<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">79932</article-id><article-id pub-id-type="doi">10.7554/eLife.79932</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Structural Biology and Molecular Biophysics</subject></subj-group></article-categories><title-group><article-title>Deep mutational scanning and machine learning reveal structural and molecular rules governing allosteric hotspots in homologous proteins</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes" id="author-280535"><name><surname>Leander</surname><given-names>Megan</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes" id="author-280536"><name><surname>Liu</surname><given-names>Zhuang</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4695-7142</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-151091"><name><surname>Cui</surname><given-names>Qiang</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6214-5211</contrib-id><email>qiangcui@bu.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes" id="author-210158"><name><surname>Raman</surname><given-names>Srivatsan</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-2461-1589</contrib-id><email>sraman4@wisc.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03ydkyb10</institution-id><institution>Department of Biochemistry, University of Wisconsin-Madison</institution></institution-wrap><addr-line><named-content content-type="city">Madison</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02866re76</institution-id><institution>Department of Physics, Boston University</institution></institution-wrap><addr-line><named-content content-type="city">Boston</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02866re76</institution-id><institution>Department of Chemistry, Boston University</institution></institution-wrap><addr-line><named-content content-type="city">Boston</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01y2jtd41</institution-id><institution>Department of Bacteriology, University of Wisconsin-Madison</institution></institution-wrap><addr-line><named-content content-type="city">Madison</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01y2jtd41</institution-id><institution>Department of Chemical and Biological Engineering, University of Wisconsin-Madison</institution></institution-wrap><addr-line><named-content content-type="city">Madison</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Faraldo-Gómez</surname><given-names>José D</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institutes of Health</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Faraldo-Gómez</surname><given-names>José D</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institutes of Health</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>13</day><month>10</month><year>2022</year></pub-date><pub-date pub-type="collection"><year>2022</year></pub-date><volume>11</volume><elocation-id>e79932</elocation-id><history><date date-type="received" iso-8601-date="2022-05-03"><day>03</day><month>05</month><year>2022</year></date><date date-type="accepted" iso-8601-date="2022-10-13"><day>13</day><month>10</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at .</event-desc><date date-type="preprint" iso-8601-date="2022-05-01"><day>01</day><month>05</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.05.01.490188"/></event></pub-history><permissions><copyright-statement>© 2022, Leander, Liu et al</copyright-statement><copyright-year>2022</copyright-year><copyright-holder>Leander, Liu et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-79932-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-79932-figures-v2.pdf"/><abstract><p>A fundamental question in protein science is where allosteric hotspots – residues critical for allosteric signaling – are located, and what properties differentiate them. We carried out deep mutational scanning (DMS) of four homologous bacterial allosteric transcription factors (aTFs) to identify hotspots and built a machine learning model with this data to glean the structural and molecular properties of allosteric hotspots. We found hotspots to be distributed protein-wide rather than being restricted to ‘pathways’ linking allosteric and active sites as is commonly assumed. Despite structural homology, the location of hotspots was not superimposable across the aTFs. However, common signatures emerged when comparing hotspots coincident with long-range interactions, suggesting that the allosteric mechanism is conserved among the homologs despite differences in molecular details. Machine learning with our large DMS datasets revealed global structural and dynamic properties to be a strong predictor of whether a residue is a hotspot than local and physicochemical properties. Furthermore, a model trained on one protein can predict hotspots in a homolog. In summary, the overall allosteric mechanism is embedded in the structural fold of the aTF family, but the finer, molecular details are sequence-specific.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>allostery</kwd><kwd>deep mutational scanning</kwd><kwd>machine learning</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>E. coli</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>DP2GM132682</award-id><principal-award-recipient><name><surname>Raman</surname><given-names>Srivatsan</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM141930</award-id><principal-award-recipient><name><surname>Cui</surname><given-names>Qiang</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32GM08293</award-id><principal-award-recipient><name><surname>Leander</surname><given-names>Megan</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution>The Camille and Henry Dreyfus Foundations, Inc</institution></institution-wrap></funding-source><award-id>ML-21-016</award-id><principal-award-recipient><name><surname>Cui</surname><given-names>Qiang</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Deep mutational scanning of homologous proteins shows conservation in allosteric mechanisms but differences in molecular details within the protein family.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Allostery is a fundamental regulatory mechanism governing proteins involved in diverse biological functions (<xref ref-type="bibr" rid="bib9">Changeux and Edelstein, 2005</xref>). It is a fascinating property of proteins where perturbation at one site of a protein elicits a response at a distant site but one whose molecular principles remain poorly understood (<xref ref-type="bibr" rid="bib81">Wodak et al., 2019</xref>). Though allosteric proteins may employ diverse structural mechanisms to propagate the perturbation, all allosteric proteins obey a simple thermodynamic principle that binding an effector ligand stabilizes the active state over the inactive state and removing the effector ligand reverses this effect (<xref ref-type="bibr" rid="bib10">Changeux, 2012</xref>; <xref ref-type="bibr" rid="bib11">Cui and Karplus, 2008</xref>; <xref ref-type="bibr" rid="bib46">Marzen et al., 2013</xref>; <xref ref-type="bibr" rid="bib27">Hilser et al., 2012</xref>). We need to investigate the molecular nature of allostery at the residue level to understand how diverse structural mechanisms are bound by the same thermodynamic principle; that is, do common underlying molecular ‘rules’ of allostery exist? To answer this question, we need to identify the allosteric ‘hotspots’ or residues critical for allosteric signaling. However, the location of allosteric hotspots cannot be gleaned from structure alone. Our recent deep mutational scanning (DMS) analysis of bacterial transcription factor, TetR, revealed that allosteric hotspots are distributed throughout the protein with no apparent direct structural link to either the allosteric or the active site (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>). In contrast, the commonly held view is that hotspot residues tend to fall along well-defined pathways linking both sites (<xref ref-type="bibr" rid="bib55">Ota and Agard, 2005</xref>; <xref ref-type="bibr" rid="bib70">Süel et al., 2003</xref>; <xref ref-type="bibr" rid="bib69">Strickland et al., 2008</xref>; <xref ref-type="bibr" rid="bib61">Reynolds et al., 2011</xref>; <xref ref-type="bibr" rid="bib3">Amor et al., 2016</xref>). In recent years, targeting allosteric rather than active site has also emerged as an attractive therapeutic strategy, especially for drug targets that implicate ubiquitous molecules, such as ATP, as the substrate (<xref ref-type="bibr" rid="bib52">Nussinov and Tsai, 2013</xref>; <xref ref-type="bibr" rid="bib1">Abdel-Magid, 2015</xref>). Therefore, from both fundamental and biomedical application points of view, it is important to develop methodologies to systematically identify allosteric hotspot residues and to understand the molecular nature of these residues.</p><p>There are no well-established methods for determining the location of allosteric hotspots in a protein. Current experimental approaches impute allosteric hotspots from residue connectivity in crystal structures (<xref ref-type="bibr" rid="bib13">del Sol et al., 2006</xref>) or changes in NMR chemical shifts or dynamics (<xref ref-type="bibr" rid="bib73">Tzeng and Kalodimos, 2009</xref>; <xref ref-type="bibr" rid="bib42">Lisi et al., 2016</xref>; <xref ref-type="bibr" rid="bib25">Guo and Zhou, 2016</xref>). These approaches at best identify only a subset of hotspots, but may also misidentify hotspots simply because they lie in-between allosteric and active sites or show local motion. Other metrics to assign the importance of a residue such as shortest path length (<xref ref-type="bibr" rid="bib74">Vanwart et al., 2012</xref>; <xref ref-type="bibr" rid="bib65">Sethi et al., 2009</xref>) or density of connections (<xref ref-type="bibr" rid="bib77">Wang et al., 2020</xref>) are only tangentially related to allostery. Computational approaches have limitations too. Sequence co-evolution patterns reveal statistically linked residue pairs (<xref ref-type="bibr" rid="bib70">Süel et al., 2003</xref>; <xref ref-type="bibr" rid="bib61">Reynolds et al., 2011</xref>), but face ambiguity regarding the origin of co-evolution, which can be driven by folding stability rather than allostery. Molecular dynamics simulations (<xref ref-type="bibr" rid="bib11">Cui and Karplus, 2008</xref>; <xref ref-type="bibr" rid="bib25">Guo and Zhou, 2016</xref>; <xref ref-type="bibr" rid="bib57">Papaleo et al., 2016</xref>) in combination with analysis such as community network analysis (<xref ref-type="bibr" rid="bib65">Sethi et al., 2009</xref>; <xref ref-type="bibr" rid="bib62">Rivalta and Batista, 2021</xref>; <xref ref-type="bibr" rid="bib50">Nierzwicki et al., 2021</xref>) or Markov state models (<xref ref-type="bibr" rid="bib37">Kuzmanic et al., 2020</xref>) has been used to identify allosteric hotspots or cryptic allosteric sites, but these approaches are often not comprehensive. In recent years, DMS has emerged as a powerful tool to understand protein function by measuring the impact of mutational perturbations using high-throughput experiments (<xref ref-type="bibr" rid="bib18">Fowler et al., 2010</xref>; <xref ref-type="bibr" rid="bib19">Fowler and Fields, 2014</xref>; <xref ref-type="bibr" rid="bib64">Sarkisyan et al., 2016</xref>; <xref ref-type="bibr" rid="bib17">Flynn et al., 2020</xref>; <xref ref-type="bibr" rid="bib68">Starr et al., 2020</xref>; <xref ref-type="bibr" rid="bib28">Huss et al., 2021</xref>). DMS is particularly useful to study a systemic property like allostery because it permits an unbiased examination of every residue of a protein without a priori assumptions about its functional role (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>; <xref ref-type="bibr" rid="bib30">Jones et al., 2019</xref>; <xref ref-type="bibr" rid="bib71">Tack et al., 2020</xref>; <xref ref-type="bibr" rid="bib16">Faure et al., 2022</xref>; <xref ref-type="bibr" rid="bib47">McCormick et al., 2021</xref>). Combining DMS with statistical tools allows us to recognize complex underlying patterns describing the molecular rules of allostery.</p><p>In this study, we used DMS to identify allosteric hotspots by systematically dissecting the functional contribution of each residue to allosteric signaling. Using this approach, we compared hotspots across four distant homologs in the TetR-like family of allosteric transcription factors (aTFs). We found that the location of allosteric hotspots is unique to each homolog despite similarities in allosteric signaling within this family. However, a common pattern emerges when comparing hotspots clustered around residues participating in long-range interactions (LRIs) suggesting that non-bonded, LRIs play a defining role in the transmission of signal between allosteric and active sites. We leveraged the DMS data to train a machine learning model (a feedforward neural network [NN] integrated with a GA for feature selection) using a broad set of local and global properties of the hotspots for classifying whether a residue is a hotspot. By analyzing the performance of different NN models and identifying features that dictate the accuracy of classification, we gained insights into factors that are likely essential to allostery. Finally, we explore the transferability of the NN model among homologous proteins. This helps elucidate to what degree the mechanism of allostery is conserved among proteins in the same family and the information content of models required to make a meaningful prediction of allosteric hotspots. Our study lays the foundation for combining DMS and machine learning to infer molecular mechanism of allostery within a protein family.</p></sec><sec id="s2" sec-type="results|discussion"><title>Results and discussion</title><sec id="s2-1"><title>Identifying allosteric hotspots across homologs</title><p>aTF is an ideal model system because of its simple one-component signal transduction mechanism that can be converted into a reporter-based high-throughput screen to measure allosteric activity (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>; <xref ref-type="bibr" rid="bib51">Nishikawa et al., 2021</xref>). We chose to study the TetR family of transcription regulators because they are a large and remarkably diverse family of proteins found in almost every bacterial host with diverse ligand and DNA specificities (<xref ref-type="bibr" rid="bib12">Cuthbertson and Nodwell, 2013</xref>). As a result, the allosteric mechanisms of these proteins have evolved under different selection pressures exerted by their environments. Despite their diversity, all TetR family proteins (&gt;100 in PDB) share a similar protein structure which suggests their structure is versatile and robust to preserve allostery while accommodating diverse sequences (<xref ref-type="bibr" rid="bib12">Cuthbertson and Nodwell, 2013</xref>; <xref ref-type="bibr" rid="bib21">Fukami-Kobayashi et al., 2003</xref>). Therefore, the TetR family serves as a good model system to investigate structural properties of allostery common within the family while minimizing sequence-dependent effects. We chose four aTFs – TetR, TtgR, MphR, and RolR – with high structural similarity (between 1 and 3 Å Cα root mean squared distance [RMSD]) but low sequence identity (between 14% and 19%, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>).</p><p>To probe the functional impact of mutations on aTFs, we developed a high-throughput pooled screen in <italic>Escherichia coli</italic> where the activity of mutants can be measured by the expression level of GFP regulated by an aTF-regulated promoter. Allosteric activity was quantified as the fold induction ratio of GFP expression with and without the inducer. Fold induction of wild-type aTFs was TetR: 49-fold (ligand: anhydrotetracycline [aTC]), TtgR: 25-fold (ligand: naringenin [Nar]), MphR: 100-fold (ligand: erythromycin [Ery]), and RolR: 15-fold (ligand: resorcinol [Res]). We mutated each aTF using commercially available chip oligonucleotides to encode a comprehensive library of point mutants by single-site saturation mutagenesis of each residue (~200 residues per aTF × 19 mutants/residue = 3800 mutants per aTF, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). We designate aTF mutations that constitutively lock the protein in an inactive allosteric state as ‘dead variants’. This may occur because the mutation stabilizes the inactive state by increasing the thermodynamic gap between inactive and active states. The dead variants are well-folded proteins that bind to DNA and repress transcription but cannot be induced with the ligand. From each aTF library, we enriched dead variants by sorting low GFP cells after incubation with their corresponding ligand (<xref ref-type="fig" rid="fig1">Figure 1A</xref>; <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). The sorted populations were deep sequenced in triplicate to identify the allosterically dead variants (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>; <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). In other words, we use cell sorting as a binary classifier; that is, does the mutation disrupt allostery or not. We capture the effect size on individual residues, not individual mutations, by counting the number of dead mutations at a residue position. This is an important consideration because it safeguards us from minor inconsistencies that inevitably arise from cell sorting.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Allosteric hotspots in four bacterial allosteric transcription factors (aTFs) identified using deep mutational scanning (DMS).</title><p>(<bold>A</bold>) Nonfluorescent cells in the TetR, TtgR, MphR, and RolR single-mutant library were sorted (gray bar) in the presence (light shade) and absence (dark shade) of 1 µM anhydrotetracycline (aTC), 500 µM naringenin (Nar), 1 mM erythromycin (Ery), and 7.5 mM resorcinol (Res), respectively, and sequenced to identify dead variants. Sorting gates were defined by the wild-type uninduced population for each homolog. (<bold>B</bold>) Allosteric hotspots (red points) for each aTF is shown with residue numbers along x axis and a weighted score along y axis based on the number of dead mutations at a residue position. Secondary structures of the aTFs are illustrated below and colored according to regions (blue: DBD, orange: hinge helix connecting LBD and DBD, gray: LBD and purple: dimer interface). Residue conservation is shown and colored by conserved residues (green), not conserved (gray) and conserved overlapping with hotspot (red). (<bold>C</bold>) Allosteric hotspots mapped on to the structure of TetR, TtgR, MphR, and RolR (ligand-contacting residues excluded).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Experimental scheme for deep mutational scanning.</title><p>Protein-wide, single-site saturation mutagenesis of four TetR-like family allosteric transcription factors (aTFs) – TetR, TtgR, RolR, and MphR – using reporter-based screening followed by deep sequencing.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp1-v2.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>A detailed summary of all single-mutant phenotypes for every position within the proteins.</title><p>Heatmaps detailing the effect of all single mutants at every position in (<bold>A</bold>) TetR, (<bold>B</bold>) TtgR, (<bold>C</bold>) MphR, and (<bold>D</bold>) RolR are shown. Wild-type residues are black, mutations that do not affect protein function are white, mutants classified as dead in two or all three replicates are orange and blue, respectively. Variants not present in the dataset are gray.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp2-v2.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Histograms of weighted scores and thresholds for identifying hotspots.</title><p>The distribution of weighted scores for every position in (<bold>A</bold>) TetR, (<bold>B</bold>) TtgR, (<bold>C</bold>) MphR, and (<bold>D</bold>) RolR is shown. Box and whisker plots above each histogram illustrate the spread of the data where outliers are shown as circles (red line) and all positions above Q3 (orange line) were designated allosteric hotspots.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp3-v2.tif"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Correlation of weighted scores between a ×5 or ×10 read count threshold.</title><p>The correlation of weighted scores for every position using a ×5 or ×10 read count threshold is shown for (<bold>A</bold>) TetR, (<bold>B</bold>) TtgR, (<bold>C</bold>) MphR, and (<bold>D</bold>) RolR. The red and orange lines illustrate the spread of the data using interquartile range where outliers are plotted above the red line and all positions in the top quartile are designated allosteric hotspots.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp4-v2.tif"/></fig><fig id="fig1s5" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 5.</label><caption><title>Distribution of allosteric hotspots in TetR homologs.</title><p>The percent of hotspots in the four main structural regions of the TetR homologs. Regions were broken into groups based on the crystal structures of TetR (PDB ID: 4AC0), TtgR (PDB ID: 2UXU), MphR (PDB ID: 3FRQ), and RolR (PDB ID: 3AQT). Potential ligand-binding residues are included in the statistics but are not considered hotspots.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp5-v2.tif"/></fig><fig id="fig1s6" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 6.</label><caption><title>Conservation of allosteric hotspots.</title><p>Average conservation score of all positions considered inactive or having no effect in (<bold>A</bold>) TetR, (<bold>B</bold>) TtgR, (<bold>C</bold>) MphR, and (<bold>D</bold>) RolR. Data show as mean ± SEM.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp6-v2.tif"/></fig><fig id="fig1s7" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 7.</label><caption><title>Comparison of experimental hotspots with predictions made by the Ohm server.</title><p>Allosteric hotspots of TetR, TtgR, MphR, and RolR determined by experiments (<bold>A, C, E, G</bold>) and the Ohm webserver (<bold>B, D, F, H</bold>) differ significantly. The Ohm webserver identifies critical residues along the signal propagation pathways between the bound effectors (ligands) to the active sites (DNA-binding residues) as allosteric hotspots. Specifically, in the Ohm calculation, signals are started from effector molecules, which are then propagated through residue contacts, with the active sites as signal sinks. Such signal propagation simulation is repeated for 10<sup>4</sup> times, and the residues that appear most frequently in pathways connecting effectors and active sites are recognized as hotspots. The number of hotspots identified by Ohm calculation is made equal to experimentally determined hotspots. The accuracy of hotspot identification of Ohm calculation is 0.08 for TetR; 0.12 for TtgR; 0.31 for MphR, and 0.40 for RolR.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig1-figsupp7-v2.tif"/></fig></fig-group><p>Next, we wanted to establish criteria to designate a residue as an allosteric hotspot. The importance of a residue for allosteric signaling is proportional to the number of dead variants at that position. For example, a residue with 15 dead variants is more important than one with five dead variants. We cannot choose an arbitrary threshold for the number of dead variants as this threshold may change for each aTF. Therefore, we created a simple scoring system where each residue was given a score based on the number of dead variants at that position and the confidence a variant is fully dead. The latter criterion captures variants that show weak allosteric activity. A higher positional score indicates the higher importance of a residue in allosteric signaling. We designated residues falling in the highest quartile (top 25% scoring residues) in the interquartile distribution of scores as allosteric hotspots for each aTF. The spread of residue scores varied between aTFs. The highest quartile was well separated for TetR, TtgR, and MphR, and less so for RolR, giving us higher confidence in the assignment of hotspots in the former groups (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). We note that the lower fold induction (dynamic range) of RolR makes it particularly challenging to separate the dead variants from the rest. We designated 53, 51, 48, and 57 residues as hotspots in TetR, TtgR, MphR, and RolR, respectively. To assess the robustness of our classification of hotspots, we determined the number of hotspots at two different sequencing thresholds – ×5 and ×10. At ×5 and ×10, the number of hotspots is – TetR: 53, 51; TtgR: 51, 51; MphR: 48, 48, and RolR: 57, 60, respectively (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>). Changing the threshold has a modest impact on the overall number of hotspots and the regions of functional importance are consistent at both thresholds. After excluding ligand-contacting residues from consideration, as mutations at these residues appeared dead likely due to loss of ligand affinity, we were left with 41, 43, 29, and 51 hotspots in TetR, TtgR, MphR, and RolR, respectively. We note that changing the read threshold does not change the identity of the hotspots falling in the top quartile indicating the robustness of our conclusions.</p><p>The location of hotspots on the structure was unique to each aTF despite their structural homology. Hotspots of TetR and MphR were concentrated in the C-terminal half of the protein (<xref ref-type="fig" rid="fig1">Figure 1B</xref>), whereas those of TtgR and RolR were distributed across the entire structure. This result exposes a key difference between understanding the allosteric mechanism at the level of protein structure vs. individual residues. The structural mechanism of allostery may be specified once a protein fold is specified. However, the residues involved in ‘executing’ the structural mechanism may be unique to different sequences folding into that structure. In other words, nature has created degenerate molecular pathways to transmit the allosteric signal within the same protein structure.</p><p>To understand the common structural mechanism in the family, we classified the hotspots based on their secondary structure location (<xref ref-type="fig" rid="fig1">Figure 1B</xref>) into α helices 1 through 9 (or 10): DNA-binding domain (DBD = α1, -2, -3), the ligand-binding domain (LBD = α5, -6, -7), the alpha helix connecting DBD and LBD (α4), and the dimer interface (α8, -9 for TetR, TtgR and MphR and α8, -9, -10 for RolR). Three structural similarities emerged in the location of hotspots across aTFs. First, a high fraction of hotspots, relative to the segment’s length, were at the dimer interface (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>). This suggests that allosteric signaling through the dimer interface is likely a conserved mechanism in TetR family. Second, a high fraction of hotspots was on α4 suggesting α4 acts as a mechanical link that transmits allosteric signals from the LBD to the DBD. Third, very few hotspots were in the DBD compared to other regions. This suggests the DBD is a standalone domain whose interaction with DNA is controlled by allosteric forces originating outside the DBD. An evolutionary perspective strengthens this hypothesis. The evolution of aTFs has occurred through a series of gene duplication events resulting in mixing and matching LBDs and DBDs (<xref ref-type="bibr" rid="bib60">Pougach et al., 2014</xref>; <xref ref-type="bibr" rid="bib83">Yuan et al., 2022</xref>). Thus, the DBDs likely exist as standalone domains that respond to large thermodynamic changes (e.g., inducer binding). Taken together, these observations show that although the hotspots are not superimposable across aTFs, the TetR family likely shares a conserved structural mechanism where the allosteric signal travels from the LBD through the dimer interface and α4 to the DBD, while the DBD itself acts as an internally rigid module that docks on DNA. Detailed biophysical or molecular dynamics characterization will be required to further validate our conclusions (<xref ref-type="bibr" rid="bib22">Gandhi et al., 2008</xref>).</p><p>Since residues important for function (e.g., binding, catalysis, etc.) tend to be conserved in sequence, we assessed if allosteric hotspots too are conserved. We compared hotspots to close sequence homologs (&gt;50% sequence identity) and did not find statistically higher sequence conservation in hotspots over non-hotspots (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>). This reinforces our earlier conclusion that though the structural mechanism of allosteric signaling may be conserved within this family, the residues participating in signal transduction may be specific for each aTF. In other words, allosteric sites are not necessarily conserved, though allostery itself may be conserved. We also compared the experimental hotspots with predictions made by the Ohm webserver (<xref ref-type="bibr" rid="bib77">Wang et al., 2020</xref>). The Ohm webserver is an efficient computational tool that analyzes the propagation of structural perturbation in proteins to identify allostery network and hotspot residues. The overlap between predictions and experiments is modest and involves mostly DBD residues while the experimental hotspots are distributed across the protein (<xref ref-type="fig" rid="fig1s7">Figure 1—figure supplement 7</xref>). This highlights the limitation of focusing on the mechanistic model that involves propagation of conformational distortions.</p></sec><sec id="s2-2"><title>LRIs reveal similarities in allosteric mechanism</title><p>We investigated what underlying property of protein structure might explain the preference of hotspots for certain sites. We considered the defining characteristic of allostery, that is, cooperative action between spatially distant residues. Cooperative action occurs through molecular forces transmitted between bonded and non-bonded interactions. Forces transmitted through bonded interactions tend to dissipate over short distances. However, non-bonded interactions, particularly between residues farther in primary sequence, likely facilitate transmission of force over longer distances (<xref ref-type="bibr" rid="bib48">Miyazawa and Jernigan, 1996</xref>). Therefore, we examined the location of allosteric hotspots with respect to residues involved in non-bonded LRI.</p><p>We generated contact maps of residue-residue interactions and selected LRIs as residues separated by 10 or more positions in sequence but within 8 Å in Cα- Cα distance. For each aTF, we then compared LRI residues and allosteric hotspots. In all four aTFs, hotspots constituted a higher fraction of LRIs than non-hotspots (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>; p=0.07). Hotspots in TetR and MphR are especially enriched in LRIs – 28% hotspots vs. 17% non-hotspots. This gap is smaller in TtgR and RolR – 23% hotspots and 20% non-hotspots. It is worth noting that in addition to allostery, LRIs play an important role in protein folding and stability. Thus, it is not surprising that only a fraction of LRIs overlaps with allosteric hotspots. But this subset of LRIs may offer insight into the mechanism of allostery in the TetR family. Therefore, we grouped the LRIs using standard, unsupervised k-means clustering on the contact maps. The optimum number of clusters for each aTF was determined using a standard Elbow method that iteratively calculates the variance within clusters for different numbers of clusters. The optimal number of clusters is the point yielding diminishing returns (higher variance within a cluster) that is not worth the cost of adding new clusters and was found to be 10 clusters independently for each aTF (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>).</p><p>The 10 clusters of LRIs represented distinct local regions of the protein (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). To evaluate the relative importance of different signaling pathways, we ranked the LRI clusters based on the fraction of unique hotspots out of all residues in that cluster (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). The dimer interface emerged as the first-ranked cluster in TetR, MphR, and RolR (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), suggesting that the dimer interface is a dominant signaling pathway. The first-ranked cluster of TtgR was not the dimer interface which is consistent with far fewer hotspots found at the dimer interface of TtgR (<xref ref-type="fig" rid="fig2">Figures 2</xref> and <xref ref-type="fig" rid="fig1">1C</xref>). The second tier of cluster rankings contained regions between helices α4, -5, -6, and -7 of the LBD and the linker helix between LBD and DBD. These clusters likely represent allosteric forces emanating from the LBD upon ligand binding. No cluster stands out as dominant within this tier. LRI clusters within the DBD were ranked near or at the bottom of the rankings for all homologs. LRIs have long been known to play a key role in protein folding and stability. Our results show that LRIs are also critical for the propagation of allosteric signals. These results also show that though allosteric hotspots may not be superimposable across distant homologs, local clusters of LRIs share similar patterns between homologs. As homologs get closer in sequence, regional similarities in allosteric signaling may give way to the superimposability of individual hotspots.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Hotspots enriched among long-range interactions (LRIs).</title><p>(<bold>A</bold>) Residue-residue contact map showing LRIs within each homolog. The LRIs are grouped by color, following standard k-means clustering, representing different regions of the protein. Inset shows ranking of LRI clusters based on the percentage of unique hotspots within each cluster. (<bold>B</bold>) The general location of each LRI cluster on the protein structure (color scheme same as panel A).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Hotspot interactions are more likely to be long range than those of non-hotspots.</title><p>The percent of hotspot and non-hotspot residues participating in long-range interactions (LRIs) in each homolog protein.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Elbow method to determine the optimal number of clusters.</title><p>The optimal number of clusters to use for the k-means clustering of long-range interactions (LRIs) in each homolog was determined by iteratively calculating the variance within clusters for 1–25 clusters.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig2-figsupp2-v2.tif"/></fig></fig-group></sec><sec id="s2-3"><title>Physiochemical properties of dead mutations</title><p>We investigated if mutations to certain amino acids were enriched among dead variants over non-dead variants. We computed the percentage of each amino acid among mutations that were dead vs. not-dead from all four aTF datasets (~12,000 mutations). Aromatic amino acids (Phe, Trp, and Tyr) were enriched among dead variants (20%) over the not-dead group (15%) (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Mutations to proline were also enriched, albeit to a lesser degree, among the dead variants (5%) vs. the not-dead group (3.8%). These trends change slightly at the protein level, for example, the branched nonpolar leucine is the most enriched mutation in dead variants of TetR (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Mutational preferences and physicochemical properties of dead variants.</title><p>(<bold>A</bold>) Percentage of mutations (final mutated state) among dead (red) and not-dead (gray) variants from deep mutational scanning (DMS) data for all four homologs combined. (<bold>B</bold>) Comparison of physicochemical properties – polarizability, solvent-accessible surface area (SASA), mass, hydrophilicity, hydrophobicity, and polarity – between dead (red) and not-dead variants (gray). Average values aggregated over all four DMS datasets shown. Data represented as mean ± SEM. (<bold>C</bold>) Structural models of the K199Y (top) and C144F (bottom) mutations in TetR. Residues in the mutant structures are colored teal and the two monomers are colored white and dark gray.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Enrichment of mutations in allosterically dead or no effect variants.</title><p>Mutations in (<bold>A</bold>) TetR, (<bold>B</bold>) TtgR, (<bold>C</bold>) MphR, and (<bold>D</bold>) RolR were separated based on their effect on protein function, dead (red) or no effect (gray), and the proportion of each of the 20 amino acids within each set calculated to identify enrichments in allosterically dead or neutral variants.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig3-figsupp1-v2.tif"/></fig></fig-group><p>Next, we compared differences between both groups in six common physicochemical properties of amino acids. We did not observe any statistically significant differences in hydrophilicity, hydrophobicity, and polarity between both groups (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). However, we observed statistically significant differences in polarizability, solvent-accessible surface area (SASA), and mass (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). This is consistent with aromatic residues having larger mass and SASA, and greater polarizability due to the π-electron cloud. Although aromatic residues are also hydrophobic, hydrophobicity itself is not a differentiator between dead vs. not-dead groups. The enrichment of aromatic amino acids, and to a lesser extent strongly aliphatic amino acids (Ile and Leu) among the dead variants, hints at a relationship between residue-residue interaction energy and allosteric signaling. These amino acids (Trp, Phe, Tyr, Ile, and Leu) have the highest interaction energies among all amino acids in the PDB of (−4 to –6 RT units) (<xref ref-type="bibr" rid="bib8">Chan et al., 2004</xref>). At the other end, mutations to small branched amino acids (Ser, Cys, and Ala), which have low interaction energies, were most depleted among dead variants (12%) vs. the not-dead group (18%), suggesting that substitutions to small branched amino acids are least likely to inactivate allosteric signaling.</p><p>To understand at an atomic level why aromatic mutations are consistently enriched for in dead variants, we examined modeled phenylalanine and tyrosine inactivating mutations in TetR. These mutations were chosen as they were consistently among the most prevalent mutations in dead variants. In the inactive C144F TetR variant, F144 formed strong aromatic interactions at the dimer interface with its neighbor F140 and H139 in the other monomer, potentially stabilizing the inactive state of the protein through increased dimerization interactions (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). Similarly, the inactive Y199 mutation may stabilize the inactive state by creating increased interactions and surface area between the monomers.</p><p>We concluded that the interaction energy of the large hydrophobic sidechains provides an enthalpic gain that stabilizes the allosteric OFF state of the protein. The resulting increase in energy gap, relative to wild-type, makes the variant unresponsive to ligand-induced allosteric activation. Thus, evolution of allosteric proteins is constrained to sequence variations that maintain an appropriate energy gap between ON and OFF states. Our results suggest that a few RT units can tip this delicate balance toward the inactive OFF state.</p></sec><sec id="s2-4"><title>Discriminative features of allosteric hotspots vary among homologous proteins</title><p>While the above analyses revealed several interesting features of hotspot residues, the partial overlap of these features between hotspot and non-hotspot residues suggests that additional features are required to make reliable predictions of allosteric hotspots. Prior to establishing such a predictive model, it is important to first understand what features are most likely to differentiate hotspot from non-hotspot residues. Accordingly, we assembled a comprehensive list of 27 features that are potentially relevant to the classification of a protein site as an allosteric hotspot (see Materials and methods) (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). These include eight intrinsic physicochemical properties of amino acids such as charge and hydrophobicity, and eight local structural properties, such as solvent accessibility, local structural entropy (LSE) (<xref ref-type="bibr" rid="bib29">Jenik et al., 2012</xref>), and frustration index (<xref ref-type="bibr" rid="bib7">Chakrabarty and Parekh, 2016</xref>). Since allostery is fundamentally about cooperativity between distant sites in a protein, we also included 11 global features that describe long-range structural or dynamical properties; for example, residue centrality (<xref ref-type="bibr" rid="bib4">Bahar and Rader, 2005</xref>), which measures the degree of connectedness of a residue when the protein is represented as a graph; a residue’s distance to important regions in the protein, such as the DNA/ligand-binding sites and local peaks of residual centralities (see Materials and methods); motional covariance between a residue and the ligand/DNA-binding residues evaluated using an elastic network model (<xref ref-type="bibr" rid="bib82">Xia et al., 2010</xref>).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Machine learning identifies structural and molecular features that differentiate allosteric hotspots.</title><p>(<bold>A</bold>) The full list of 27 features is shown at the top. The F scores (measure of importance) of the features for each of the four allosteric transcription factors (aTFs) is shown below. (<bold>B</bold>) Frequency of appearance of the 27 features in the top ten 1–10 feature combinations ranked by F1 score for each protein. Row 2–28 corresponds to feature 1–27, row 1 is the average F1 score of the top ten 1–10 feature combinations. (<bold>C</bold>) Predictions made by the model based on the best fivefold cross-validation performance achieved for each aTF (red: true positive; cyan: false positive; black: false negative; rest: true negative). The features used in the best models are 2, 19, 21, 23–26 for TetR; 4, 7, 10, 15, 19, 21, 24, 25 for MphR; 2, 10, 12, 21, 25 for RolR, and 9, 10, 13, 23, 25 for TtgR.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Global features have the highest Jensen-Shannon divergence (JSD).</title><p>The full list of 27 features is shown at the top. The JSDs (measure of importance) of the features for each of the four allosteric transcription factors (aTFs) is shown below. JSD is a measure of similarity between two probability distributions P and Q, which is bound between 0 (P and Q are the same) and 1 (P and Q have no overlap). The larger the JSD, the more different the two distributions are, and thus the features with larger JSDs are more discriminative for hotspot residues. JSD is a symmetrized and smoothed version of the more familiar Kullback-Liebler divergence defined as JSD(P||Q) = { D<sub>KL</sub>(P||M)+D<sub>KL</sub>(Q||M) }/2, where M = (P+Q)/2 is the average of two distributions and D<sub>KL</sub> is the Kullback-Liebler divergence (KL divergence) which also measures similarity between two distributions. KL divergence is defined as D<sub>KL</sub>(P||M) = ∑<sub>x</sub>P(x)*log<sub>2</sub>[P(x)/M(x)], x are points of the probability space where discrete distributions P and M are defined.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Distributions of TetR’s hotspots’ and non-hotspots’ z-scored feature values for feature 1–27.</title><p>The 27 plots correspond to the distributions of TetR’s hotspots’ (hs) and non-hotspots’ (non-hs) z-scored feature values for feature 1–27 as labeled by figure titles. The distributions of hotspots and non-hotspots are normalized by their populations, thus the y axis of the figures are probabilities. Z-scored feature j value of a residue n (Z<sub>nj</sub>) is defined as the difference between its raw feature j value (R<sub>nj</sub>) and the average raw feature j values of all residues (avg_R<sub>i</sub>), divided by the standard deviation of raw feature j values of all residues, Z<sub>nj</sub> = (R<sub>nj</sub> - avg_R<sub>j</sub>)/std_R<sub>j</sub>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp2-v2.tif"/></fig><fig id="fig4s3" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 3.</label><caption><title>Distributions of MphR’s hotspots’ and non-hotspots’ z-scored feature values for feature 1–27.</title><p>The 27 plots correspond to the distributions of MphR’s hotspots’ (hs) and non-hotspots’ (non-hs) z-scored feature values for feature 1–27 as labeled by figure titles. The distributions of hotspots and non-hotspots are normalized by their populations, thus the y axis of the figures are probabilities. Z-scored feature j value of a residue n (Z<sub>nj</sub>) is defined as the difference between its raw feature j value (R<sub>nj</sub>) and the average raw feature j values of all residues (avg_R<sub>i</sub>), divided by the standard deviation of raw feature j values of all residues, Z<sub>nj</sub> = (R<sub>nj</sub> − avg_R<sub>j</sub>)/std_R<sub>j</sub>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp3-v2.tif"/></fig><fig id="fig4s4" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 4.</label><caption><title>Distributions of TtgR’s hotspots’ and non-hotspots’ z-scored feature values for feature 1–27.</title><p>The 27 plots correspond to the distributions of TtgR’s hotspots’ (hs) and non-hotspots’ (non-hs) z-scored feature values for feature 1–27 as labeled by figure titles. The distributions of hotspots and non-hotspots are normalized by their populations, thus the y axis of the figures are probabilities. Z-scored feature j value of a residue n (Z<sub>nj</sub>) is defined as the difference between its raw feature j value (R<sub>nj</sub>) and the average raw feature j values of all residues (avg_R<sub>i</sub>), divided by the standard deviation of raw feature j values of all residues, Z<sub>nj</sub> = (R<sub>nj</sub> − avg_R<sub>j</sub>)/std_R<sub>j</sub>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp4-v2.tif"/></fig><fig id="fig4s5" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 5.</label><caption><title>Distributions of RolR’s hotspots’ and non-hotspots’ z-scored feature values for feature 1–27.</title><p>The 27 plots correspond to the distributions of RolR’s hotspots’ (hs) and non-hotspots’ (non-hs) z-scored feature values for feature 1–27 as labeled by figure titles. The distributions of hotspots and non-hotspots are normalized by their populations, thus the y axis of the figures are probabilities. Z-scored feature j value of a residue n (Z<sub>nj</sub>) is defined as the difference between its raw feature j value (R<sub>nj</sub>) and the average raw feature j values of all residues (avg_R<sub>i</sub>), divided by the standard deviation of raw feature j values of all residues, Z<sub>nj</sub> = (R<sub>nj</sub> − avg_R<sub>j</sub>)/std_R<sub>j</sub>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp5-v2.tif"/></fig><fig id="fig4s6" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 6.</label><caption><title>Average and best F1 scores of 4–10 feature combinations converge after 10 generations in the genetic algorithm feature selection.</title><p>The plots show the average and best F1 scores for 4–10 feature combinations as a function of generation in the genetic algorithm feature selection for the four homologous allosteric transcription factors (aTFs) as labeled by figure titles.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp6-v2.tif"/></fig><fig id="fig4s7" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 7.</label><caption><title>Machine learning identifies structural and molecular features that differentiate allosteric hotspots.</title><p>Frequency of appearance of the 27 features in the top ten 1–10 feature combinations ranked by F1 score for each protein (labeled on top). Row 2–28 corresponds to feature 1–27, row 1 is the average F1 score of the top ten 1–10 feature combinations. This is the same data as that of <xref ref-type="fig" rid="fig4">Figure 4B</xref> with all the frequencies specified in the heatmap.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp7-v2.tif"/></fig><fig id="fig4s8" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 8.</label><caption><title>Positions of centrality peaks.</title><p>Plots of centrality against residue number of each protein (labeled by the title), with the four red stars label the positions of centrality peaks 1–4 from left to right. The centrality peaks are identified as positions of highest centrality within local sequence while maintaining distances between centrality peaks as large as possible. The centrality peaks 1–4 are located at residue 22, 83, 150, 193 for TetR; residue 26, 59, 104, 155 for MphR; residue 30, 71, 122, 186 for TtgR; and residue 48, 84, 133, 191 for RolR.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig4-figsupp8-v2.tif"/></fig></fig-group><p>To quantify how well these features differentiated hotspots and non-hotspots, we computed two metrics commonly used to estimate feature importance in machine learning – the F score and the Jensen-Shannon divergence (JSD). The F score measures the difference between the average (mean) of the two distributions relative to the widths of these distributions; the F score of feature i is computed as (<xref ref-type="bibr" rid="bib56">Pan et al., 2018</xref>),<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">F</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext> </mml:mtext></mml:mrow><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>in which <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> / <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the average values of hotspots/non-hotspots, and <inline-formula><mml:math id="inf3"><mml:msub><mml:mrow><mml:mi mathvariant="normal">σ</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>/</mml:mo><mml:mi mathvariant="normal">σ</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the corresponding standard deviations. A large F score indicates that a specific feature adopts significantly different values for hotspot and non-hotspot residues. Features with larger F scores better differentiate hotspots from non-hotspots than those with smaller F scores. For all four aTFs, global features have the highest F scores and physicochemical features tend to have the lowest F scores. The confidence of hotspot assignment increases with increasing dynamic range because there is a clearer separation of dead vs. not-dead. Since RolR and TtgR have lower dynamic ranges, this may be a factor in their lower F scores.</p><p>Since the distributions of the various features are not necessarily mono-modal, we also evaluated feature importance using JSD, which is different from F score in that it measures the overall similarity between two distributions of arbitrary shape rather than only the difference between the averages. A similar trend in the rankings of features was observed with JSD (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplements 1</xref>–<xref ref-type="fig" rid="fig4s5">5</xref>). These striking differences among the three classes of features clearly show that a residue’s relative location in the protein structure and its motional correlation with other residues are more indicative of its role in allostery, as compared to its molecular features and local environment. In other words, once the protein fold is specified, the contribution of structure is larger than the contribution of sequence in determining the importance of a residue in allostery. On the other hand, substantial variations in the patterns of the F score and JSD among the four proteins highlight that the distinguishing properties of hotspot residues may differ even among homologous proteins, suggesting non-trivial variations in the detailed mechanism among them.</p></sec><sec id="s2-5"><title>NN analysis further highlights convergence and divergence in allostery mechanisms among homologous proteins</title><p>Having established the importance of individual features, we set out to build models by combining these features to reliably classify whether a protein site is an allosteric hotspot because combinations of features tend to perform better than models based on a single feature (<xref ref-type="bibr" rid="bib76">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="bib53">Ofran and Rost, 2007</xref>; <xref ref-type="bibr" rid="bib14">Demerdash et al., 2009</xref>; <xref ref-type="bibr" rid="bib58">Pethe et al., 2019</xref>; <xref ref-type="bibr" rid="bib24">Gelman et al., 2021</xref>; <xref ref-type="bibr" rid="bib66">So and Karplus, 1996a</xref>). We included even the lower-ranked features (<xref ref-type="fig" rid="fig4">Figure 4A</xref>) as they may contribute to the discriminative power of a model when used in combination with other features in NNs. To search for the best feature combinations, we coupled the NNs with a genetic algorithm (GA), which has been shown to be efficient at picking out desired feature combinations when the total number of possible combinations is too large for an exhaustive search (<xref ref-type="bibr" rid="bib67">So and Karplus, 1996b</xref>; <xref ref-type="bibr" rid="bib26">Halabi et al., 2009</xref>). Specifically, we implemented the evolutionary programming algorithm to search for the best 1–10 feature combinations for NNs for each aTF. The algorithm guarantees that the average and highest fitness of the gene pool increases monotonically with evolutionary time (number of generations iterated), or remains constant upon convergence (see Materials and methods for details); these properties are essential for the convergence and proper termination of the feature optimization process. The fitness during GA optimization is evaluated as the average F1 score (distinct from the F score) of five times of fivefold cross-validation tests. The F1 score is defined in <xref ref-type="disp-formula" rid="equ2 equ3 equ4">Equations 2–4</xref>, where TP, FP, TN, and FN represent true positive rate, false positive rate, true negative rate, and false negative rate, respectively.<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">F</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">N</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><disp-formula id="equ3"><label>(3)</label><mml:math id="m3"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">s</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">F</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><disp-formula id="equ4"><label>(4)</label><mml:math id="m4"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">F</mml:mi></mml:mrow><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mrow><mml:mo>∗</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mo>∗</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">s</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">s</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>Both the average and best fitness scores converge after 10 generations for 4–10 feature combinations (<xref ref-type="fig" rid="fig4s6">Figure 4—figure supplement 6</xref>). As the total numbers of all 1–3 feature combinations are moderate, they are evaluated exhaustively without using the genetic algorithm.</p><p>The F1 scores of the best model and that of a random model for each protein are: 0.83 and 0.19 for TetR, 0.82 and 0.16 for MphR, 0.64 and 0.26 for RolR, and 0.54 and 0.21 for TtgR; the F1 score of a random model is given by the fraction of residues being identified as allosteric hotspots in the DMS experiments. The prediction results of the best models are further visualized by mapping the best fivefold cross-validation performance onto the crystal structures of the four proteins (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). The results illustrate that all best models significantly outperform their corresponding random models, demonstrating the effectiveness of combining subsets of the 27 features in describing hotspots. However, the performances of the models are not uniform across the four proteins, with RolR and TtgR exhibiting significant numbers of false positive hotspots predicted by the GA-NN models (blue regions in <xref ref-type="fig" rid="fig4">Figure 4C</xref>). This is consistent with the above observation that the same set of features may have different F scores and JSDs for different proteins, which suggests that the four proteins, despite being homologous, likely feature different detailed allosteric mechanisms.</p><p>Next, to identify the key features for differentiating hotspots from non-hotspots, we examined the frequency of appearance of all features in the top 10 NN models (ranked by F1 score) using 1–10 features in each case (<xref ref-type="fig" rid="fig4">Figure 4B</xref> and <xref ref-type="fig" rid="fig4s7">Figure 4—figure supplement 7</xref>). For example, models containing four features for TetR are dominated by features 25, 23, 19, and 17 (see vertically at number of features = 4 for TetR, <xref ref-type="fig" rid="fig4">Figure 4B</xref>). The average F1 score of the top 10 models using n features reach its maximum at 0.81 when n=7 for TetR, 0.8 when n=8 for MphR, 0.63 when n=5 for RolR, and 0.52 when n=5 for TtgR (see top row of <xref ref-type="fig" rid="fig4">Figure 4B</xref> and <xref ref-type="fig" rid="fig4s7">Figure 4—figure supplement 7</xref>). These are optimal models as further increasing the number of features in the NN likely incurs the problem of overfitting. Examining the vertical trends (<xref ref-type="fig" rid="fig4">Figure 4B</xref>), we note that for a given number (n) of features, the top features do not always have overwhelmingly higher frequencies of appearance, suggesting a considerable degree of redundancy in the features that are able to differentiate hotspot residues. Nevertheless, the five most frequent features in the optimal models of the four proteins consist of 71.4% global features, 14.8% local features, and 14.8% physicochemical properties of wild-type amino acids, further supporting the notion that global features are more indicative of the role of a residue in allostery. For example, distances to centrality peaks are among the most frequent global features for all four aTFs; this observation is consistent with the expectation that a residue close to highly connected regions in a protein has a high chance of being a hotspot, since its mutations are likely to cause significant perturbations to protein structure and/or energetics.</p><p>A closer examination of the most frequent features in the optimal models reveals considerable variations. For example, distances to DNA and ligand are more important for TetR and RolR than for the other two proteins. Motional correlations with DNA-binding region and other residues are important in MphR, while their significance is modest in TetR and RolR, and very low in TtgR. Moreover, while the five most frequent features are all global in nature for TetR, the third most frequent feature is solvent accessibility (local feature) for MphR; the most frequent feature is charge (physicochemical feature) for RolR, and the top three most frequent features for TtgR are LSE (local feature), frustration index (local feature), and hydrophobicity (physicochemical feature) (<xref ref-type="fig" rid="fig4">Figure 4B</xref> and <xref ref-type="fig" rid="fig4s7">Figure 4—figure supplement 7</xref>). These observations highlight that being able to exploit the synergy between different classes of features is crucial for an NN model to achieve high performance, and the role of a residue in allostery is likely determined by a range of factors, with the weights of different factors being substantially different even among homologous proteins.</p><p>Therefore, compared to the F score analysis for individual features, the GA-NN analysis has revealed a more nuanced view of properties indicative of allosteric hotspots. To rationalize these observations, we note that all aTFs are structurally divided into LBDs and DBDs, thus allostery relies on both intra-domain properties and inter-domain couplings. While inter-domain couplings dictate the communication between allosteric and active sites, corresponding to a ‘contact relayed signal’ view of allostery (<xref ref-type="bibr" rid="bib77">Wang et al., 2020</xref>; <xref ref-type="bibr" rid="bib49">Motlagh et al., 2014</xref>); intra-domain properties affect allostery by shifting the populations of different thermodynamic states, as described by the classical MWC (Monod-Wyman-Changeux) model and its recent variations (<xref ref-type="bibr" rid="bib46">Marzen et al., 2013</xref>; <xref ref-type="bibr" rid="bib27">Hilser et al., 2012</xref>; <xref ref-type="bibr" rid="bib44">Luo et al., 2021</xref>). Thus, an allosteric hotspot might contribute to one of these two types of properties or both (<xref ref-type="bibr" rid="bib22">Gandhi et al., 2008</xref>). Features indicative of long-range motional correlations are likely discriminative for hotspots important to inter-domain coupling, while features reflecting spatial location of a site (e.g., distance to regions of high centrality) are likely more discriminative for hotspots important to intra-domain properties. Therefore, we speculate that if hotspots of an aTF are mostly important only to inter- or intra-domain properties, global features are highly effective in building high-performing models, like in the case of TetR and MphR. However, if most hotspot residues contribute to both inter- and intra-domain properties, a high level of cooperativity and epistasis is likely essential within the aTF, making hotspot identification intrinsically harder, such as in the case of RolR and TtgR. For these latter cases, global features alone become less effective, and local or intrinsic features like charge, hydrophobicity, LSE, and frustration index, which represent finer description of the local interactions of a site, appear more frequently in the top-performing models.</p></sec><sec id="s2-6"><title>Transfer learning improves cross-protein predictions</title><p>Next, we explored the possibility of predicting hotspots on TetR homolog using models trained with data of other TetR homologs (cross-protein prediction [CPP]) with and without transfer learning (TL). For a given train-test combination (e.g., train on TetR and test on MphR), the prediction accuracy is recorded as its CPP performance. Then, the model is further trained with 10% of the data for the test protein and used to predict on the remaining 90% data, and the prediction result is recorded as its CPP_TL performance (see Materials and methods).</p><p>CPP_TL significantly outperforms their CPP counterparts as well as the corresponding best models trained with 10% data for the test protein only (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). Using MphR as an example, the NN trained with data of TetR, RolR, and TtgR without TL gives the worst performance among the models tested (<xref ref-type="fig" rid="fig5">Figure 5B</xref>), while the model trained with 10% MphR data leads to rather low precision as well, which is not unexpected since it has only seen 10% data during training. By contrast, when the model trained with homologous proteins is further refined with 10% data of MphR, it yields a performance close to the optimal NN model trained specifically for MphR using the same set of features and all MphR data. These observations demonstrate that an NN trained for one protein, although not directly applicable to a homologous protein with modest sequence identity (the highest pairwise sequence identity is 19.2%, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>), can still learn useful information about allostery in the latter. It’s also noted that NNs trained with data of three proteins in general outperform NNs trained with the data of only one of them in predicting on the fourth protein (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). This result can be attributed to the expectation that NNs trained on the data of more proteins are less biased by the distinct characteristics of any single member, thus can learn common rules of allostery in all homologous proteins, leading to better prediction on a new protein unseen during training. In other words, while direct CPP performance is low due to the divergence in detailed allostery mechanism among homologous aTFs as discussed in the last subsection, the observation that TL can be effective suggests that such divergence can be represented by a limited amount of mutation data.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Cross-protein prediction – predicting allosteric hotspots in one homolog using data from other homologs.</title><p>(<bold>A</bold>) Best cross-protein predictions without (CPP, yellow) and with transfer learning (CPP_TL, green) achieved for each protein using models trained with 1–10 features and different training data. The title of each heatmap specifies the target protein being predicted. The label of each row indicates the training dataset used (a protein name means data from that one protein and homologs means data from all other three proteins besides the target protein). The first row reports the best fivefold cross-validation performance achieved using 1–10 features on the target protein, and the first column (marked ‘0’) is the performance of a random model for comparison. The row of 10%_TargetProtein indicates the best performance of neural networks (NNs) trained with only 10% data of the target protein in predicting the rest 90% data. (<bold>B</bold>) Comparison of hotspot predictions of MphR using different models (all employing features 17, 20, 21, 25, 26). Residue numbers of MphR are marked horizontally. Experimental data is the first row; ‘MphR’ shows the result of fivefold cross-validation performance of the NN on the MphR data; ‘10% MphR’ shows the performance of NN trained with 10% MphR data in predicting the rest 90%; ‘Homologs – CPP’ and ‘Homologs – CPP_TL’ shows cross-protein prediction without and with transfer learning of NN trained with data of the other three homologs in predicting MphR.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig5-v2.tif"/></fig><p>Only a fraction of the thousands of TetR-like proteins has three-dimensional (3D) structures. TL could be a powerful tool to predict hotspots in proteins with unknown structure but that have limited experimental data. To explore this idea, we repeated TL studies based on homology models rather than the crystal structure for all four aTFs (<xref ref-type="fig" rid="fig6">Figure 6</xref>). Specifically, we generated homology models using two different protein templates for each of the four proteins, and recalculated all structure-based features in each case. We observed a general trend of increasing model performance with increasing sequence similarity between template protein and MphR and decreasing model performance with increasing RMSD between template protein and MphR (<xref ref-type="fig" rid="fig6">Figure 6</xref> and <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>). Nevertheless, all relative performances are above 83%, which is remarkable considering that the lowest sequence identity between a modeled protein and its template is only 15.4% (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>), suggesting that the TL approach can be effective for predicting allostery in a homologous protein even in the absence of high-resolution structural information.</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Predicting allosteric hotspots using homology models.</title><p>(<bold>A</bold>) Correlation between relative performance and the identity between the template protein and the target protein for modeling. (<bold>B</bold>) Correlation between relative performance and the root mean squared distance (RMSD) between the template protein and the target protein for modeling. R squared shows the coefficient of determination of the corresponding linear regression (red: templates for TetR; blue: templates for MphR; purple: templates for RolR; orange: templates for TtgR).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig6-v2.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Sequence identity and root mean squared distance (RMSD) between template protein and target protein are anticorrelated.</title><p>Correlation of sequence identity and RMSD between the four allosteric transcription factors (aTFs) and their corresponding templates used in generating homology models. R squared shows the coefficient of determination of the corresponding linear regression (red: templates for TetR; blue: templates for MphR; purple: templates for RolR; orange: templates for TtgR).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-79932-fig6-figsupp1-v2.tif"/></fig></fig-group></sec><sec id="s2-7"><title>Comparison with sequence-based featurization</title><p>In recent years, deep representation learning has emerged as an effective method for protein featurization. This approach, which generates a representation of a given sequence, is based on information extracted from the known protein sequence universe (<xref ref-type="bibr" rid="bib66">So and Karplus, 1996a</xref>; <xref ref-type="bibr" rid="bib6">Biswas et al., 2021</xref>; <xref ref-type="bibr" rid="bib23">Garruss et al., 2021</xref>; <xref ref-type="bibr" rid="bib20">Freschlin et al., 2022</xref>; <xref ref-type="bibr" rid="bib2">Alley et al., 2019</xref>). In contrast, our approach is based on features derived from structural and physicochemical properties of amino acids. We sought to compare the performance of our model and a state-of-the-art sequence-based method, UniRep, which represents a protein sequence by a 1900-dimension vector (see Materials and methods for details) (<xref ref-type="bibr" rid="bib80">Werten et al., 2016</xref>). Since UniRep features are based on protein sequences, an NN model trained with these features can be used to predict the protein phenotype upon every mutation (<xref ref-type="table" rid="table1">Table 1</xref>). For the prediction of hotspots (<xref ref-type="table" rid="table2">Table 2</xref>), we first rank all sites based on the fraction of predicted dead mutations for each site. The top N sites are then identified as hotspots, where N is the number of hotspots determined from DMS experiment for the protein of interest. We then evaluate the performance of the model by comparing the list of predicted hotspots with experimentally identified ones. When the UniRep features are combined with our 27 site features, the model can be used to predict mutation phenotypes and allosteric hotspots with the same procedure.</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Mutation phenotype prediction performance<sup>a</sup>.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom">TetR</th><th align="left" valign="bottom">MphR</th><th align="left" valign="bottom">RolR</th><th align="left" valign="bottom">TtgR</th></tr></thead><tbody><tr><td align="left" valign="bottom">UniRep1900</td><td align="char" char="plusmn" valign="bottom">0.50±0.01</td><td align="char" char="plusmn" valign="bottom">0.65±0.00</td><td align="char" char="plusmn" valign="bottom">0.57±0.01</td><td align="char" char="plusmn" valign="bottom">0.43±0.01</td></tr><tr><td align="left" valign="bottom">feat1927</td><td align="char" char="plusmn" valign="bottom">0.53±0.00</td><td align="char" char="plusmn" valign="bottom">0.69±0.00</td><td align="char" char="plusmn" valign="bottom">0.59±0.02</td><td align="char" char="plusmn" valign="bottom">0.44±0.00</td></tr><tr><td align="left" valign="bottom">random</td><td align="char" char="." valign="bottom">0.11</td><td align="char" char="." valign="bottom">0.12</td><td align="char" char="." valign="bottom">0.09</td><td align="char" char="." valign="bottom">0.07</td></tr></tbody></table><table-wrap-foot><fn><p>a. Performances are evaluated as the average performance of five times of fivefold cross-validation tests; Unirep1900 and feat1927 show best NN performance using only Unirep features and using Unirep features in combination with 27 physical features, respectively. Data are presented as average ± std.</p></fn></table-wrap-foot></table-wrap><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>Hotspot prediction performance<sup>a</sup>.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom">TetR</th><th align="left" valign="bottom">MphR</th><th align="left" valign="bottom">RolR</th><th align="left" valign="bottom">TtgR</th></tr></thead><tbody><tr><td align="left" valign="bottom">feat27</td><td align="char" char="plusmn" valign="bottom">0.83±0.02</td><td align="char" char="plusmn" valign="bottom">0.82±0.02</td><td align="char" char="plusmn" valign="bottom">0.64±0.02</td><td align="char" char="plusmn" valign="bottom">0.54±0.03</td></tr><tr><td align="left" valign="bottom">UniRep1900</td><td align="char" char="plusmn" valign="bottom">0.61±0.07</td><td align="char" char="plusmn" valign="bottom">0.50±0.02</td><td align="char" char="plusmn" valign="bottom">0.32±0.03</td><td align="char" char="plusmn" valign="bottom">0.35±0.03</td></tr><tr><td align="left" valign="bottom">random</td><td align="char" char="." valign="bottom">0.19</td><td align="char" char="." valign="bottom">0.16</td><td align="char" char="." valign="bottom">0.26</td><td align="char" char="." valign="bottom">0.21</td></tr></tbody></table><table-wrap-foot><fn><p>a. Feat27 represents the fitness of the best-performing feature combination emerged in feature selection with the GA-NN approach. Performances are evaluated as the average performance of five times of fivefold cross-validation tests, and presented as average ± std.</p></fn></table-wrap-foot></table-wrap><p>As summarized in <xref ref-type="table" rid="table1 table2">Tables 1–2</xref>, UniRep feature-based models perform significantly better than the random baselines in both mutation phenotype prediction and hotspot prediction for all four homologous aTFs, highlighting the ability of such models in distilling fundamental features of a protein. However, there is a noticeable gap between hotspot prediction performance using UniRep features and our optimal models. This gap can be understood from the results of our above analysis using F score and JSD that global structure-based features are more indicative of allostery than sequence-based features. While our 27 features provide explicit and comprehensive descriptions of a residues’ physicochemical property, location, motional correlation, and local environment in the context of the entire protein structure, sequence-based featurization attempts to infer such information from the sequence universe in order to establish an interpretable model without using structural information explicitly. When the 1900 UniRep features are combined with the 27 physical features in our model, a marginal yet consistent improvement in mutation phenotype prediction is observed (<xref ref-type="table" rid="table1">Table 1</xref>). This suggests that combining sequence-based features and features of clearer physical meaning can lead to improved predictive power. TL also proves to be effective in boosting the performance of cross-protein mutation prediction when the UniRep features and the 27 physical features are combined (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>).</p></sec><sec id="s2-8"><title>Concluding remarks</title><p>Prediction of residues essential to protein allostery is of great fundamental and biomedical significance. An important question is to what degree allosteric hotspots and, therefore, mechanistic details of allostery, are conserved among homologous proteins. In this study, by combining DMS and machine learning analyses of four homologous aTFs, we have gained new understanding of this question. DMS has enabled a systemic, function-centric, approach to identify allosteric hotspots in proteins. Analysis of the distribution and basic properties of allosteric hotspots in the four aTFs has revealed key insights. First, hotspot residues are distributed across the structure rather than being limited to the specific pathway(s) that connect the inducer and DNA-binding sites as commonly assumed in allostery models. Nonetheless, they are relatively enriched near the dimerization interface and α4 helix at the LBD/DBD interface, highlighting the role of these regions in signal transmission (<xref ref-type="bibr" rid="bib32">Jumper et al., 2021</xref>). Second, we observe that LRIs (in terms of sequence separation) are more prevalent among hotspots, suggesting that in addition to being important to folding and stability, LRIs are also relevant to propagating allostery signals. Third, a systematic analysis of F scores of a diverse set (<xref ref-type="bibr" rid="bib19">Fowler and Fields, 2014</xref>) of protein site features suggests that in all four homologs, global structural and dynamic properties such as distance to centrality peaks, motion covariance with inducer/DNA-binding site residues are more useful than local and intrinsic physicochemical properties for differentiating hotspot from non-hotspot residues. The importance of global properties to the identification of hotspots is further confirmed by GA-NN models that optimize the combination of features to best classify whether a protein site is an allosteric hotspot. Fourth, combined with TL, the GA-NN model trained for one protein can lead to a reasonable prediction of hotspots in a homolog. Further, GA-NN models built using homology-modeled structures rather than actual crystal structure also perform well. These results support the idea that a generally similar allostery mechanism is at play in these homologous proteins.</p><p>On the other hand, we also observe a notable degree of divergence in allosteric hotspot distribution and the features that best define them among the homologous proteins. For example, hotspots in TetR and MphR are more concentrated in the C-terminal end of the protein, while the distributions are more even across the structure and sequence in TtgR and RolR. Related to this difference, the top F scores for different features are substantially lower in TtgR and RolR, highlighting that the hotspots in these two proteins are less distinctive (in terms of the features we have examined) than in TetR and MphR; similarly, the accuracy of the GA-NN model is generally less compelling for TtgR and RolR, when compared to TetR and MphR. Moreover, while global properties are most important to an NN model for the prediction of hotspots in TetR and MphR, local and intrinsic physical properties also contribute in TtgR and RolR. As discussed above, these differences might suggest a higher level of complexity in the allostery mechanism in TtgR and RolR, in which the hotspot residues may contribute to both intra-domain properties and inter-domain coupling. Therefore, allostery mechanisms in homologous proteins may differ substantially in fine details, an observation that has significant implication to the prediction of allostery in related proteins. Along this line, it is satisfying to observe that the TL approach can be rather effective for CPPs; the fact that the performance is not highly sensitive to the resolution of the structural model is particularly encouraging, especially considering recent advances in protein structure predictions (<xref ref-type="bibr" rid="bib54">Orth et al., 2000</xref>).</p><p>Finally, we acknowledge that the accuracy of the GA-NN model, especially for CPP, which is most meaningful from the perspective of application, is not uniform. As mentioned above, the accuracy is generally less compelling for TtgR and RolR, which apparently feature more uniformly distributed hotspot residues and therefore present a higher degree of challenge for both mechanistic understanding and hotspot prediction. To further improve the accuracy of prediction, it is likely worthwhile to include features that better encode the 3D structure with, for example, convolution NN models. Moreover, we have considered only one structural state for each protein, while structural variations among different functional states (e.g., inducer bound vs. DNA bound) are expected to be informative (<xref ref-type="bibr" rid="bib32">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib36">Kosuri et al., 2013</xref>). Finally, while we found limited value in including generic sequence-based features in the current work, it is possible that a more protein-family-specific set of sequence-based features can better contribute. Ultimately, we envision that by judiciously combining different types of experimental data and machine learning techniques in the framework of theoretical models (e.g., variations of the MWC model), we are able to not only predict but interpret the contribution of hotspot residues, which will pave the way for rational engineering of allostery to achieve desired biological function.</p></sec></sec><sec id="s3" sec-type="materials|methods"><title>Materials and methods</title><sec id="s3-1"><title>Plasmid construction</title><p>We constructed a sensor plasmid with TtgR (Uniprot #Q88N29) and RolR (Uniprot #Q8NR95) cloned into a low-copy backbone (SC101 origin of replication) carrying spectinomycin resistance. The ttgR gene was driven by a variant of promoter apFAB61 and Bba_J61132 RBS while the apFAB50 promoter and BBa_J61119 RBS expressed rolR (<xref ref-type="bibr" rid="bib72">Terán et al., 2003</xref>). On a second reporter plasmid, superfolder (sf) GFP was cloned into a high-copy backbone (ColE1 origin of replication) carrying kanamycin resistance. In the TtgR reporter, sfGFP was under the control of the native promoter driving ttgA expression (<xref ref-type="bibr" rid="bib40">Li et al., 2012</xref>) modified to contain canonical –10 (5’-TATAAT-3’) and –35 (5’-TTGACA-3’) and the g10 RBS. sfGFP in the RolR reporter was driven by the lac operon promoter with rolO (<xref ref-type="bibr" rid="bib63">Rogers et al., 2015</xref>) upstream of –35 and Bujard RBS. To control for plasmid copy number, red fluorescent protein (RFP) was constitutively expressed with the BBa_J23106 promoter and Plotkin RBS (<xref ref-type="bibr" rid="bib72">Terán et al., 2003</xref>) in a divergent orientation to sfGFP. Plasmid construction for TetR(B) was previously described (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>) and pJRK-H-mphR from the Church lab was obtained for MphR (<xref ref-type="bibr" rid="bib45">Magoč and Salzberg, 2011</xref>).</p></sec><sec id="s3-2"><title>Library synthesis</title><p>Comprehensive single-mutant libraries of TetR, TtgR, MphR, and RolR were generated by replacing all wild-type residues to all other 19 canonical amino acids starting at position 2 (total mutant sequences – TetR: 3914; TtgR: 3971; MphR: 3667; RolR: 4332). Oligonucleotides encoding each single point mutation were synthesized as single-stranded Oligo Pools from Twist Bioscience and Agilent. Due to limitations in synthesis length, oligonucleotide pools were organized into six to seven subpools spanning the encoding region for each homolog and were encoded and amplified as previously described (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>). Regions of the sensor plasmids corresponding to the oligonucleotide subpools were amplified with primers linearizing the backbone, adding a BsaI restriction site, and removing the wild-type sequence. Vector backbones were further digested with DpnI, BsaI, and Antarctic phosphatase before library assembly.</p><p>We assembled mutant sub-libraries by combining the linearized sensor backbone with each oligo subpool at a molar ratio of 1:5 using Golden Gate Assembly Kit (New England Biolabs; 37°C for 5 min and 60°C for 5 min, repeated ×30). Reactions were dialyzed with water on silica membranes (0.025 μm pores) for 1 hr before transformed into DH10B cells (New England Biolabs). Library sizes of at least 100,000 colony-forming units (CFU) were considered successful. MphR libraries were complete at this point. Cells (New England Biolabs) containing the reporter pColE1_sfGFP_RFP_kanR (DH5α for TetR and RolR, and DH10B for TtgR) were transformed with extracted plasmids to obtain libraries of at least 100,000 CFU.</p></sec><sec id="s3-3"><title>Fluorescence-activated cell sorting</title><p>Library cultures for each subpool were grown in triplicate for 16 hr at 37°C in lysogeny broth (LB) containing 50 µg/mL kanamycin and 100 µg/mL spectinomycin for TetR, TtgR, and RolR; MphR cultures were maintained with 100 µg/mL carbenicillin. Libraries were seeded from a 50 µL aliquot of glycerol stocks and grown to an OD<sub>600</sub> ~ 0.2 before being split in two and induced with 1 µM aTC, 500 µM Nar, 1 mM Ery, or 7.5 mM Res, and grown overnight. Saturated (un)induced sub-library cultures were split into two groups and pooled based location in the gene for sorting and sequencing: sub-libraries 1-3 covered the N terminus while sub-libraries 4-6/7 covered the C terminus of the homologs. Pooled sub-libraries were diluted 1:50 in ×1 phosphate buffered saline and fluorescence intensity was measured on an SH800S Cell Sorter (Sony). Remaining uninduced cultures were spun down and plasmids were extracted for next-generation sequencing to represent the presorted library, identifying all variants present in the library. For sorting, we first gated cells to remove debris and doublets and selected for variants constitutively expressing RFP; this gate was skipped for MphR which did not express RFP. The induction profile of each wild-type homolog was used as reference in drawing gates on GFP fluorescence (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Uninduced and induced pooled sub-libraries were sorted between ~10 and 1000 RFU (based on fluorescence distribution of repressed, DNA-bound wild-type TetR homologs; <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>) to identify nonfluorescent, inactive variants. A total of 500,000 events were sorted for each gated population and cells were recovered in 5 mL of LB for 1 hr before antibiotics were added and cultures grown for an additional 6 hr until an OD<sub>600</sub> ~ 0.2 was reached when cells were spun down and plasmids extracted for sequencing. Each library was grown, sorted, and sequenced in triplicate.</p></sec><sec id="s3-4"><title>NGS preparation and analysis</title><p>In total, three conditions were sequenced in triplicate for each homolog sub-library: (1) the presorted population, (2) the sorted nonfluorescent, uninduced population, and (3) the sorted nonfluorescent, induced population. Sub-libraries were prepared for sequencing with plasmids extracted from the each of the three populations, amplified with two primer sets in a two-step PCR, and sequenced using a 2×250 Illumina MiSeq run as previously described (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>). Paired-end Illumina sequencing reads were merged with FLASH (Fast Length Adjustment of SHort reads) using the default software parameters (<xref ref-type="bibr" rid="bib15">Edgar and Flyvbjerg, 2015</xref>). Phred quality scores were used to compute the total number of expected errors for each merged read (<xref ref-type="bibr" rid="bib59">Potter et al., 2018</xref>). Reads exceeding the maximum expected error threshold of 1 were removed.</p><p>Before analysis, two separate normalizations were performed on the total sequence reads to compare and draw common thresholds (1) between experimental conditions and replicates and (2) across proteins. First, total sequencing reads were normalized to 200k total (100k for each sub-library) across all three conditions and replicates for each homolog. Next, reads were normalized to account for differences in theoretical size of each protein’s single-mutant library. For example, reads of RolR (4332 possible mutants) increased by ×1.18 relative to MphR (3667 possible mutants) for a total of 236k reads. A read threshold of 5 was then applied across all replicates, conditions, and proteins to reduce sequencing noise; increasing this threshold to 10 reads did not significantly affect final analyses or positions identified as hotspots (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>).</p><p>Variants that did not have at least 5 reads in all replicates of the presorted population were not considered present in the dataset (gray, <xref ref-type="fig" rid="fig1s2">Figure 1—figure supplements 2</xref>–<xref ref-type="fig" rid="fig1s3">3</xref>). To be classified as ‘dead’ within a single replicate, variants must have at least 5 reads in both the induced and uninduced sorted populations. There was good correlation between replicates in the number of dead variants identified a every position within the protein (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). Dead variants were then given a score of 0, 1, or 2 based on how many replicates within a protein they were identified as dead in as a measure of confidence in calling these variants dead. These scores were then used to calculate a weighted score for every position in the protein based on the number and confidence of dead variants at that position using <xref ref-type="disp-formula" rid="equ5">Equation 5</xref>:<disp-formula id="equ5"><label>(5)</label><mml:math id="m5"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mrow><mml:mo>∗</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">D</mml:mi></mml:mrow><mml:msub><mml:mn>1</mml:mn><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>∗</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">D</mml:mi></mml:mrow><mml:msub><mml:mn>2</mml:mn><mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mo>∗</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">D</mml:mi></mml:mrow><mml:msub><mml:mn>3</mml:mn><mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">t</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>At position x, D1 is the number of variants dead in one replicate, D2 is the number dead in two replicates, D3 is the number dead in three replicates, and Total is the total number of variants present in the dataset. Dead variants present in only one replicate were not considered confident enough to include in the weighted score and were discarded. The interquartile range of weighted scores for every protein was calculated and positions with weighted scores above the calculated Q3 were identified as allosteric hotspots (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>). Sequencing data uploaded here. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.7020077">https://doi.org/10.5281/zenodo.7020077</ext-link>.</p></sec><sec id="s3-5"><title>Sequence conservation</title><p>Sequence conservation of TetR(B) was previously calculated (<xref ref-type="bibr" rid="bib39">Leander et al., 2020</xref>). Homologs of TtgR, MphR, and RolR were identified using HMM search (<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/Tools/hmmer/">https://www.ebi.ac.uk/Tools/hmmer/</ext-link>) (<xref ref-type="bibr" rid="bib38">Larkin et al., 2007</xref>) against UniProtKB database with individual sequences used as queries. Sequences with alignment coverage less than 95% of full-length TtgR, MphR, and RolR were removed from consideration. The remaining sequences were aligned using Clustal Omega (<xref ref-type="bibr" rid="bib78">Waterhouse et al., 2009</xref>). After applying a redundancy cutoff of 90%, we were left with 500–6000 which was used to evaluate sequence conservation score within Jalview (<xref ref-type="bibr" rid="bib43">Livingstone and Barton, 1993</xref>). Conservation score in Jalview is computed by AMAS tool (<xref ref-type="bibr" rid="bib75">Vehlow et al., 2011</xref>) and positions with a score of 7 or more were termed highly conserved. Two-sample t-tests were used to compare the average conservation score of residues classified as dead or no effect for each homolog. Ligand-contacting residues, defined as contacts within 5 Å of the ligand, were removed when calculating average conservation.</p></sec><sec id="s3-6"><title>Mapping, clustering, and ranking LRIs</title><p>Contact maps of TetR homologs were generated using CMView (<xref ref-type="bibr" rid="bib35">Kellogg et al., 2011</xref>). Crystal structures of TetR (PDB ID: 4AC0), TtgR (PDB ID: 2UXU), MphR (PDB ID: 3FRQ), and RolR (PDB ID: 3AQT) dimers were obtained and removed from ligands and water molecules. Structures were edited to combine the two monomers and renumber residues to identify intermolecular dimer interactions. Interactions between α carbon atoms within 8 Å of were identified and a minimum sequence separation of 10 residues was set to select for LRIs. For each homolog, k-means clustering was used to identify subgroups of LRIs based on location similarity in the contact map. The elbow method was used to determine the optimal number of clusters in which the within-cluster sum of squares was minimized (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>); 10 clusters were chosen for each homolog. Clusters within each contact map were then ranked based on the percent of unique hotspots within the cluster (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). A paired t-test was used to compare the percentage of hotspot and non-hotspot residues within all four homologs participating in LRIs.</p></sec><sec id="s3-7"><title>Amino acid physiochemical properties</title><p>Physicochemical properties of mutations were compared by binning all substitutions that were dead or had no effect, removing ligand-contacting residues, across all four TetR homologs and calculating the average hydrophilicity, hydrophobicity, polarity, mass, SASA, and polarizability. A two-sample t-test was used to compare the means of the dead and no effect mutations for each of the six properties.</p></sec><sec id="s3-8"><title>ΔΔG calculations and structural models of mutations</title><p>The crystal structure of TetR(B) with bound [Minocycline:Mg]<sup>+</sup> dimer structure was obtained and water molecules removed before calculations run; the bound ligand was also removed from TetR(B). All modeling calculations were performed using the Rosetta molecular modeling suite v3.9. Single-point mutants were generated using the standard ddg_monomer application (<xref ref-type="bibr" rid="bib34">Kawashima and Kanehisa, 2000</xref>), which enables local conformational to minimize energy. Calculations were run at every position in protein for all 20 amino acids, generating 50 possible mutant and wild-type structural models for each protein variant. Structures with the lowest total energy from the 50 mutant and wild-type models were used to calculate ΔΔG and served as models for structural analysis.</p></sec><sec id="s3-9"><title>Calculation of physicochemical features (#1–8)</title><p>The eight physicochemical properties of wild-type amino acids, molecular weight, number of electrostatic charges, hydrophobicity, aromaticity, number of potential hydrogen bond, polarity, polarizability, and flexibility are obtained from the AAindex database (<xref ref-type="bibr" rid="bib79">Waterhouse et al., 2018</xref>).</p></sec><sec id="s3-10"><title>Calculation of local structural features (#9–16)</title><p>The PDB structures or the modeled structures, generated using the SWISS-MODEL webserver (<xref ref-type="bibr" rid="bib5">Baxa et al., 2014</xref>), of the four homologs were used for the calculation of any structure-based features in the corresponding cases. Local atomic density of a residue R was calculated as the number of atoms from other residues that are within 5 Å to any atom of the residue R. Backbone entropy loss and sidechain entropy loss of a residue, measuring the loss of conformational entropy of a residue upon protein folding, were calculated with the PLOPS webserver based on the crystal structure of the protein (<xref ref-type="bibr" rid="bib31">Joosten et al., 2011</xref>). SASA was calculated with the DSSP webserver (<xref ref-type="bibr" rid="bib33">Kabsch and Sander, 1983</xref>; <xref ref-type="bibr" rid="bib41">Li et al., 2017</xref>). The number of potential hydrogen bonds of a residue was calculated with the WHAT IF webserver maintained by the Vrient group at the Radboud University (<ext-link ext-link-type="uri" xlink:href="https://swift.cmbi.umcn.nl/servers/html/index.html">The WHAT IF Web Interface (umcn.nl)</ext-link>). The single-residue frustration index was calculated with AWSEM-MD Frustratometer based on the crystal structure of the protein (<xref ref-type="bibr" rid="bib7">Chakrabarty and Parekh, 2016</xref>). It measures how energetically favorable the wild-type residue is for its position in the 3D structure of the protein, when compared with the other 19 possible amino acid choices. X-ray crystallographic B-factor of a residue was obtained from the ENM webserver (<xref ref-type="bibr" rid="bib41">Li et al., 2017</xref>). LSE measures the likelihood of the local sequence around the residue to change secondary structure. The calculation of LSE for a residue follows the method described by Hwang et al., which is based on the probability of the four 4-residue sequences that contain the target residue to assume eight different secondary structures as observed in the protein data bank (<xref ref-type="bibr" rid="bib29">Jenik et al., 2012</xref>).</p></sec><sec id="s3-11"><title>Calculation of global structural features (#17–27)</title><p>The correlation of motion of a residue with the DNA or ligand region was calculated as the maximum absolute value of correlation it has with any of the 10 residues that are closest to DNA/ligand. Orientational cross-correlations between residue fluctuations are calculated using the ENM server (<xref ref-type="bibr" rid="bib41">Li et al., 2017</xref>), the values vary from –1 (fully anticorrelated motions) to +1 (fully correlated). Maximum correlation of motion of a residue was calculated as the average of the five largest absolute values of correlation the residue has with any other residue. Distance of a residue to DNA was calculated by first modeling a DNA sequence of 15 nucleotide pairs to the proteins studied through structural alignment with the PDB structure of 1QPI. The distance between the α carbon of a residue and the closest DNA nucleotide, where the position of a nucleotide is represented by the position of its center of mass, was then calculated. Distance of a residue to ligand was calculated as the distance between the α carbon of the residue to the closest center of mass of a ligand.</p><p>Centrality scores of each residue were calculated using the Network Analysis of Protein Structures (NAPS) server (<ext-link ext-link-type="uri" xlink:href="http://bioinf.iiit.ac.in/NAPS/">http://bioinf.iiit.ac.in/NAPS/</ext-link>) (<xref ref-type="bibr" rid="bib4">Bahar and Rader, 2005</xref>). The unweighted atom pair contact network of each structure was generated using a 0–5 Å threshold and node centrality was measured by closeness, or the shortest distance of one position to all others in the network. Distances to centrality peaks were measured by first identifying four major peaks in <xref ref-type="fig" rid="fig4s8">Figure 4—figure supplement 8</xref> for each of the four homologs. Distances to peaks 1–4 for a residue are the distances between the α carbon of the residue to those of the four residues at the four peak positions (<xref ref-type="fig" rid="fig4s8">Figure 4—figure supplement 8</xref>).</p><p>Sequence propagation of a residue R was defined as the largest sequence separation between R and all other residues within 5 Å of R, as was measured by the distance between α carbon atoms.</p></sec><sec id="s3-12"><title>Machine learning methods</title><sec id="s3-12-1"><title>Architecture of the NN</title><p>We used the Keras machine learning package to build and train fully connected feedforward NNs. All implemented NNs (except those involving UniRep features) have one hidden layer of 10 neurons (with RELU activation) and 2 neurons (with softmax activation) in the output layer. Xavier initialization, Adams optimizer, categorical cross-entropy loss function, and a learning rate of 0.0007 were used in all cases.</p><p>NNs using UniRep features alone (labeled as UniRep1900) and NNs using UniRep features in combination with the 27 physical features (labeled as feat1927) have the same architecture as the NNs mentioned above except for an additional batch normalization layer in front of the hidden layer of 10 neurons. Hyperparameters (learning rate, epochs, class weights in the loss function) are tuned in a grid search to maximize performance.</p></sec><sec id="s3-12-2"><title>Evaluation of different feature combinations for a given dataset</title><p>For a given dataset (e.g., data of a single protein), performance of different feature combinations is evaluated through fivefold cross-validation. Specifically, in fivefold cross-validation, the given dataset is randomly divided into five equal partitions. Each partition is used as the test set once, while the other four partitions are used as the training set for the NN. The fivefold cross-validation performance is then evaluated as the average of the five test F1 scores. The fitness of a given feature combination is evaluated as the average performance of five times of fivefold cross-validation.</p></sec><sec id="s3-12-3"><title>Selection of best feature combinations</title><p>For a given dimension of the feature space (p), we used a genetic algorithm to select the best feature combinations in terms of their fitnesses for a dataset for the given p. Specifically, we start with a randomly generated initial gene pool (generation 1) containing 300 genes, with each gene being a different p-feature combination and its fitness evaluated and recorded. A point mutation (change of 1 feature) is then made to each gene (parent) to generate 300 new genes (sons) that have not been evaluated before. The sons are then evaluated, and the 300 fittest genes are selected from the composite pool (parents plus sons) to form the next generation.</p></sec></sec><sec id="s3-13"><title>Cross-protein predictions</title><p>When making predictions on a test protein A using NNs trained with data of a different protein B, an NN is trained with all data of protein B using a certain feature combination, and make predictions on all data of protein A to obtain a test F1 score. Such CPP procedure is carried out for five times for a given feature combination, and its performance is evaluated as the five-time-average F1 score. All the top 300 p-feature combinations for protein B (the last generation obtained through the genetic algorithm optimization) are evaluated for CPP on other proteins, with p=1–10.</p></sec><sec id="s3-14"><title>CPPs with TL</title><p>Making CPPs with TL contains one more step than the above CPP procedure. Specifically, for a test protein A, its data is randomly partitioned into 10 equal subsets. For each subset, an NN trained with all data of protein B using a certain feature combination is further trained with one subset of data of protein A. The NN is then used to make predictions on the other nine subsets to obtain one test F1 score. The CPP (with TL) performance of the feature combination is then evaluated as the 10-time-average F1 score.</p></sec><sec id="s3-15"><title>Mutation phenotype and hotspot prediction using UniRep features</title><p>NNs using UniRep1900 features and feat1927 can be readily used for mutation phenotype prediction as UniRep generates a distinct 1900-dimensional vector to represent each mutant sequence. Feat1927 is generated by concatenating the 1900 UniRep features for a mutant and the 27 features of the mutation site. The performance of mutation phenotype predictions is evaluated as the average performance of five times of fivefold cross-validation tests.</p><p>The performance of hotspot prediction is calculated based on mutation phenotype prediction result. Specifically, in a fivefold cross-validation for hotspot prediction, all residues of a protein are divided into five random partitions, each serving as test residues once while the other four partitions serving as training residues. For each test residue, the model generates prediction for the phenotype of all of its mutations, with x percent of the mutations being predicted to be dead. Thus in a fivefold cross-validation test, the model generates an x value for every residue in the protein. Then, the top N residues (N is the number of true hotspots of a given repressor determined experimentally) with highest x values are identified as hotspots by the model, we then calculate the F1 score of the result by comparing with the list of true hotspots. The reported performance is the average F1 score of five times of such fivefold cross-validation test.</p></sec></sec></body><back><sec sec-type="additional-information" id="s4"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>Reviewing editor, <italic>eLife</italic></p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Data curation, Software, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Resources, Software, Formal analysis, Supervision, Funding acquisition, Visualization, Writing - original draft, Project administration, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Resources, Supervision, Funding acquisition, Investigation, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s5"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Pairwise sequence identity and similarity.</title></caption><media xlink:href="elife-79932-supp1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>R squared correlation of deads identified at each position between replicates.</title></caption><media xlink:href="elife-79932-supp2-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Cluster rankings.</title></caption><media xlink:href="elife-79932-supp3-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Template information.</title></caption><media xlink:href="elife-79932-supp4-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Cross-protein prediction of mutation phenotype.</title></caption><media xlink:href="elife-79932-supp5-v2.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-79932-mdarchecklist1-v2.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s6"><title>Data availability</title><p>Data included in the manuscript.</p></sec><ack id="ack"><title>Acknowledgements</title><p>This work is funded by NIH Director’s New Innovator Award DP2GM132682 (SR) and Shaw Scientist Award (SR), NIH Molecular Biophysics Training Program T32 GM08293 (ML), and R35-GM141930 (QC). Development of the machine learning model was partially supported by grant ML-21–016 from the Dreyfus foundation (QC). Computational resources from the Extreme Science and Engineering Discovery Environment (XSEDE), which is supported by NSF grant number ACI-1548562, are greatly appreciated; part of the computational work was performed on the Shared Computing Cluster which is administered by Boston University’s Research Computing Services (URL: <ext-link ext-link-type="uri" xlink:href="https://www.bu.edu/tech/support/research/">https://www.bu.edu/tech/support/research/</ext-link>).</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abdel-Magid</surname><given-names>AF</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Allosteric modulators: an emerging concept in drug discovery</article-title><source>ACS Medicinal Chemistry Letters</source><volume>6</volume><fpage>104</fpage><lpage>107</lpage><pub-id pub-id-type="doi">10.1021/ml5005365</pub-id><pub-id pub-id-type="pmid">25699154</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alley</surname><given-names>EC</given-names></name><name><surname>Khimulya</surname><given-names>G</given-names></name><name><surname>Biswas</surname><given-names>S</given-names></name><name><surname>AlQuraishi</surname><given-names>M</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Unified rational protein engineering with sequence-based deep representation learning</article-title><source>Nature Methods</source><volume>16</volume><fpage>1315</fpage><lpage>1322</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0598-1</pub-id><pub-id pub-id-type="pmid">31636460</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Amor</surname><given-names>BRC</given-names></name><name><surname>Schaub</surname><given-names>MT</given-names></name><name><surname>Yaliraki</surname><given-names>SN</given-names></name><name><surname>Barahona</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Prediction of allosteric sites and mediating interactions through bond-to-bond propensities</article-title><source>Nature Communications</source><volume>7</volume><elocation-id>12477</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms12477</pub-id><pub-id pub-id-type="pmid">27561351</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bahar</surname><given-names>I</given-names></name><name><surname>Rader</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Coarse-Grained normal mode analysis in structural biology</article-title><source>Current Opinion in Structural Biology</source><volume>15</volume><fpage>586</fpage><lpage>592</lpage><pub-id pub-id-type="doi">10.1016/j.sbi.2005.08.007</pub-id><pub-id pub-id-type="pmid">16143512</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baxa</surname><given-names>MC</given-names></name><name><surname>Haddadian</surname><given-names>EJ</given-names></name><name><surname>Jumper</surname><given-names>JM</given-names></name><name><surname>Freed</surname><given-names>KF</given-names></name><name><surname>Sosnick</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Loss of conformational entropy in protein folding calculated using realistic ensembles and its implications for NMR-based calculations</article-title><source>PNAS</source><volume>111</volume><fpage>15396</fpage><lpage>15401</lpage><pub-id pub-id-type="doi">10.1073/pnas.1407768111</pub-id><pub-id pub-id-type="pmid">25313044</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Biswas</surname><given-names>S</given-names></name><name><surname>Khimulya</surname><given-names>G</given-names></name><name><surname>Alley</surname><given-names>EC</given-names></name><name><surname>Esvelt</surname><given-names>KM</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Low-N protein engineering with data-efficient deep learning</article-title><source>Nature Methods</source><volume>18</volume><fpage>389</fpage><lpage>396</lpage><pub-id pub-id-type="doi">10.1038/s41592-021-01100-y</pub-id><pub-id pub-id-type="pmid">33828272</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chakrabarty</surname><given-names>B</given-names></name><name><surname>Parekh</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Naps: network analysis of protein structures</article-title><source>Nucleic Acids Research</source><volume>44</volume><fpage>W375</fpage><lpage>W382</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw383</pub-id><pub-id pub-id-type="pmid">27151201</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname><given-names>CH</given-names></name><name><surname>Liang</surname><given-names>HK</given-names></name><name><surname>Hsiao</surname><given-names>NW</given-names></name><name><surname>Ko</surname><given-names>MT</given-names></name><name><surname>Lyu</surname><given-names>PC</given-names></name><name><surname>Hwang</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Relationship between local structural entropy and protein thermostability</article-title><source>Proteins</source><volume>57</volume><fpage>684</fpage><lpage>691</lpage><pub-id pub-id-type="doi">10.1002/prot.20263</pub-id><pub-id pub-id-type="pmid">15532068</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Changeux</surname><given-names>JP</given-names></name><name><surname>Edelstein</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Allosteric mechanisms of signal transduction</article-title><source>Science</source><volume>308</volume><fpage>1424</fpage><lpage>1428</lpage><pub-id pub-id-type="doi">10.1126/science.1108595</pub-id><pub-id pub-id-type="pmid">15933191</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Changeux</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Allostery and the Monod-Wyman-Changeux model after 50 years</article-title><source>Annual Review of Biophysics</source><volume>41</volume><fpage>103</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1146/annurev-biophys-050511-102222</pub-id><pub-id pub-id-type="pmid">22224598</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname><given-names>Q</given-names></name><name><surname>Karplus</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Allostery and cooperativity revisited</article-title><source>Protein Science</source><volume>17</volume><fpage>1295</fpage><lpage>1307</lpage><pub-id pub-id-type="doi">10.1110/ps.03259908</pub-id><pub-id pub-id-type="pmid">18560010</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cuthbertson</surname><given-names>L</given-names></name><name><surname>Nodwell</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The TetR family of regulators</article-title><source>Microbiology and Molecular Biology Reviews</source><volume>77</volume><fpage>440</fpage><lpage>475</lpage><pub-id pub-id-type="doi">10.1128/MMBR.00018-13</pub-id><pub-id pub-id-type="pmid">24006471</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>del Sol</surname><given-names>A</given-names></name><name><surname>Fujihashi</surname><given-names>H</given-names></name><name><surname>Amoros</surname><given-names>D</given-names></name><name><surname>Nussinov</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Residues crucial for maintaining short paths in network communication mediate signaling in proteins</article-title><source>Molecular Systems Biology</source><volume>2</volume><fpage>1</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1038/msb4100063</pub-id><pub-id pub-id-type="pmid">16738564</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Demerdash</surname><given-names>ONA</given-names></name><name><surname>Daily</surname><given-names>MD</given-names></name><name><surname>Mitchell</surname><given-names>JC</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Structure-based predictive models for allosteric hot spots</article-title><source>PLOS Computational Biology</source><volume>5</volume><elocation-id>e1000531</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1000531</pub-id><pub-id pub-id-type="pmid">19816556</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname><given-names>RC</given-names></name><name><surname>Flyvbjerg</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Error filtering, pair assembly and error correction for next-generation sequencing reads</article-title><source>Bioinformatics</source><volume>31</volume><fpage>3476</fpage><lpage>3482</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv401</pub-id><pub-id pub-id-type="pmid">26139637</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Faure</surname><given-names>AJ</given-names></name><name><surname>Domingo</surname><given-names>J</given-names></name><name><surname>Schmiedel</surname><given-names>JM</given-names></name><name><surname>Hidalgo-Carcedo</surname><given-names>C</given-names></name><name><surname>Diss</surname><given-names>G</given-names></name><name><surname>Lehner</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Mapping the energetic and allosteric landscapes of protein binding domains</article-title><source>Nature</source><volume>604</volume><fpage>175</fpage><lpage>183</lpage><pub-id pub-id-type="doi">10.1038/s41586-022-04586-4</pub-id><pub-id pub-id-type="pmid">35388192</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Flynn</surname><given-names>JM</given-names></name><name><surname>Rossouw</surname><given-names>A</given-names></name><name><surname>Cote-Hammarlof</surname><given-names>P</given-names></name><name><surname>Fragata</surname><given-names>I</given-names></name><name><surname>Mavor</surname><given-names>D</given-names></name><name><surname>Hollins</surname><given-names>C</given-names></name><name><surname>Bank</surname><given-names>C</given-names></name><name><surname>Bolon</surname><given-names>DN</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Comprehensive fitness maps of hsp90 show widespread environmental dependence</article-title><source>eLife</source><volume>9</volume><elocation-id>e53810</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.53810</pub-id><pub-id pub-id-type="pmid">32129763</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fowler</surname><given-names>D</given-names></name><name><surname>Araya</surname><given-names>CL</given-names></name><name><surname>Fleishman</surname><given-names>SJ</given-names></name><name><surname>Kellogg</surname><given-names>EH</given-names></name><name><surname>Stephany</surname><given-names>JJ</given-names></name><name><surname>Baker</surname><given-names>D</given-names></name><name><surname>Fields</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>High-resolution mapping of protein sequence-function relationships</article-title><source>Nature Methods</source><volume>7</volume><fpage>741</fpage><lpage>746</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1492</pub-id><pub-id pub-id-type="pmid">20711194</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fowler</surname><given-names>D.M.</given-names></name><name><surname>Fields</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Deep mutational scanning: a new style of protein science</article-title><source>Nature Methods</source><volume>11</volume><fpage>801</fpage><lpage>807</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3027</pub-id><pub-id pub-id-type="pmid">25075907</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Freschlin</surname><given-names>CR</given-names></name><name><surname>Fahlberg</surname><given-names>SA</given-names></name><name><surname>Romero</surname><given-names>PA</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Machine learning to navigate fitness landscapes for protein engineering</article-title><source>Current Opinion in Biotechnology</source><volume>75</volume><elocation-id>102713</elocation-id><pub-id pub-id-type="doi">10.1016/j.copbio.2022.102713</pub-id><pub-id pub-id-type="pmid">35413604</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fukami-Kobayashi</surname><given-names>K</given-names></name><name><surname>Tateno</surname><given-names>Y</given-names></name><name><surname>Nishikawa</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Parallel evolution of ligand specificity between LacI/GalR family repressors and periplasmic sugar-binding proteins</article-title><source>Molecular Biology and Evolution</source><volume>20</volume><fpage>267</fpage><lpage>277</lpage><pub-id pub-id-type="doi">10.1093/molbev/msg038</pub-id><pub-id pub-id-type="pmid">12598694</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gandhi</surname><given-names>PS</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name><name><surname>Mathews</surname><given-names>FS</given-names></name><name><surname>Di Cera</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Structural identification of the pathway of long-range communication in an allosteric enzyme</article-title><source>PNAS</source><volume>105</volume><fpage>1832</fpage><lpage>1837</lpage><pub-id pub-id-type="doi">10.1073/pnas.0710894105</pub-id><pub-id pub-id-type="pmid">18250335</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garruss</surname><given-names>AS</given-names></name><name><surname>Collins</surname><given-names>KM</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Deep representation learning improves prediction of laci-mediated transcriptional repression</article-title><source>PNAS</source><volume>118</volume><elocation-id>e2022838118</elocation-id><pub-id pub-id-type="doi">10.1073/pnas.2022838118</pub-id><pub-id pub-id-type="pmid">34187888</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gelman</surname><given-names>S</given-names></name><name><surname>Fahlberg</surname><given-names>SA</given-names></name><name><surname>Heinzelman</surname><given-names>P</given-names></name><name><surname>Romero</surname><given-names>PA</given-names></name><name><surname>Gitter</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Neural networks to learn protein sequence-function relationships from deep mutational scanning data</article-title><source>PNAS</source><volume>118</volume><elocation-id>e2104878118</elocation-id><pub-id pub-id-type="doi">10.1073/pnas.2104878118</pub-id><pub-id pub-id-type="pmid">34815338</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname><given-names>J</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Protein allostery and conformational dynamics</article-title><source>Chemical Reviews</source><volume>116</volume><fpage>6503</fpage><lpage>6515</lpage><pub-id pub-id-type="doi">10.1021/acs.chemrev.5b00590</pub-id><pub-id pub-id-type="pmid">26876046</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Halabi</surname><given-names>N</given-names></name><name><surname>Rivoire</surname><given-names>O</given-names></name><name><surname>Leibler</surname><given-names>S</given-names></name><name><surname>Ranganathan</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Protein sectors: evolutionary units of three-dimensional structure</article-title><source>Cell</source><volume>138</volume><fpage>774</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2009.07.038</pub-id><pub-id pub-id-type="pmid">19703402</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hilser</surname><given-names>VJ</given-names></name><name><surname>Wrabl</surname><given-names>JO</given-names></name><name><surname>Motlagh</surname><given-names>HN</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Structural and energetic basis of allostery</article-title><source>Annual Review of Biophysics</source><volume>41</volume><fpage>585</fpage><lpage>609</lpage><pub-id pub-id-type="doi">10.1146/annurev-biophys-050511-102319</pub-id><pub-id pub-id-type="pmid">22577828</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huss</surname><given-names>P</given-names></name><name><surname>Meger</surname><given-names>A</given-names></name><name><surname>Leander</surname><given-names>M</given-names></name><name><surname>Nishikawa</surname><given-names>K</given-names></name><name><surname>Raman</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Mapping the functional landscape of the receptor binding domain of T7 bacteriophage by deep mutational scanning</article-title><source>eLife</source><volume>10</volume><elocation-id>e63775</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.63775</pub-id><pub-id pub-id-type="pmid">33687327</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jenik</surname><given-names>M</given-names></name><name><surname>Parra</surname><given-names>RG</given-names></name><name><surname>Radusky</surname><given-names>LG</given-names></name><name><surname>Turjanski</surname><given-names>A</given-names></name><name><surname>Wolynes</surname><given-names>PG</given-names></name><name><surname>Ferreiro</surname><given-names>DU</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Protein frustratometer: a tool to localize energetic frustration in protein molecules</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>W348</fpage><lpage>W351</lpage><pub-id pub-id-type="doi">10.1093/nar/gks447</pub-id><pub-id pub-id-type="pmid">22645321</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Jones</surname><given-names>EM</given-names></name><name><surname>Lubock</surname><given-names>NB</given-names></name><name><surname>Venkatakrishnan</surname><given-names>A</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Tseng</surname><given-names>AM</given-names></name><name><surname>Paggi</surname><given-names>JM</given-names></name><name><surname>Latorraca</surname><given-names>NR</given-names></name><name><surname>Cancilla</surname><given-names>D</given-names></name><name><surname>Satyadi</surname><given-names>M</given-names></name><name><surname>Davis</surname><given-names>JE</given-names></name><name><surname>Babu</surname><given-names>MM</given-names></name><name><surname>Dror</surname><given-names>RO</given-names></name><name><surname>Kosuri</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Structural and Functional Characterization of G Protein-Coupled Receptors with Deep Mutational Scanning</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/623108</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Joosten</surname><given-names>RP</given-names></name><name><surname>te Beek</surname><given-names>TAH</given-names></name><name><surname>Krieger</surname><given-names>E</given-names></name><name><surname>Hekkelman</surname><given-names>ML</given-names></name><name><surname>Hooft</surname><given-names>RWW</given-names></name><name><surname>Schneider</surname><given-names>R</given-names></name><name><surname>Sander</surname><given-names>C</given-names></name><name><surname>Vriend</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A series of PDB related databases for everyday needs</article-title><source>Nucleic Acids Research</source><volume>39</volume><fpage>D411</fpage><lpage>D419</lpage><pub-id pub-id-type="doi">10.1093/nar/gkq1105</pub-id><pub-id pub-id-type="pmid">21071423</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Back</surname><given-names>T</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Pacholska</surname><given-names>M</given-names></name><name><surname>Berghammer</surname><given-names>T</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Silver</surname><given-names>D</given-names></name><name><surname>Vinyals</surname><given-names>O</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction with alphafold</article-title><source>Nature</source><volume>596</volume><fpage>583</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id><pub-id pub-id-type="pmid">34265844</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kabsch</surname><given-names>W</given-names></name><name><surname>Sander</surname><given-names>C</given-names></name></person-group><year iso-8601-date="1983">1983</year><article-title>Dictionary of protein secondary structure: pattern recognition of hydrogen-bonded and geometrical features</article-title><source>Biopolymers</source><volume>22</volume><fpage>2577</fpage><lpage>2637</lpage><pub-id pub-id-type="doi">10.1002/bip.360221211</pub-id><pub-id pub-id-type="pmid">6667333</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kawashima</surname><given-names>S</given-names></name><name><surname>Kanehisa</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>AAindex: amino acid index database</article-title><source>Nucleic Acids Research</source><volume>28</volume><elocation-id>374</elocation-id><pub-id pub-id-type="doi">10.1093/nar/28.1.374</pub-id><pub-id pub-id-type="pmid">10592278</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kellogg</surname><given-names>EH</given-names></name><name><surname>Leaver-Fay</surname><given-names>A</given-names></name><name><surname>Baker</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Role of conformational sampling in computing mutation-induced changes in protein structure and stability</article-title><source>Proteins</source><volume>79</volume><fpage>830</fpage><lpage>838</lpage><pub-id pub-id-type="doi">10.1002/prot.22921</pub-id><pub-id pub-id-type="pmid">21287615</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kosuri</surname><given-names>S</given-names></name><name><surname>Goodman</surname><given-names>DB</given-names></name><name><surname>Cambray</surname><given-names>G</given-names></name><name><surname>Mutalik</surname><given-names>VK</given-names></name><name><surname>Gao</surname><given-names>Y</given-names></name><name><surname>Arkin</surname><given-names>AP</given-names></name><name><surname>Endy</surname><given-names>D</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Composability of regulatory sequences controlling transcription and translation in <italic>Escherichia coli</italic></article-title><source>PNAS</source><volume>110</volume><fpage>14024</fpage><lpage>14029</lpage><pub-id pub-id-type="doi">10.1073/pnas.1301301110</pub-id><pub-id pub-id-type="pmid">23924614</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kuzmanic</surname><given-names>A</given-names></name><name><surname>Bowman</surname><given-names>GR</given-names></name><name><surname>Juarez-Jimenez</surname><given-names>J</given-names></name><name><surname>Michel</surname><given-names>J</given-names></name><name><surname>Gervasio</surname><given-names>FL</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Investigating cryptic binding sites by molecular dynamics simulations</article-title><source>Accounts of Chemical Research</source><volume>53</volume><fpage>654</fpage><lpage>661</lpage><pub-id pub-id-type="doi">10.1021/acs.accounts.9b00613</pub-id><pub-id pub-id-type="pmid">32134250</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Larkin</surname><given-names>MA</given-names></name><name><surname>Blackshields</surname><given-names>G</given-names></name><name><surname>Brown</surname><given-names>NP</given-names></name><name><surname>Chenna</surname><given-names>R</given-names></name><name><surname>McGettigan</surname><given-names>PA</given-names></name><name><surname>McWilliam</surname><given-names>H</given-names></name><name><surname>Valentin</surname><given-names>F</given-names></name><name><surname>Wallace</surname><given-names>IM</given-names></name><name><surname>Wilm</surname><given-names>A</given-names></name><name><surname>Lopez</surname><given-names>R</given-names></name><name><surname>Thompson</surname><given-names>JD</given-names></name><name><surname>Gibson</surname><given-names>TJ</given-names></name><name><surname>Higgins</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Clustal W and clustal X version 2.0</article-title><source>Bioinformatics</source><volume>23</volume><fpage>2947</fpage><lpage>2948</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btm404</pub-id><pub-id pub-id-type="pmid">17846036</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leander</surname><given-names>M</given-names></name><name><surname>Yuan</surname><given-names>Y</given-names></name><name><surname>Meger</surname><given-names>A</given-names></name><name><surname>Cui</surname><given-names>Q</given-names></name><name><surname>Raman</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Functional plasticity and evolutionary adaptation of allosteric regulation</article-title><source>PNAS</source><volume>117</volume><fpage>25445</fpage><lpage>25454</lpage><pub-id pub-id-type="doi">10.1073/pnas.2002613117</pub-id><pub-id pub-id-type="pmid">32999067</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>T</given-names></name><name><surname>Zhao</surname><given-names>K</given-names></name><name><surname>Huang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>D</given-names></name><name><surname>Jiang</surname><given-names>CY</given-names></name><name><surname>Zhou</surname><given-names>N</given-names></name><name><surname>Fan</surname><given-names>Z</given-names></name><name><surname>Liu</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The TetR-type transcriptional repressor rolr from Corynebacterium glutamicum regulates resorcinol catabolism by binding to a unique operator, rolo</article-title><source>Applied and Environmental Microbiology</source><volume>78</volume><fpage>6009</fpage><lpage>6016</lpage><pub-id pub-id-type="doi">10.1128/AEM.01304-12</pub-id><pub-id pub-id-type="pmid">22706057</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Chang</surname><given-names>YY</given-names></name><name><surname>Lee</surname><given-names>JY</given-names></name><name><surname>Bahar</surname><given-names>I</given-names></name><name><surname>Yang</surname><given-names>LW</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>DynOmics: dynamics of structural proteome and beyond</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>W374</fpage><lpage>W380</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx385</pub-id><pub-id pub-id-type="pmid">28472330</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lisi</surname><given-names>GP</given-names></name><name><surname>Manley</surname><given-names>GA</given-names></name><name><surname>Hendrickson</surname><given-names>H</given-names></name><name><surname>Rivalta</surname><given-names>I</given-names></name><name><surname>Batista</surname><given-names>VS</given-names></name><name><surname>Loria</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Dissecting dynamic allosteric pathways using chemically related small-molecule activators</article-title><source>Structure</source><volume>24</volume><fpage>1155</fpage><lpage>1166</lpage><pub-id pub-id-type="doi">10.1016/j.str.2016.04.010</pub-id><pub-id pub-id-type="pmid">27238967</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Livingstone</surname><given-names>CD</given-names></name><name><surname>Barton</surname><given-names>GJ</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Protein sequence alignments: a strategy for the hierarchical analysis of residue conservation</article-title><source>Computer Applications in the Biosciences</source><volume>9</volume><fpage>745</fpage><lpage>756</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/9.6.745</pub-id><pub-id pub-id-type="pmid">8143162</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Jiang</surname><given-names>G</given-names></name><name><surname>Yu</surname><given-names>T</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Vo</surname><given-names>L</given-names></name><name><surname>Ding</surname><given-names>H</given-names></name><name><surname>Su</surname><given-names>Y</given-names></name><name><surname>Qian</surname><given-names>WW</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Peng</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>ECNet is an evolutionary context-integrated deep learning framework for protein engineering</article-title><source>Nature Communications</source><volume>12</volume><elocation-id>5743</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-021-25976-8</pub-id><pub-id pub-id-type="pmid">34593817</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Magoč</surname><given-names>T</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Flash: fast length adjustment of short reads to improve genome assemblies</article-title><source>Bioinformatics</source><volume>27</volume><fpage>2957</fpage><lpage>2963</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr507</pub-id><pub-id pub-id-type="pmid">21903629</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Marzen</surname><given-names>S</given-names></name><name><surname>Garcia</surname><given-names>HG</given-names></name><name><surname>Phillips</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Statistical mechanics of Monod-Wyman-Changeux (MWC) models</article-title><source>Journal of Molecular Biology</source><volume>425</volume><fpage>1433</fpage><lpage>1460</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2013.03.013</pub-id><pub-id pub-id-type="pmid">23499654</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McCormick</surname><given-names>JW</given-names></name><name><surname>Russo</surname><given-names>MA</given-names></name><name><surname>Thompson</surname><given-names>S</given-names></name><name><surname>Blevins</surname><given-names>A</given-names></name><name><surname>Reynolds</surname><given-names>KA</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Structurally distributed surface sites tune allosteric regulation</article-title><source>eLife</source><volume>10</volume><elocation-id>e68346</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.68346</pub-id><pub-id pub-id-type="pmid">34132193</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miyazawa</surname><given-names>S</given-names></name><name><surname>Jernigan</surname><given-names>RL</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>Residue-Residue potentials with a favorable contact pair term and an unfavorable high packing density term, for simulation and threading</article-title><source>Journal of Molecular Biology</source><volume>256</volume><fpage>623</fpage><lpage>644</lpage><pub-id pub-id-type="doi">10.1006/jmbi.1996.0114</pub-id><pub-id pub-id-type="pmid">8604144</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Motlagh</surname><given-names>HN</given-names></name><name><surname>Wrabl</surname><given-names>JO</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Hilser</surname><given-names>VJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The ensemble nature of allostery</article-title><source>Nature</source><volume>508</volume><fpage>331</fpage><lpage>339</lpage><pub-id pub-id-type="doi">10.1038/nature13001</pub-id><pub-id pub-id-type="pmid">24740064</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nierzwicki</surname><given-names>L</given-names></name><name><surname>East</surname><given-names>KW</given-names></name><name><surname>Morzan</surname><given-names>UN</given-names></name><name><surname>Arantes</surname><given-names>PR</given-names></name><name><surname>Batista</surname><given-names>VS</given-names></name><name><surname>Lisi</surname><given-names>GP</given-names></name><name><surname>Palermo</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Enhanced specificity mutations perturb allosteric signaling in CRISPR-Cas9</article-title><source>eLife</source><volume>10</volume><elocation-id>e73601</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.73601</pub-id><pub-id pub-id-type="pmid">34908530</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nishikawa</surname><given-names>KK</given-names></name><name><surname>Hoppe</surname><given-names>N</given-names></name><name><surname>Smith</surname><given-names>R</given-names></name><name><surname>Bingman</surname><given-names>C</given-names></name><name><surname>Raman</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Epistasis shapes the fitness landscape of an allosteric specificity switch</article-title><source>Nature Communications</source><volume>12</volume><elocation-id>5562</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-021-25826-7</pub-id><pub-id pub-id-type="pmid">34548494</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nussinov</surname><given-names>R</given-names></name><name><surname>Tsai</surname><given-names>CJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Allostery in disease and in drug discovery</article-title><source>Cell</source><volume>153</volume><fpage>293</fpage><lpage>305</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.03.034</pub-id><pub-id pub-id-type="pmid">23582321</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ofran</surname><given-names>Y</given-names></name><name><surname>Rost</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Protein-protein interaction hotspots carved into sequences</article-title><source>PLOS Computational Biology</source><volume>3</volume><elocation-id>e119</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.0030119</pub-id><pub-id pub-id-type="pmid">17630824</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Orth</surname><given-names>P</given-names></name><name><surname>Schnappinger</surname><given-names>D</given-names></name><name><surname>Hillen</surname><given-names>W</given-names></name><name><surname>Saenger</surname><given-names>W</given-names></name><name><surname>Hinrichs</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Structural basis of gene regulation by the tetracycline inducible Tet repressor-operator system</article-title><source>Nature Structural Biology</source><volume>7</volume><fpage>215</fpage><lpage>219</lpage><pub-id pub-id-type="doi">10.1038/73324</pub-id><pub-id pub-id-type="pmid">10700280</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ota</surname><given-names>N</given-names></name><name><surname>Agard</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Intramolecular signaling pathways revealed by modeling anisotropic thermal diffusion</article-title><source>Journal of Molecular Biology</source><volume>351</volume><fpage>345</fpage><lpage>354</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2005.05.043</pub-id><pub-id pub-id-type="pmid">16005893</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Zhan</surname><given-names>W</given-names></name><name><surname>Deng</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Computational identification of binding energy hot spots in protein-RNA complexes using an ensemble approach</article-title><source>Bioinformatics</source><volume>34</volume><fpage>1473</fpage><lpage>1480</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btx822</pub-id><pub-id pub-id-type="pmid">29281004</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Papaleo</surname><given-names>E</given-names></name><name><surname>Saladino</surname><given-names>G</given-names></name><name><surname>Lambrughi</surname><given-names>M</given-names></name><name><surname>Lindorff-Larsen</surname><given-names>K</given-names></name><name><surname>Gervasio</surname><given-names>FL</given-names></name><name><surname>Nussinov</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The role of protein loops and linkers in conformational dynamics and allostery</article-title><source>Chemical Reviews</source><volume>116</volume><fpage>6391</fpage><lpage>6423</lpage><pub-id pub-id-type="doi">10.1021/acs.chemrev.5b00623</pub-id><pub-id pub-id-type="pmid">26889708</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pethe</surname><given-names>MA</given-names></name><name><surname>Rubenstein</surname><given-names>AB</given-names></name><name><surname>Khare</surname><given-names>SD</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Data-driven supervised learning of a viral protease specificity landscape from deep sequencing and molecular simulations</article-title><source>PNAS</source><volume>116</volume><fpage>168</fpage><lpage>176</lpage><pub-id pub-id-type="doi">10.1073/pnas.1805256116</pub-id><pub-id pub-id-type="pmid">30587591</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Potter</surname><given-names>SC</given-names></name><name><surname>Luciani</surname><given-names>A</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Park</surname><given-names>Y</given-names></name><name><surname>Lopez</surname><given-names>R</given-names></name><name><surname>Finn</surname><given-names>RD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>HMMER web server: 2018 update</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>W200</fpage><lpage>W204</lpage><pub-id pub-id-type="doi">10.1093/nar/gky448</pub-id><pub-id pub-id-type="pmid">29905871</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pougach</surname><given-names>K</given-names></name><name><surname>Voet</surname><given-names>A</given-names></name><name><surname>Kondrashov</surname><given-names>FA</given-names></name><name><surname>Voordeckers</surname><given-names>K</given-names></name><name><surname>Christiaens</surname><given-names>JF</given-names></name><name><surname>Baying</surname><given-names>B</given-names></name><name><surname>Benes</surname><given-names>V</given-names></name><name><surname>Sakai</surname><given-names>R</given-names></name><name><surname>Aerts</surname><given-names>J</given-names></name><name><surname>Zhu</surname><given-names>B</given-names></name><name><surname>Van Dijck</surname><given-names>P</given-names></name><name><surname>Verstrepen</surname><given-names>KJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Duplication of a promiscuous transcription factor drives the emergence of a new regulatory network</article-title><source>Nature Communications</source><volume>5</volume><elocation-id>4868</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms5868</pub-id><pub-id pub-id-type="pmid">25204769</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reynolds</surname><given-names>KA</given-names></name><name><surname>McLaughlin</surname><given-names>RN</given-names></name><name><surname>Ranganathan</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Hot spots for allosteric regulation on protein surfaces</article-title><source>Cell</source><volume>147</volume><fpage>1564</fpage><lpage>1575</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2011.10.049</pub-id><pub-id pub-id-type="pmid">22196731</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rivalta</surname><given-names>I</given-names></name><name><surname>Batista</surname><given-names>VS</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Community network analysis of allosteric proteins</article-title><source>Methods in Molecular Biology</source><volume>2253</volume><fpage>137</fpage><lpage>151</lpage><pub-id pub-id-type="doi">10.1007/978-1-0716-1154-8_9</pub-id><pub-id pub-id-type="pmid">33315222</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rogers</surname><given-names>JK</given-names></name><name><surname>Guzman</surname><given-names>CD</given-names></name><name><surname>Taylor</surname><given-names>ND</given-names></name><name><surname>Raman</surname><given-names>S</given-names></name><name><surname>Anderson</surname><given-names>K</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Synthetic biosensors for precise gene control and real-time monitoring of metabolites</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>7648</fpage><lpage>7660</lpage><pub-id pub-id-type="doi">10.1093/nar/gkv616</pub-id><pub-id pub-id-type="pmid">26152303</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sarkisyan</surname><given-names>KS</given-names></name><name><surname>Bolotin</surname><given-names>DA</given-names></name><name><surname>Meer</surname><given-names>MV</given-names></name><name><surname>Usmanova</surname><given-names>DR</given-names></name><name><surname>Mishin</surname><given-names>AS</given-names></name><name><surname>Sharonov</surname><given-names>GV</given-names></name><name><surname>Ivankov</surname><given-names>DN</given-names></name><name><surname>Bozhanova</surname><given-names>NG</given-names></name><name><surname>Baranov</surname><given-names>MS</given-names></name><name><surname>Soylemez</surname><given-names>O</given-names></name><name><surname>Bogatyreva</surname><given-names>NS</given-names></name><name><surname>Vlasov</surname><given-names>PK</given-names></name><name><surname>Egorov</surname><given-names>ES</given-names></name><name><surname>Logacheva</surname><given-names>MD</given-names></name><name><surname>Kondrashov</surname><given-names>AS</given-names></name><name><surname>Chudakov</surname><given-names>DM</given-names></name><name><surname>Putintseva</surname><given-names>EV</given-names></name><name><surname>Mamedov</surname><given-names>IZ</given-names></name><name><surname>Tawfik</surname><given-names>DS</given-names></name><name><surname>Lukyanov</surname><given-names>KA</given-names></name><name><surname>Kondrashov</surname><given-names>FA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Local fitness landscape of the green fluorescent protein</article-title><source>Nature</source><volume>533</volume><fpage>397</fpage><lpage>401</lpage><pub-id pub-id-type="doi">10.1038/nature17995</pub-id><pub-id pub-id-type="pmid">27193686</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sethi</surname><given-names>A</given-names></name><name><surname>Eargle</surname><given-names>J</given-names></name><name><surname>Black</surname><given-names>AA</given-names></name><name><surname>Luthey-Schulten</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Dynamical networks in trna: protein complexes</article-title><source>PNAS</source><volume>106</volume><fpage>6620</fpage><lpage>6625</lpage><pub-id pub-id-type="doi">10.1073/pnas.0810961106</pub-id><pub-id pub-id-type="pmid">19351898</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>So</surname><given-names>SS</given-names></name><name><surname>Karplus</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1996">1996a</year><article-title>Evolutionary optimization in quantitative structure-activity relationship: an application of genetic neural networks</article-title><source>Journal of Medicinal Chemistry</source><volume>39</volume><fpage>1521</fpage><lpage>1530</lpage><pub-id pub-id-type="doi">10.1021/jm9507035</pub-id><pub-id pub-id-type="pmid">8691483</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>So</surname><given-names>SS</given-names></name><name><surname>Karplus</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1996">1996b</year><article-title>Genetic neural networks for quantitative structure-activity relationships: improvements and application of benzodiazepine affinity for benzodiazepine/GABAA receptors</article-title><source>Journal of Medicinal Chemistry</source><volume>39</volume><fpage>5246</fpage><lpage>5256</lpage><pub-id pub-id-type="doi">10.1021/jm960536o</pub-id><pub-id pub-id-type="pmid">8978853</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Starr</surname><given-names>TN</given-names></name><name><surname>Greaney</surname><given-names>AJ</given-names></name><name><surname>Hilton</surname><given-names>SK</given-names></name><name><surname>Ellis</surname><given-names>D</given-names></name><name><surname>Crawford</surname><given-names>KHD</given-names></name><name><surname>Dingens</surname><given-names>AS</given-names></name><name><surname>Navarro</surname><given-names>MJ</given-names></name><name><surname>Bowen</surname><given-names>JE</given-names></name><name><surname>Tortorici</surname><given-names>MA</given-names></name><name><surname>Walls</surname><given-names>AC</given-names></name><name><surname>King</surname><given-names>NP</given-names></name><name><surname>Veesler</surname><given-names>D</given-names></name><name><surname>Bloom</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Deep mutational scanning of SARS-cov-2 receptor binding domain reveals constraints on folding and ACE2 binding</article-title><source>Cell</source><volume>182</volume><fpage>1295</fpage><lpage>1310</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2020.08.012</pub-id><pub-id pub-id-type="pmid">32841599</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Strickland</surname><given-names>D</given-names></name><name><surname>Moffat</surname><given-names>K</given-names></name><name><surname>Sosnick</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Light-activated DNA binding in a designed allosteric protein</article-title><source>PNAS</source><volume>105</volume><fpage>10709</fpage><lpage>10714</lpage><pub-id pub-id-type="doi">10.1073/pnas.0709610105</pub-id><pub-id pub-id-type="pmid">18667691</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Süel</surname><given-names>GM</given-names></name><name><surname>Lockless</surname><given-names>SW</given-names></name><name><surname>Wall</surname><given-names>MA</given-names></name><name><surname>Ranganathan</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Evolutionarily conserved networks of residues mediate allosteric communication in proteins</article-title><source>Nature Structural Biology</source><volume>10</volume><fpage>59</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/nsb881</pub-id><pub-id pub-id-type="pmid">12483203</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Tack</surname><given-names>DS</given-names></name><name><surname>Tonner</surname><given-names>PD</given-names></name><name><surname>Pressman</surname><given-names>A</given-names></name><name><surname>Olson</surname><given-names>ND</given-names></name><name><surname>Levy</surname><given-names>SF</given-names></name><name><surname>Romantseva</surname><given-names>EF</given-names></name><name><surname>Alperovich</surname><given-names>N</given-names></name><name><surname>Vasilyeva</surname><given-names>O</given-names></name><name><surname>Ross</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The Genotype-Phenotype Landscape of an Allosteric Protein</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.09.30.320812</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Terán</surname><given-names>W</given-names></name><name><surname>Felipe</surname><given-names>A</given-names></name><name><surname>Segura</surname><given-names>A</given-names></name><name><surname>Rojas</surname><given-names>A</given-names></name><name><surname>Ramos</surname><given-names>JL</given-names></name><name><surname>Gallegos</surname><given-names>MT</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Antibiotic-Dependent induction of Pseudomonas putida DOT-T1E ttgabc efflux pump is mediated by the drug binding repressor TtgR</article-title><source>Antimicrobial Agents and Chemotherapy</source><volume>47</volume><fpage>3067</fpage><lpage>3072</lpage><pub-id pub-id-type="doi">10.1128/AAC.47.10.3067-3072.2003</pub-id><pub-id pub-id-type="pmid">14506010</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tzeng</surname><given-names>SR</given-names></name><name><surname>Kalodimos</surname><given-names>CG</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Dynamic activation of an allosteric regulatory protein</article-title><source>Nature</source><volume>462</volume><fpage>368</fpage><lpage>372</lpage><pub-id pub-id-type="doi">10.1038/nature08560</pub-id><pub-id pub-id-type="pmid">19924217</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vanwart</surname><given-names>AT</given-names></name><name><surname>Eargle</surname><given-names>J</given-names></name><name><surname>Luthey-Schulten</surname><given-names>Z</given-names></name><name><surname>Amaro</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Exploring residue component contributions to dynamical network models of allostery</article-title><source>Journal of Chemical Theory and Computation</source><volume>8</volume><fpage>2949</fpage><lpage>2961</lpage><pub-id pub-id-type="doi">10.1021/ct300377a</pub-id><pub-id pub-id-type="pmid">23139645</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vehlow</surname><given-names>C</given-names></name><name><surname>Stehr</surname><given-names>H</given-names></name><name><surname>Winkelmann</surname><given-names>M</given-names></name><name><surname>Duarte</surname><given-names>JM</given-names></name><name><surname>Petzold</surname><given-names>L</given-names></name><name><surname>Dinse</surname><given-names>J</given-names></name><name><surname>Lappe</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>CMView: interactive contact MAP visualization and analysis</article-title><source>Bioinformatics</source><volume>27</volume><fpage>1573</fpage><lpage>1574</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr163</pub-id><pub-id pub-id-type="pmid">21471016</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Liu</surname><given-names>C</given-names></name><name><surname>Deng</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Enhanced prediction of hot spots at protein-protein interfaces using extreme gradient boosting</article-title><source>Scientific Reports</source><volume>8</volume><elocation-id>14285</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-018-32511-1</pub-id><pub-id pub-id-type="pmid">30250210</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Jain</surname><given-names>A</given-names></name><name><surname>McDonald</surname><given-names>LR</given-names></name><name><surname>Gambogi</surname><given-names>C</given-names></name><name><surname>Lee</surname><given-names>AL</given-names></name><name><surname>Dokholyan</surname><given-names>NV</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Mapping allosteric communications within individual proteins</article-title><source>Nature Communications</source><volume>11</volume><elocation-id>3862</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-020-17618-2</pub-id><pub-id pub-id-type="pmid">32737291</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waterhouse</surname><given-names>AM</given-names></name><name><surname>Procter</surname><given-names>JB</given-names></name><name><surname>Martin</surname><given-names>DMA</given-names></name><name><surname>Clamp</surname><given-names>M</given-names></name><name><surname>Barton</surname><given-names>GJ</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Jalview version 2 -- a multiple sequence alignment editor and analysis workbench</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1189</fpage><lpage>1191</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp033</pub-id><pub-id pub-id-type="pmid">19151095</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waterhouse</surname><given-names>A</given-names></name><name><surname>Bertoni</surname><given-names>M</given-names></name><name><surname>Bienert</surname><given-names>S</given-names></name><name><surname>Studer</surname><given-names>G</given-names></name><name><surname>Tauriello</surname><given-names>G</given-names></name><name><surname>Gumienny</surname><given-names>R</given-names></name><name><surname>Heer</surname><given-names>FT</given-names></name><name><surname>de Beer</surname><given-names>TAP</given-names></name><name><surname>Rempfer</surname><given-names>C</given-names></name><name><surname>Bordoli</surname><given-names>L</given-names></name><name><surname>Lepore</surname><given-names>R</given-names></name><name><surname>Schwede</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>SWISS-MODEL: homology modelling of protein structures and complexes</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>W296</fpage><lpage>W303</lpage><pub-id pub-id-type="doi">10.1093/nar/gky427</pub-id><pub-id pub-id-type="pmid">29788355</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Werten</surname><given-names>S</given-names></name><name><surname>Schneider</surname><given-names>J</given-names></name><name><surname>Palm</surname><given-names>GJ</given-names></name><name><surname>Hinrichs</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Modular organisation of inducer recognition and allostery in the tetracycline repressor</article-title><source>The FEBS Journal</source><volume>283</volume><fpage>2102</fpage><lpage>2114</lpage><pub-id pub-id-type="doi">10.1111/febs.13723</pub-id><pub-id pub-id-type="pmid">27028290</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wodak</surname><given-names>SJ</given-names></name><name><surname>Paci</surname><given-names>E</given-names></name><name><surname>Dokholyan</surname><given-names>NV</given-names></name><name><surname>Berezovsky</surname><given-names>IN</given-names></name><name><surname>Horovitz</surname><given-names>A</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Hilser</surname><given-names>VJ</given-names></name><name><surname>Bahar</surname><given-names>I</given-names></name><name><surname>Karanicolas</surname><given-names>J</given-names></name><name><surname>Stock</surname><given-names>G</given-names></name><name><surname>Hamm</surname><given-names>P</given-names></name><name><surname>Stote</surname><given-names>RH</given-names></name><name><surname>Eberhardt</surname><given-names>J</given-names></name><name><surname>Chebaro</surname><given-names>Y</given-names></name><name><surname>Dejaegere</surname><given-names>A</given-names></name><name><surname>Cecchini</surname><given-names>M</given-names></name><name><surname>Changeux</surname><given-names>JP</given-names></name><name><surname>Bolhuis</surname><given-names>PG</given-names></name><name><surname>Vreede</surname><given-names>J</given-names></name><name><surname>Faccioli</surname><given-names>P</given-names></name><name><surname>Orioli</surname><given-names>S</given-names></name><name><surname>Ravasio</surname><given-names>R</given-names></name><name><surname>Yan</surname><given-names>L</given-names></name><name><surname>Brito</surname><given-names>C</given-names></name><name><surname>Wyart</surname><given-names>M</given-names></name><name><surname>Gkeka</surname><given-names>P</given-names></name><name><surname>Rivalta</surname><given-names>I</given-names></name><name><surname>Palermo</surname><given-names>G</given-names></name><name><surname>McCammon</surname><given-names>JA</given-names></name><name><surname>Panecka-Hofman</surname><given-names>J</given-names></name><name><surname>Wade</surname><given-names>RC</given-names></name><name><surname>Di Pizio</surname><given-names>A</given-names></name><name><surname>Niv</surname><given-names>MY</given-names></name><name><surname>Nussinov</surname><given-names>R</given-names></name><name><surname>Tsai</surname><given-names>CJ</given-names></name><name><surname>Jang</surname><given-names>H</given-names></name><name><surname>Padhorny</surname><given-names>D</given-names></name><name><surname>Kozakov</surname><given-names>D</given-names></name><name><surname>McLeish</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Allostery in its many disguises: from theory to applications</article-title><source>Structure</source><volume>27</volume><fpage>566</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.1016/j.str.2019.01.003</pub-id><pub-id pub-id-type="pmid">30744993</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xia</surname><given-names>JF</given-names></name><name><surname>Zhao</surname><given-names>XM</given-names></name><name><surname>Song</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>DS</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Apis: accurate prediction of hot spots in protein interfaces by combining protrusion index with solvent accessibility</article-title><source>BMC Bioinformatics</source><volume>11</volume><elocation-id>174</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-11-174</pub-id><pub-id pub-id-type="pmid">20377884</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname><given-names>Y</given-names></name><name><surname>Deng</surname><given-names>J</given-names></name><name><surname>Cui</surname><given-names>Q</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Molecular dynamics simulations establish the molecular basis for the broad allostery hotspot distributions in the tetracycline repressor</article-title><source>Journal of the American Chemical Society</source><volume>144</volume><fpage>10870</fpage><lpage>10887</lpage><pub-id pub-id-type="doi">10.1021/jacs.2c03275</pub-id><pub-id pub-id-type="pmid">35675441</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.79932.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Faraldo-Gómez</surname><given-names>José D</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institutes of Health</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/2022.05.01.490188" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/2022.05.01.490188"/></front-stub><body><p>This article seeks to address a key question in protein biophysics: are the amino acid positions involved in allosteric mechanisms conserved across homologs of a protein family? Or do these mechanisms involve distinct amino acid patterns that vary amongst homologs? To address this question, the authors follow an innovative multidisciplinary approach that combines deep mutational scanning with machine learning; the findings of this study will be highly relevant to protein engineers and biophysicists.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.79932.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Faraldo-Gómez</surname><given-names>José D</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institutes of Health</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2022.05.01.490188">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2022.05.01.490188v1">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Deep mutational scanning and machine learning reveal structural and molecular rules governing allosteric hotspots in homologous proteins&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, and the evaluation has been overseen by José Faraldo-Gómez as the Senior Editor. All reviewers have opted to remain anonymous.</p><p>The reviewers have discussed their reviews with one another and with the Senior Editor, and the consensus is to invite you to submit a revised version of your manuscript that addresses the concerns enumerated below – particularly but not exclusively those put forward by Reviewer #3.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>The paper is well written, the experiments are appropriately performed. I would like to encourage the authors to make the raw data available. I don't have any suggestions for changes to the manuscript and think that it makes a valuable contribution to the literature while noting that it leaves a few questions open-ended and that some of the speculation regarding the molecular basis could be tested experimentally. Obviously, this reflects my biases and interests as somebody interested also in structure and dynamics. Overall it's a very nice paper, congratulations.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>In the Public Review I provided my overall comments. Below my aim is to make the concept and method clearer to the community.</p><p>With this aim in mind, there is one thing that I am missing. That is, how the authors define 'hotspots' via DMS. I think that it is important to clarify. This will help the readers.</p><p>I would also suggest to consider a figure comparing the hotspots in this paper to the hotspots as defined by the propagation pathways from the allosteric binding site to the distal active site. Would mapping these on a structure (e.g. using some server?) work? This could help the reader in visualizing the difference between the new concept suggested here and the 'traditional' definition.</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>Figure 1 supplement 1 sets the dynamic range of the assay and seems critical to the interpretation of the experiment. From these data, it would seem that the RolR assay lacks discriminating power across mutations. Consider promoting this figure to the main text.</p><p>We found the selection of the top 25% scoring residues as &quot;allosteric hotspots&quot; somewhat arbitrary – is it possible to instead select a cutoff based on the resolution (error) of the assay? Surely the top 25% of allosteric hotspots for RolR (which has a limited dynamic range) and TetR (which has a more extensive dynamic range) have very different biophysical effect sizes, and so this is not an apples-to-apples comparison?</p><p>Consider describing the behavior of previously well-characterized aTF mutations in your assay. This would help build confidence that the assay is truly reporting on allosterically dead mutations.</p><p>Consider using Fisher's exact test to assign a p-value describing the significance of enrichment/depletion of particular residue types in figure 3A</p><p>Figure 3C does not have a legend. Also, the figure seems to show K193Y while the main text refers to an inactive Y198 mutation.</p><p>In the methods section, it was unclear how the library was broken up. It seems like it was sequenced in two parts (to cover the entire open reading frame), but was this sequencing two regions of the same cultured and sorted sample? Or was the library broken into sublibraries that were cultured and sorted in smaller batches and then sequenced?</p><p>We understand that mutations at ligand-contacting positions were not considered, since these mutations are not allosteric (they instead directly affect ligand binding). How was ligand contacting defined?</p><p>Many times, plausible explanations/ideas seemed to be asserted as fact. We feel that these claims either need a citation, more expression of their uncertainty (explaining their lack of evidence in data and the literature), or a more careful explanation:</p><p>1. Abstract, page 2: &quot;We found hotspots to be distributed protein-wide rather than being restricted to &quot;pathways&quot; linking allosteric and active sites as is commonly assumed&quot; It is to our knowledge that it has never been asserted that allosteric hotspots themselves form pathways. Rather it is the assertion (in the prior literature cited by the authors) that allosteric hotspots preferentially contact coevolving networks of residues. In many cases, these co-evolving networks don't just link allosteric to the active site, but often connect other distal surfaces (with no known allosteric function) to the active site – going beyond the idea of a single pathway that the authors seem to imply.</p><p>2. Results page 5: &quot;An aTF mutant that increases the thermodynamic gap between inactive and active states by stabilizing the inactive state will constitutively lock the protein in the inactive allosteric state. We term these &quot;dead&quot; variants. The dead variants are well-folded proteins that can bind to DNA and repress transcription but cannot be induced with ligand.&quot; The authors have made no measurements of protein stability. The only measurements made are cellular GFP levels in the presence and absence of the ligand. This needs a citation or some moderation of language.</p><p>3. Results page 6: &quot;The evolution of aTFs has occurred through a series of gene duplication events resulting in mixing and matching LBDs and DBDs. &quot; Citation needed.</p><p>4. Results page 6: &quot;Thus, the DBDs likely exist as stand-alone domains that are not allosterically &quot;wired&quot; to the rest of the protein at the residue level but instead respond to large thermodynamic changes (e.g., inducer binding).&quot; The data does not support this conclusion. The data suggests that the DBD is qualitatively depleted for mutations that abolish ligand-based activation while maintaining apo repression.</p><p>5. Results page 6: &quot;Taken together, these observations show that although the hotspots are not superimposable across aTFs, the TetR-family likely share a conserved structural mechanism where the allosteric signal travels from the LBD through the dimer interface and a4 to the DBD, while the DBD itself acts as an internally rigid module that docks on DNA.&quot; This is speculation that seems more appropriate to the discussion (rather than results) section.</p><p>6. Results page 7: &quot;These results also show that although allosteric hotspots may not be superimposable across distant homologs, local clusters of LRIs share similar patterns between homologs. As homologs get closer in sequence, regional similarities in allosteric signaling may give way to the superimposability of individual hotspots. &quot; This again seems more appropriate to the discussion (rather than results) section.</p><p>7. Results page 8: &quot;We concluded that the interaction energy of the large hydrophobic sidechains provides an enthalpic gain that stabilizes the allosteric OFF state of the protein.&quot; The thermodynamic mechanism of the mutations was not investigated in this study. We feel that it would be a stretch to form conclusions about enthalpy.</p><p>[Editors’ note: further revisions were suggested prior to acceptance, as described below.]</p><p>Thank you for resubmitting your work entitled &quot;Deep mutational scanning and machine learning reveal structural and molecular rules governing allosteric hotspots in homologous proteins&quot; for further consideration by <italic>eLife</italic>. Your revised article has been evaluated by the 3 original referees. I am glad to be able to inform the reviewers have decided to recommend that this work be published in <italic>eLife</italic>, pending revisions. As you will see below, one of the reviewers requires some additional clarifications in regard to the methodology, which might also help future readers to better appreciate the value of the work. Therefore we would like to offer you the opportunity to clarify these issues – which in my view would require editing of the manuscript and possibly moving Figure 1 S1 to the main text.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>All the comments have been addressed satisfactorily in my opinion.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>The revised manuscript addresses my comments/suggestions, and I think of the other reviewers as well. The paper is an excellent contribution to the literature in an important area and can be accepted as is. It is also an additional highly innovative and original work by the authors.</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>Thank you to the authors for a substantial revision. I very much appreciated the additions to the flow and NGS preparation/analysis methods sections, the clarifications on chip-based library construction, the more complete analysis of replicates, and the specification of the equation used to score allosterically dead variants. I also found the comparison to the ohm server predictions an interesting addition. As stated in my first review, the questions the authors seek to address – both (1) how allostery is implemented across homologs and (2)what physicochemical factors distinguish allosteric hotspots – are timely and fundamental open problems in protein biophysics. The paper contributes an enormous amount of experimental data, and the strategy of using machine learning to identify relevant features that distinguish hotspots is creative and leads to interesting results.</p><p>However, I still have substantial reservations about two aspects of the data analysis that persist from my earlier review. These are considerable enough that in parts of the manuscript I do not feel that the data support the authors' conclusions, but rather suggest something different. The first is that the strategy for deciding which mutations are allosterically dead still just doesn't make sense to me. I really might be missing something here; maybe the authors can explain. The second lies in the usage of quartiles to assign allosteric hotspots, and how this impacts the two constructs with a more limited allosteric dynamic range (RolR and TtgR).</p><p>Major concerns:</p><p>1. Strategy for assigning allosterically dead mutations. I appreciate that the authors clarified in their revisions that sequencing reads were normalized across replicates, experimental conditions, and proteins. These normalizations make sense to me, and I see that this helps in comparing counts. I also appreciate the authors' statements that &quot;we use cell sorting as a binary classifier&quot; and that by counting the number of dead mutations at a position they safeguard against noise in the data. Their point is well-taken that they seek to categorize mutations as allosterically dead/not dead, rather than using the flow data as a quantitative high-resolution measure. But I am still really stuck on understanding the use of a single threshold of 5 (or 10) to assess if a mutant is present in both the uninduced and induced sorted populations, and therefore assign it as allosterically dead. To illustrate, consider the following scenario. Let's imagine the authors ALSO sequenced the sorted induced fluorescent population (in addition to the induced nonfluorescent population). Now consider two different mutations with the following read distributions:</p><p>a. Mutant &quot;A&quot;: 10 reads in the uninduced population, 10 reads in the induced non-fluorescent population, and 0 reads in the induced fluorescent population.</p><p>b. Mutant &quot;B&quot;: 1000 reads in the uninduced population, 10 reads in the induced non-fluorescent population, and 900 reads in the induced fluorescent population.</p><p>If I understand correctly, both mutants would be classified as allosterically dead according to the authors' method. This makes sense for Mutant A, but for Mutant B…. it looks like it activates, just not completely, or maybe there is some noise in the sorting data. Is it obvious that if there are five or ten reads present it truly isn't noise? (How often do the authors observe &quot;impossible codons&quot; – meaning codons that are not part of their chip-based library – in the induced non-fluorescent population? This might set the noise threshold?) My impression is that the single threshold approach used by the authors may overestimate the number of &quot;dead&quot; mutations. It seems like it would be more correct to consider the ratio of the number of reads in the induced, sorted, non-fluorescent population relative to the number of reads in the uninduced population. One could then plot the distribution of this ratio (or maybe the log ratio), and apply a threshold to that ratio, rather than to threshold the absolute number of reads.</p><p>2. Strategy for assigning allosteric hotspots. Here the authors take the top quartile of residues according to their weighted positional score (that accounts for the number of dead mutations at a position as observed across replicates). By definition, this means that for each homolog, one-quarter of positions (however many were scored) will be called &quot;hotspots&quot;. This seems consistent with what the authors report – for a length of 200 protein, you should then get about 50 hotspots. For RolR, which seems to be a bit longer, they get a few more (57 hotspots). So when they write that &quot;changing the threshold (for assessing allosterically dead) has a modest impact on the overall number of hotspots&quot; it is not really evidence of the robustness of the threshold choice – it is just that they are still taking the top quartile. If I understand correctly, they could use pretty much any strategy they like for assigning allosterically dead/not dead mutants and the number of hotspots would be about the same. That seems like an unusual feature of the analysis choice to me.</p><p>Now, the challenge is what happens when they consider TtgR and RolR. These are the two mutants with the least dynamic range in the assay (25-fold for TtgR, 15-fold for RolR, vs 49-fold for TetR, and 100-fold for MphR). When looking at the data in figure 1 supplement 2, it is clear that TtgR and RolR seem to have fewer allosterically-dead mutations per position. The matrices are overall less &quot;stripey&quot; in the vertical direction than TetR and MphR. So, when they take the top quartile of positions for TtgR and RolR to define allosteric hotspots, the cutoffs are much lower (~0.25-0.3) than for TetR and MphR (~0.8 or so, based on figure 1 supplement 3). As a consequence, what it means to be a hotspot in RolR or TtgR seems to be different than what it means in TetR and MphR. Indeed, my interpretation of the data (based on the heat maps in figure 1 supplement 3) would have been that RolR and TtgR just have fewer hotspots overall. This quartile-based definition of hotspots may explain a number of unusual features for RolR/TtgR, including the fact that: (1) the hotspot distributions are more diffuse across the sequence and structure (Figure 1) (2) that the F-scores are lower (they are less easily distinguished by physical properties), and (3) that the GA-NN model is less compelling. IMO, the reason that the GA-NN does less well for RolR/TtgR is that the training data is labeled improperly… basically, many of the positions they are calling hotspots are just not really hotspots. I feel like this is a far simpler explanation for differences in behavior for RolR and TtgR, rather than the authors' proposal that &quot;… these differences might suggest a higher level of complexity in the allostery mechanism in TtgR and RolR, in which the hotspot resides may contribute to both intra-domain properties and inter-domain coupling&quot;. More generally, I think the choice of top-quartile means that what the authors compare across homologs is not truly apples-to-apples.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.79932.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>Above, I provided my overall comments. Below my aim is to make the concept and method clearer to the community.</p><p>With this aim in mind, there is one thing that I am missing. That is, how the authors define 'hotspots' via DMS. I think that it is important to clarify. This will help the readers.</p></disp-quote><p>Please see a detailed response to Reviewer 3 Question 11.</p><disp-quote content-type="editor-comment"><p>I would also suggest to consider a figure comparing the hotspots in this paper to the hotspots as defined by the propagation pathways from the allosteric binding site to the distal active site. Would mapping these on a structure (e.g. using some server?) work? This could help the reader in visualizing the difference between the new concept suggested here and the 'traditional' definition.</p></disp-quote><p>Thank you for the suggestion. In response to this suggestion, we predicted allosteric hotspots using Ohm webserver. The Ohm webserver is an efficient computational tool that analyzes the propagation of structural perturbation in proteins to identify allostery network and hotspot residues. The method is efficient and the resulting allostery network tends to be robust to small-scale variations in the input structure; it has been successfully applied to map allosteric networks in 20 systems for which high-resolution structures were available and the allosteric sites were known. The new supplementary figure (Figure 1 —figure supplement 7) compares allosteric hotspots between Ohm predictions and our experiments. The overlap between predictions and experiments is modest and involves mostly DNA binding domain residues while the experimental hotspots are distributed across the protein. This highlights the limitation of focusing on the mechanistic model that involves propagation of conformational distortions.</p><p>Changes to manuscript:</p><p>“We also compared the experimental hotspots with predictions made by the Ohm webserver(21). The Ohm webserver is an efficient computational tool that analyzes the propagation of structural perturbation in proteins to identify allostery network and hotspot residues. The overlap between predictions and experiments is modest and involves mostly DNA binding domain residues while the experimental hotspots are distributed across the protein (Figure 1 —figure supplement 7). This highlights the limitation of focusing on the mechanistic model that involves propagation of conformational distortions.”</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>We found the selection of the top 25% scoring residues as &quot;allosteric hotspots&quot; somewhat arbitrary – is it possible to instead select a cutoff based on the resolution (error) of the assay? Surely the top 25% of allosteric hotspots for RolR (which has a limited dynamic range) and TetR (which has a more extensive dynamic range) have very different biophysical effect sizes, and so this is not an apples-to-apples comparison?</p></disp-quote><p>The reviewer makes an important point about classifying the top 25% scoring residues as allosteric hotspots being somewhat arbitrary. The most stringent and arguably unimpeachable definition of an allosteric hotspot is if every non-native mutation results in a dead phenotype (19/20 substitutions). However, few residues meet this strict condition because some mutations are inevitably partially active. This does not mean allosteric hotspots or “lynchpin” residues do not exist in a protein, only that any definition of an allosteric hotspot must be based on some arbitrary activity threshold.</p><p>Yes, the dynamic range of TetR and RolR are different. These differences are reflected in the distribution of weighted scores for each protein. In other words, the distribution of scores for each protein captures the biophysical effect size and the resolution of the assay. Our requirement that the dead mutation is considered only if it is present + and – inducer in both replicates applies to every protein. We do not alter this condition for RolR because it has a lower dynamic range. Therefore, comparing residue scores within a protein, not across proteins, is indeed “apples-to-apples.” We use this consistent definition to assign hotspots.</p><disp-quote content-type="editor-comment"><p>Consider describing the behavior of previously well-characterized aTF mutations in your assay. This would help build confidence that the assay is truly reporting on allosterically dead mutations.</p></disp-quote><p>Previous studies, which employ clonal screening, had identified a limited number of dead mutations (Hillen et al., PMID: 7552732). We found hundreds of dead mutations nearly seven times greater than what was previously known from clonal screens. Earlier studies identified hotspots in the region between LBD and DBD, though not comprehensively. But hotspots in other regions: short motif connecting α7 and α8, dimer interface on α8, and C-terminal end on α9, were previously known. We first reported them in our earlier study (PMID: 32999067) and now in this manuscript.</p><disp-quote content-type="editor-comment"><p>Consider using Fisher's exact test to assign a p-value describing the significance of enrichment/depletion of particular residue types in figure 3A</p></disp-quote><p>In figure 3A, we are merely observing trends in the enrichment and depletion of amino acids for comparison between the “dead” and “no effect” groups. We are careful not to make statistical claims regarding the significance of the enrichment/depletion. However, in figure 3B, we make statistical claims, backed by paired t-tests, on differences in physicochemical properties of both groups.</p><disp-quote content-type="editor-comment"><p>Figure 3C does not have a legend. Also, the figure seems to show K193Y while the main text refers to an inactive Y198 mutation.</p></disp-quote><p>We regret this error. The actual PDB residue number is 199. Our analysis software reassigns residue numbers when missing density is encountered, which changed the numbering to 198. We have changed the figure and main text to 199.</p><p>Changes to manuscript: Changed residue number to 199 in figure and main text.</p><disp-quote content-type="editor-comment"><p>In the methods section, it was unclear how the library was broken up. It seems like it was sequenced in two parts (to cover the entire open reading frame), but was this sequencing two regions of the same cultured and sorted sample? Or was the library broken into sublibraries that were cultured and sorted in smaller batches and then sequenced?</p></disp-quote><p>The full DMS library was broken into 6-7 sub-libraries spanning segments of the proteins. Each sub-library was cultured independently. They were then combined into two groups for sorting and sequencing – segments 1-3 in one group and segments 4-6/7 in the other.</p><p>Changes to manuscript: We added a few sentences under Materials and methods in the “Library synthesis” section.</p><disp-quote content-type="editor-comment"><p>We understand that mutations at ligand-contacting positions were not considered, since these mutations are not allosteric (they instead directly affect ligand binding). How was ligand contacting defined?</p></disp-quote><p>Residues with atom-atom contacts within 5A from the ligand in the crystal structure were considered ligand-binding.</p><disp-quote content-type="editor-comment"><p>Many times, plausible explanations/ideas seemed to be asserted as fact. We feel that these claims either need a citation, more expression of their uncertainty (explaining their lack of evidence in data and the literature), or a more careful explanation:</p><p>1. Abstract, page 2: &quot;We found hotspots to be distributed protein-wide rather than being restricted to &quot;pathways&quot; linking allosteric and active sites as is commonly assumed&quot; It is to our knowledge that it has never been asserted that allosteric hotspots themselves form pathways. Rather it is the assertion (in the prior literature cited by the authors) that allosteric hotspots preferentially contact coevolving networks of residues. In many cases, these co-evolving networks don't just link allosteric to the active site, but often connect other distal surfaces (with no known allosteric function) to the active site – going beyond the idea of a single pathway that the authors seem to imply.</p></disp-quote><p>We agree with the reviewer that some literature, especially those from Ranganathan’s group, highlight the overlap between allosteric hotspots and coevolving residues to form “sectors” that connect protein surface, allosteric and active sites. Nevertheless, many recent publications on protein allostery continue to focus on specific pathways that link allosteric and active sites, while recognizing that multiple pathways may exist in a single system to form a network. The Ohm server (Ref. 21) discussed earlier in response to reviewer 2 question 7 is an example. Recent studies that integrate NMR relaxation measurements, MD simulations and network (community) analysis also tend to focus on such pathways (e.g., Ref. 17, 23). We have added reference to a recent study on allosteric pathways in the Cas9 protein using similar approaches (Ref. 24). Compared to these previous discussions, the hotspot distributions observed in our recent (Ref. 7) and current studies are much broader.</p><p>Changes to manuscript: New reference added to the following paper (Ref. 24).</p><p>“Enhanced specificity mutations perturb allosteric signaling in CRISPR-Cas9”, <italic>eLife</italic>, 2021</p><disp-quote content-type="editor-comment"><p>2. Results page 5: &quot;An aTF mutant that increases the thermodynamic gap between inactive and active states by stabilizing the inactive state will constitutively lock the protein in the inactive allosteric state. We term these &quot;dead&quot; variants. The dead variants are well-folded proteins that can bind to DNA and repress transcription but cannot be induced with ligand.&quot; The authors have made no measurements of protein stability. The only measurements made are cellular GFP levels in the presence and absence of the ligand. This needs a citation or some moderation of language.</p></disp-quote><p>Changes to manuscript: We changed the main text as follows.</p><p>“We designate aTF mutations that constitutively lock the protein in an inactive allosteric state as “dead variants.” This may occur because the mutation stabilizes the inactive state by increasing the thermodynamic gap between inactive and active states. The dead variants are well-folded proteins that bind to DNA and repress transcription but cannot be induced with the ligand.”</p><disp-quote content-type="editor-comment"><p>3. Results page 6: &quot;The evolution of aTFs has occurred through a series of gene duplication events resulting in mixing and matching LBDs and DBDs. &quot; Citation needed.</p></disp-quote><p>Changes to manuscript: We have added citations to two publications, refs. 39 and 40. “Parallel evolution of ligand specificity between LacI/GalR family repressors and periplasmic binding proteins”, <italic>Mol. Biol. Evol</italic>., 2009</p><p>“Duplication of promiscuous transcription factor drives the emergence of a new regulatory network”, <italic>Nature Communications</italic>, 2014</p><disp-quote content-type="editor-comment"><p>4. Results page 6: &quot;Thus, the DBDs likely exist as stand-alone domains that are not allosterically &quot;wired&quot; to the rest of the protein at the residue level but instead respond to large thermodynamic changes (e.g., inducer binding).&quot; The data does not support this conclusion. The data suggests that the DBD is qualitatively depleted for mutations that abolish ligand-based activation while maintaining apo repression.</p></disp-quote><p>Changes to manuscript: We have removed “that are not allosterically wired to the rest of the protein at the residue level” from the manuscript.</p><p>[Editors’ note: further revisions were suggested prior to acceptance, as described below.]</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>Major concerns:</p><p>1. Strategy for assigning allosterically dead mutations. I appreciate that the authors clarified in their revisions that sequencing reads were normalized across replicates, experimental conditions, and proteins. These normalizations make sense to me, and I see that this helps in comparing counts. I also appreciate the authors' statements that &quot;we use cell sorting as a binary classifier&quot; and that by counting the number of dead mutations at a position they safeguard against noise in the data. Their point is well-taken that they seek to categorize mutations as allosterically dead/not dead, rather than using the flow data as a quantitative high-resolution measure. But I am still really stuck on understanding the use of a single threshold of 5 (or 10) to assess if a mutant is present in both the uninduced and induced sorted populations, and therefore assign it as allosterically dead. To illustrate, consider the following scenario. Let's imagine the authors ALSO sequenced the sorted induced fluorescent population (in addition to the induced nonfluorescent population). Now consider two different mutations with the following read distributions:</p><p>a. Mutant &quot;A&quot;: 10 reads in the uninduced population, 10 reads in the induced non-fluorescent population, and 0 reads in the induced fluorescent population.</p><p>b. Mutant &quot;B&quot;: 1000 reads in the uninduced population, 10 reads in the induced non-fluorescent population, and 900 reads in the induced fluorescent population.</p><p>If I understand correctly, both mutants would be classified as allosterically dead according to the authors' method. This makes sense for Mutant A, but for Mutant B…. it looks like it activates, just not completely, or maybe there is some noise in the sorting data. Is it obvious that if there are five or ten reads present it truly isn't noise? (How often do the authors observe &quot;impossible codons&quot; – meaning codons that are not part of their chip-based library – in the induced non-fluorescent population? This might set the noise threshold?) My impression is that the single threshold approach used by the authors may overestimate the number of &quot;dead&quot; mutations. It seems like it would be more correct to consider the ratio of the number of reads in the induced, sorted, non-fluorescent population relative to the number of reads in the uninduced population. One could then plot the distribution of this ratio (or maybe the log ratio), and apply a threshold to that ratio, rather than to threshold the absolute number of reads.</p></disp-quote><p>We did not sequence the induced fluorescent population. Therefore, we cannot answer this question from data. However, there may be a flaw in the reviewer’s line of reasoning. The induced fluorescent population spans 4-6 orders of magnitude of fluorescence, depending on the protein. As a result, drawing the gate for the activity becomes difficult. To detect weak activity, as the reviewer points out, we would have drawn the gate adjacent to the non-fluorescent gate. This would dramatically increase false positives in the screen.</p><disp-quote content-type="editor-comment"><p>2. Strategy for assigning allosteric hotspots. Here the authors take the top quartile of residues according to their weighted positional score (that accounts for the number of dead mutations at a position as observed across replicates). By definition, this means that for each homolog, one-quarter of positions (however many were scored) will be called &quot;hotspots&quot;. This seems consistent with what the authors report – for a length of 200 protein, you should then get about 50 hotspots. For RolR, which seems to be a bit longer, they get a few more (57 hotspots). So when they write that &quot;changing the threshold (for assessing allosterically dead) has a modest impact on the overall number of hotspots&quot; it is not really evidence of the robustness of the threshold choice – it is just that they are still taking the top quartile. If I understand correctly, they could use pretty much any strategy they like for assigning allosterically dead/not dead mutants and the number of hotspots would be about the same. That seems like an unusual feature of the analysis choice to me.</p></disp-quote><p>We may have a philosophical difference with Reviewer 3 on the concept of allosteric hotspots. We get the impression that Reviewer 3 considers the classification of a residue as a hotspot as a yes/no binary such that a residue is either a hotspot or not, and for a given protein, there is a fixed number of hotspots. While this view is not necessarily wrong, it ignores the notion of the magnitude of the contribution of a residue toward allostery, i.e., the ‘effect size’ of the residue. In contrast, we consider the effect size of a residue to fall along a spectrum – some residues are important lynchpins in allosteric signaling, whereas others may have an insignificant effect. Therefore, instead of binary classification, we first sought to rank the residues based on effect size. We then drew a cut-off for effect size on what we consider a hotspot based on the interquartile distribution of scores. In other words, though the hotspot classification is binary, the effect size is continuous.</p><p>To address the question that reviewer 3 posed on changing the read threshold not impacting the number of hotspots. The key takeaway from figure 1 —figure supplement 4 is that changing the read threshold does not change the identity of hotspots falling in the top quartile. This shows that varying the threshold does not impact the effect size of each residue. Our conclusions are, thus, robust to changing thresholds.</p><p>Changes to the manuscript: We have added the following sentence in the manuscript to clarify</p><p>“We note that changing the read threshold does not change the identity of hotspots falling in the top quartile indicating the robustness of our conclusions.”</p><disp-quote content-type="editor-comment"><p>Now, the challenge is what happens when they consider TtgR and RolR. These are the two mutants with the least dynamic range in the assay (25-fold for TtgR, 15-fold for RolR, vs 49-fold for TetR, and 100-fold for MphR). When looking at the data in figure 1 supplement 2, it is clear that TtgR and RolR seem to have fewer allosterically-dead mutations per position. The matrices are overall less &quot;stripey&quot; in the vertical direction than TetR and MphR. So, when they take the top quartile of positions for TtgR and RolR to define allosteric hotspots, the cutoffs are much lower (~0.25-0.3) than for TetR and MphR (~0.8 or so, based on figure 1 supplement 3). Indeed, my interpretation of the data (based on the heat maps in figure 1 supplement 3) would have been that RolR and TtgR just have fewer hotspots overall. This quartile-based definition of hotspots may explain a number of unusual features for RolR/TtgR, including the fact that: (1) the hotspot distributions are more diffuse across the sequence and structure (Figure 1) (2) that the F-scores are lower (they are less easily distinguished by physical properties), and (3) I feel like this is a far simpler explanation for differences in behavior for RolR and TtgR, rather than the authors' proposal that &quot;… these differences might suggest a higher level of complexity in the allostery mechanism in TtgR and RolR, in which the hotspot resides may contribute to both intra-domain properties and inter-domain coupling&quot;. More generally, I think the choice of top-quartile means that what the authors compare across homologs is not truly apples-to-apples.</p></disp-quote><p>Yes, the dynamic range of the four aTFs is different. These differences are reflected in the distribution of weighted scores for each protein and captures the effect size of residues within that protein.</p><p>Let’s assume we could somehow measure the biophysical force associated with allostery disruption (e.g., conformational change) for a mutation. A protein with weaker allosteric activation may have a smaller force associated with a mutation than one with stronger allosteric activation. One can compare the force of disruption within a protein to create a ranked list of effect sizes of residues, but not across proteins. However, patterns of hotspots can be compared across proteins because each hotspot list is internally calibrated for that protein. In summary, the comparison of weighted scores within a protein is apples-to-apples, and the comparison of patterns (not scores) across proteins is also apples-to-apples.</p><disp-quote content-type="editor-comment"><p>As a consequence, what it means to be a hotspot in RolR or TtgR seems to be different than what it means in TetR and MphR.</p></disp-quote><p>Correct, and this is indeed the point. What it means to be a hotspot is a property of the protein, not a fixed threshold applicable to all proteins.</p><disp-quote content-type="editor-comment"><p>That the GA-NN model is less compelling. IMO, the reason that the GA-NN does less well for RolR/TtgR is that the training data is labeled improperly… basically, many of the positions they are calling hotspots are just not really hotspots.</p></disp-quote><p>We disagree with the reviewer that the training data is labeled improperly. The confidence of hotspot assignment increases with increasing dynamic range because there is a clearer separation of dead vs. not dead. Since RolR has lower dynamic range, this may be a factor in the lower F-scores in Figure 4A. This doesn’t mean the ‘data is improperly labeled’; it means the resolution of the assay is lower for RolR compared to TetR.</p><p>Changes to the manuscript: We have added the following sentence in the manuscript to clarify</p><p>“The confidence of hotspot assignment increases with increasing dynamic range because there is a clearer separation of dead vs. not dead. Since RolR and TtgR have lower dynamic ranges, this may be a factor in their lower F-scores.”</p></body></sub-article></article>