<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">94029</article-id><article-id pub-id-type="doi">10.7554/eLife.94029</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.94029.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Structural Biology and Molecular Biophysics</subject></subj-group></article-categories><title-group><article-title>Reliable protein–protein docking with AlphaFold, Rosetta, and replica exchange</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Harmalkar</surname><given-names>Ameya</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6863-9634</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Lyskov</surname><given-names>Sergey</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6380-6712</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Gray</surname><given-names>Jeffrey J</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6380-2324</contrib-id><email>jgray@jhu.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Department of Chemical and Biomolecular Engineering, The Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Program in Molecular Biophysics, The Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Data Science and AI Institute, Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qwgg493</institution-id><institution>Boston University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qwgg493</institution-id><institution>Boston University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>27</day><month>05</month><year>2025</year></pub-date><volume>13</volume><elocation-id>RP94029</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-11-06"><day>06</day><month>11</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-11-25"><day>25</day><month>11</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.07.28.551063"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-02-09"><day>09</day><month>02</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.94029.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2025-02-04"><day>04</day><month>02</month><year>2025</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.94029.2"/></event></pub-history><permissions><copyright-statement>© 2024, Harmalkar et al</copyright-statement><copyright-year>2024</copyright-year><copyright-holder>Harmalkar et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-94029-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-94029-figures-v1.pdf"/><abstract><p>Despite the recent breakthrough of AlphaFold (AF) in the field of protein sequence-to-structure prediction, modeling protein interfaces and predicting protein complex structures remains challenging, especially when there is a significant conformational change in one or both binding partners. Prior studies have demonstrated that AF-multimer (AFm) can predict accurate protein complexes in only up to 43% of cases (Yin et al., 2022). In this work, we combine AF as a structural template generator with a physics-based replica exchange docking algorithm to better sample conformational changes. Using a curated collection of 254 available protein targets with both unbound and bound structures, we first demonstrate that AF confidence measures (pLDDT) can be repurposed for estimating protein flexibility and docking accuracy for multimers. We incorporate these metrics within our ReplicaDock 2.0 protocol to complete a robust in silico pipeline for accurate protein complex structure prediction. AlphaRED (AlphaFold-initiated Replica Exchange Docking) successfully docks failed AF predictions, including 97 failure cases in Docking Benchmark Set 5.5. AlphaRED generates CAPRI acceptable-quality or better predictions for 63% of benchmark targets. Further, on a subset of antigen-antibody targets, which is challenging for AFm (20% success rate), AlphaRED demonstrates a success rate of 43%. This new strategy demonstrates the success possible by integrating deep learning-based architectures trained on evolutionary information with physics-based enhanced sampling. The pipeline is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/Graylab/AlphaRED">https://github.com/Graylab/AlphaRED</ext-link>.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>protein docking</kwd><kwd>protein interactions</kwd><kwd>structure prediction</kwd><kwd>AlphaFold</kwd><kwd>replica exchange</kwd><kwd>RosettaDock</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35-GM141881</award-id><principal-award-recipient><name><surname>Harmalkar</surname><given-names>Ameya</given-names></name><name><surname>Lyskov</surname><given-names>Sergey</given-names></name><name><surname>Gray</surname><given-names>Jeffrey J</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>New protein-protein docking algorithm combining deep learning (AlphaFold2) and physics (Rosetta and enhanced sampling) achieves high success rates.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>In silico protein structure prediction, that is<italic>,</italic> sequence to structure, tackles one of the core questions in structural biology. AlphaFold (AF) (<xref ref-type="bibr" rid="bib21">Jumper et al., 2021</xref>) has brought a paradigm shift in the field of structural biology by intertwining deep learning (DL) tools with evolutionary data to predict single-chain structures with high accuracy. Further, AlphaFold-multimer (AFm) (<xref ref-type="bibr" rid="bib11">Evans et al., 2021</xref>) and related work <xref ref-type="bibr" rid="bib3">Baek et al., 2021</xref>; <xref ref-type="bibr" rid="bib41">Tsaban et al., 2022</xref> have demonstrated the utility of AF to predict protein complexes. The association of proteins to form transient or stable protein complexes often involves binding-induced conformational changes. Capturing conformational dynamics of protein–protein interactions is another grand challenge in structural biology, and many physics-based (computational) approaches have been used to tackle this challenge (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>). Computational tools have sampled the uncharted landscape of protein–protein interactions by emulating kinetic mechanisms such as conformer selection and induced-fit and identifying energetically stable binding states. However, these tools are hampered by the accuracy of the energy functions and the limitations of time and length scales for sampling. In fact, AFm predicted accurate protein complexes in only 43% of cases in one recent study (<xref ref-type="bibr" rid="bib48">Yin et al., 2022</xref>). As the development of DL-based tools have unveiled ground-breaking performance in structure prediction, integration of a biophysical context has potential to strengthen prediction of protein assemblies and binding pathways.</p><p>Blind docking challenges prior to AF, particularly CASP13-CAPRI and CASP14-CAPRI experiments, reported high-quality predictions for only 8% targets (<xref ref-type="bibr" rid="bib24">Lensink et al., 2019</xref>, <xref ref-type="bibr" rid="bib25">Lensink et al., 2021</xref>). With the availability of AF and AFm, the CASP15-CAPRI experiment stood as its first blind assessment for the prediction of protein complexes and higher-order assemblies (<xref ref-type="bibr" rid="bib26">Lensink et al., 2023</xref>). In this round, the docking community relied on AF and AFm for single-structure or complex predictions. Given that AF generates a static three-dimensional structure, it has been unclear whether conformational diversity could be captured by AF. In other terms, given a protein sequence, could AF generate ensembles of structures that include both unbound and bound conformations? Additionally, can AF reveal intrinsic conformational heterogeneity?</p><p>To diversify model complexes generated with AFm in the recent round of CASP15, predictors employed tuning parameters such as dropout (<xref ref-type="bibr" rid="bib45">Wallner, 2022</xref>), higher recycles on inference (<xref ref-type="bibr" rid="bib32">Mirdita et al., 2022</xref>), or modulating the MSA inputs (<xref ref-type="bibr" rid="bib10">Del Alamo et al., 2022</xref>, <xref ref-type="bibr" rid="bib46">Wayment-Steele et al., 2022</xref>) with the amino acid sequence. While these approaches demonstrated the ability to generate broader conformational ensembles, AFm performance still worsens with a higher degree of conformational flexibility between unbound and bound targets (<xref ref-type="bibr" rid="bib48">Yin et al., 2022</xref>). Prediction accuracies especially deteriorated in bound complex regions involving loop motions, concerted motions between domains, rearrangement of secondary structures, or hinge-like domain motions, that is, large-scale conformational changes, which are also challenging for conventional docking methods (<xref ref-type="bibr" rid="bib37">Saldaño et al., 2022</xref>).</p><p>Unlike state-of-the-art docking algorithms, AF’s output models incorporate a residue-specific estimate of prediction accuracy. This suggests a few interesting questions: (1) Do the residue-specific estimates from AF/AFm relate to potential metrics demonstrating conformational flexibility? (2) Can AF/AFm metrics deduce information about docking accuracy? (3) Can we create a docking pipeline for in silico complex structure prediction incorporating AFm to convert sequence to structure to docked complexes?</p><p>Recent work in physics-based docking approaches tested induced-fit docking (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>), large ensembles (<xref ref-type="bibr" rid="bib29">Marze et al., 2018</xref>), and fast Fourier transforms with improved energy functions (<xref ref-type="bibr" rid="bib47">Yan et al., 2020</xref>) to capture conformational changes and better dock protein structures. Coupling temperature replica exchange with induced-fit docking, ReplicaDock 2.0 (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>) achieved successful local docking predictions on 80% of rigid (unbound-to-bound root mean square deviation, RMSD<sub>UB</sub>&lt; 1.1 Å) and 61% medium (1.1 ≤ RMSD<sub>UB</sub> &lt;2.2Å) targets in the Docking Benchmark 5.0 set (<xref ref-type="bibr" rid="bib44">Vreven et al., 2015</xref>). However, like most state-of-the-art physics-based docking methods, ReplicaDock 2.0 performance was limited for highly flexible targets: 33% success rate on targets with RMSD<sub>UB</sub> ≥ 2.2Å. Promisingly, by focusing backbone moves on known mobile residues (i.e<italic>.,</italic> residues that exhibit conformational changes upon binding), ReplicaDock 2.0 sampling substantially improved the docking accuracy. But the flexible residues must first, somehow, be identified. Additionally, physics-based docking is quite slow (6–8 h on a 24-core CPU cluster) compared to recent DL-based docking tools (0.1–10 min on a single NVIDIA GPU). However, docking-specific DL tools such as EquiDock (<xref ref-type="bibr" rid="bib13">Ganea et al., 2021</xref>) and dMASIF (<xref ref-type="bibr" rid="bib40">Sverrisson et al., 2020</xref>) do not allow for protein flexibility, and recent tools like GeoDock (<xref ref-type="bibr" rid="bib8">Chu et al., 2023</xref>) and DockGPT <xref ref-type="bibr" rid="bib31">McPartlon and Xu, 2023</xref> have very limited backbone flexibility. Further, all of these DL docking tools have low success rates on unbound docking targets such as those in Docking Benchmark 5.5 (<xref ref-type="bibr" rid="bib8">Chu et al., 2023</xref>).</p><p>In this work, we combine the features of a top DL approach (AFm; <xref ref-type="bibr" rid="bib11">Evans et al., 2021</xref>) with physics-based docking schemes (ReplicaDock 2.0; <xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>) to systematically dock protein interfaces. The overarching goal is to create a one-stop, fully automated pipeline for simple, reproducible, and accurate modeling of protein complexes. We investigate the aforementioned questions and create a protocol to resolve AFm failures and capture binding-induced conformational changes. We first assess the utility of AFm confidence metrics to detect conformational flexibility and binding site confidence. Next, we feed these metrics and the AFm-generated structural template to ReplicaDock 2.0, creating a pipeline we call AlphaRED (AlphaFold-initiated Replica Exchange Docking). We test AlphaRED’s docking accuracy on a curated set of benchmark targets of bound and unbound protein structures of varying levels of binding-induced conformational change, including antibody–antigen interfaces, which additionally challenge AF2m due to the lack of evolutionary information across the interface (<xref ref-type="bibr" rid="bib49">Yin and Pierce, 2023</xref>; <xref ref-type="bibr" rid="bib36">Ruffolo et al., 2023</xref>). In summary, we to assess the promise of combining the best of DL and biophysical approaches for predicting challenging protein complexes.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Dataset curation</title><p>We curated a dataset for conformational flexibility from the Docking Benchmark Set 5.5 (DB5.5) (<xref ref-type="bibr" rid="bib44">Vreven et al., 2015</xref>), which comprises experimentally characterized (X-ray or cryo-EM) structures of bound protein complexes and their corresponding unbound protein subunits. Each protein target (with unbound and bound structures) is classified based on their unbound-to-bound root-mean-square-deviation (RMSD<sub>UB</sub>) as rigid (RMSD<sub>UB</sub> ≤ 1.2 Å), medium (1.2 Å &lt; RMSD<sub>UB</sub> ≤ 2.2 Å), or difficult (RMSD<sub>UB</sub> ≥ 2.2 Å). Furthermore, due to the poor performance of AF and other predictor groups in predicting antibody–antigen targets in the recent CASP15-CAPRI round <xref ref-type="bibr" rid="bib7">CASP15, 2022</xref>, we identified a subset comprising only antibody–antigen complexes (including single-domain antibodies or nanobodies) extracting all 67 antibody–antigen structures from the DB5.5 (<xref ref-type="bibr" rid="bib15">Guest et al., 2020</xref>, <xref ref-type="bibr" rid="bib44">Vreven et al., 2015</xref>) set. The comprehensive dataset includes 254 protein targets exhibiting binding-induced conformational changes.</p><p>For each protein target, we extracted the amino acid sequences from the bound structure and predicted a corresponding three-dimensional complex structure with the ColabFold implementation (<ext-link ext-link-type="uri" xlink:href="https://github.com/YoshitakaMo/localcolabfold">https://github.com/YoshitakaMo/localcolabfold</ext-link>; <xref ref-type="bibr" rid="bib33">Moriwaki, 2023</xref>) of the AlphaFold multimer v2.3.0 (released in March 2023) for the 254 benchmark targets from DB5.5. Being trained on experimentally characterized structures deposited in the PDB, AF is expected to produce models analogous to the PDB structures. Since most of the benchmark targets in DB5.5 were included in AF training, there would be training bias associated with their predictions (i.e., our measured success rates are an upper bound). However, since both unbound and bound structures exist for the benchmark targets in the PDB, we first investigated whether AFm exhibits any bias toward either unbound or bound forms for the same protein sequence. <xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref> compares the Cα-RMSD of all protein partners (calculated on a per-chain basis) of the AFm predicted complex structures from the bound (B) and unbound (U) crystal structures on a log-log scale (a few AFm predicted models were 20 Å apart from both bound and unbound structures). As evident from Supplementary Fig.S1A, the protein partners from the AFm top-ranked model deviate from both unbound and bound forms and skew more often toward the bound state. Antibody–antigen targets further demonstrate a similar trend, however with fewer targets predicted within sub-angstrom accuracy to the bound form (29.7% for Ab-Ag targets as opposed to 41% for DB5.5). We also calculated the TM-scores (<xref ref-type="bibr" rid="bib51">Zhang and Skolnick, 2004</xref>) of the AFm predicted complex structures with respect to the bound and the unbound crystal structures (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>). As TM-scores reflect a global comparison between structures and are less sensitive to local structural deviations, no strong conclusions could be derived. This is in agreement with our intuition that since both unbound and bound states of proteins will share a similar fold, and AF can predict structures with high TM-scores in most cases, gauging the conformational deviations with TM-scores would be inconclusive.</p></sec><sec id="s2-2"><title>AlphaFold pLDDT provides a predictive confidence measure for backbone flexibility</title><p>AF employs multiple sequence alignments with a multi-track attention-based architecture to predict three-dimensional structures of proteins and complexes. Further, for each structural prediction, it provides a residue-level confidence measure: the predicted local-distance difference test (pLDDT), estimating the agreement between predicted model to an experimental structure based on the Cα LDDT test (‘Methods’). Tunyasuvunakool et al<italic>.</italic> analyzed pLDDT confidence measures for the human proteome demonstrating the correlation between lower pLDDT scores with higher disordered regions in protein structures (<xref ref-type="bibr" rid="bib42">Tunyasuvunakool et al., 2021</xref>). Building on this observation, we evaluated whether there is a correlation between AF pLDDT confidence metric and the experimental metrics of conformational change between unbound and bound structures. In this regard, we compared the computational (AF-pLDDT) and experimental (per-residue RMSD and LDDT) metrics against each other.</p><p>As a reference, we first superimposed the unbound partners over the bound structures and calculated residue-wise Cα deviations to determine the per-residue RMSD<sub>BU</sub> values. LDDT<sub>BU</sub> was measured by calculating the local distance differences in the unbound structure relative to the bound form. These metrics capture the extent of motion in the unbound–bound transitions for each of the protein targets. Next, we compared the per-residue pLDDT score from AFm predicted monomer models with the experimental metrics. <xref ref-type="fig" rid="fig1">Figure 1A and B</xref> shows the results for two representative protein targets: kinase-associated phosphatase in complex with phospho-CDK2 (1FQ1; <xref ref-type="bibr" rid="bib39">Song et al., 2001</xref>) and TGF-β receptor with FKBP12 domain (1B6C; <xref ref-type="bibr" rid="bib19">Huse et al., 1999</xref>). In both cases, pLDDT confidence scores correlate with the experimental measurements of binding: pLDDT decreases as LDDT<sub>BU</sub> decreases and RMSD<sub>BU</sub> increases. This is further illustrated with the AF2 predicted structures of the two targets superimposed over the bound structures (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). In regions of low confidence/pLDDT (highlighted in <italic>red</italic>), the prediction is inaccurate, but higher confidence/pLDDT regions (highlighted in <italic>blue</italic>) have high accuracy of prediction with the bound form. The results for the benchmark set (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplements 1</xref> and <xref ref-type="fig" rid="fig1s2">2</xref>) show similar trends for most targets. The pLDDT, thus, can suggest protein residues that move upon binding.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Comparison of AlphaFold-multimer (AFm) predicted local-distance difference test (pLDDT) with structural metrics.</title><p>(<bold>A</bold>) AlphaFold pLDDT plotted against LDDT<sub>BU</sub>. LDDT<sub>BU</sub> is calculated by comparing the unbound and bound environment for each residue. High scores correlate with high pLDDT (<italic>red</italic>). (<bold>B</bold>) Per-residue root-mean-square-deviation between unbound–bound structures (Per-Residue RMSD<sub>BU</sub>) vs. AlphaFold pLDDT for two example complex structures. Higher RMSDs correlate with lower pLDDT. (<bold>C</bold>) Structures for two targets (PDB ID: 1B6C and 1FQ1) with the experimental bound form (<italic>gray</italic>) and the AlphaFold-multimer predicted model (<italic>red–white–blue</italic> in <bold>A</bold> and <bold>B</bold>). In both cases, the residues with low pLDDT scores (<italic>red</italic>) are the residues with incorrect conformation and more conformational change.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Comparison of AlphaFold-multimer (AFm) predicted local-distance difference test (pLDDT) with structural metrics.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig1-figsupp1-v1.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Comparison of AlphaFold-multimer (AFm) predicted local-distance difference test (pLDDT) with structural metrics.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig1-figsupp2-v1.tif"/></fig></fig-group></sec><sec id="s2-3"><title>Interface-pLDDT correlates with DockQ and discriminates poorly docked structures</title><p>When the prediction accuracy is lower, it is often evident from lower confidence metrics (such as average pLDDT or PAE). However, for AFm complex predictions, the confidence metrics of the overall prediction do not correlate with the accuracy of the docked prediction, that is, even if the complex exhibits higher confidence, the docking interfaces could be incorrect. <xref ref-type="fig" rid="fig2">Figure 2</xref> shows a few examples of failed AFm predictions including rigid (2FJU [<xref ref-type="bibr" rid="bib20">Jezyk et al., 2006</xref>]), medium (5VNW [<xref ref-type="bibr" rid="bib30">McMahon et al., 2018</xref>]), and flexible targets (1IB1 [<xref ref-type="bibr" rid="bib43">Vetter et al., 1999</xref>], 2FJG [<xref ref-type="bibr" rid="bib12">Fuh et al., 2006</xref>]). In all the examples, the AFm model (highlighted in <italic>red</italic> to <italic>blue</italic> based on residue-wise pLDDT) is superimposed over an individual binding partner, and the bound structure is highlighted in <italic>pale green</italic>. AFm models predict the individual subunits (protein partners) accurately in almost all scenarios; however, the docking orientation is incorrect.</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>AlphaFold multimer predictions with reference to bound experimentally characterized structures.</title><p>Four targets with poor DockQ scores and high interface root mean square deviations (RMSDs): (i) activated Rac1 bound to phospholipase Cβ2 (2FJU) – rigid target (RMSD<sub>UB</sub> = 1.04 Å); (ii) nanobody bound to serum albumin (5VNW) – medium target (RMSD<sub>UB</sub> = 1.49 Å); (iii) 14-3-3 zeta Isoform:serotonin N-acetyltransferase complex (1IB1) – difficult target (RMSD<sub>UB</sub> = 2.09 Å)l and (iv) G6 antibody in complex with the VEGF antigen – difficult target (RMSD<sub>UB</sub> = 2.51 Å). Bound structure in <italic>green</italic> and AlphaFold prediction colored by residue-wise pLDDT in <italic>red → blue</italic> (low confidence → high confidence).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig2-v1.tif"/></fig><p>We investigated whether any of the AF predictive metrics could be repurposed for distinguishing native-like binding sites from non-native ones. That is, can one utilize pLDDT or PAE from AFm models to determine whether the predicted docked complex has the accurate binding orientation? Thus, we evaluated accuracy with the DockQ score, the standard metric for docking model quality (<xref ref-type="bibr" rid="bib4">Basu and Wallner, 2016</xref>). DockQ ∈[0,1] combines interface RMSD (Irms), fraction of native-like contacts (<inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>), and ligand-RMSD (Lrms). DockQ scores above 0.23 correspond to models with a CAPRI quality of ‘acceptable’ or higher. As an acceptable quality target implies docked decoys are in the near-native binding region, we chose a binary classification of success with a threshold of DockQ = 0.23. We then tested how well DockQ correlated with several AFm-derived metrics: (1) interface residues: the number of interface residues (atoms of residues on one partner within 8 Å from an atom on another partner); (2) interface contacts: the number of interface contacts between the residues on the interface (Cβ atoms within 5 Å); (3) average pLDDT, determined by averaging over the per-residue LDDT score of the entire protein complex; and (4) interface-pLDDT, determined by averaging the per-residue LDDT score only over the predicted interfacial residues (as identified in case <italic>a</italic>).</p><p><xref ref-type="fig" rid="fig3">Figure 3A</xref> shows the classification accuracy of each of these metrics with a receiver-operating characteristics curve. The interface-pLDDT metric stands out with a higher true positive rate (TPR) with an area under curve (AUC) of 0.86. With interface-pLDDT as a discriminating metric, we tested multiple thresholds to estimate the optimum cutoff for distinguishing near-native structures (defined as an interface-RMSD &lt;4 Å) from the predictions. <xref ref-type="fig" rid="fig3">Figure 3B</xref> summarizes the performance with a confusion matrix for the chosen interface-pLDDT cutoff of 85. 79% of the targets are classified accurately with a precision of 75%, thereby validating the utility of interface-pLDDT as a discriminating metric to rank the docking quality of the AFm complex structure predictions. With newer structure prediction tools such as AlphaFold3 (<xref ref-type="bibr" rid="bib1">Abramson et al., 2024</xref>) and ESM3 (<xref ref-type="bibr" rid="bib18">Hayes et al., 2024</xref>) being released, investigating features that could predict flexible residues or interface site would be valuable as this information may guide local docking. This discrimination is also evident in the highlighted interface residues in <xref ref-type="fig" rid="fig2">Figure 2</xref>, where the AFm predicted models have lower confidence at predicted interfaces (<italic>red</italic>). Finally, we show the trend between DockQ scores and interface-pLDDT for each target in <xref ref-type="fig" rid="fig3">Figure 3C</xref>. The interface-pLDDT threshold of 85 (<italic>dashed line</italic>) thus can serve as the AF-derived metric to distinguish acceptable quality docked predictions from incorrect models.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Interface predicted local-distance difference test (interface-pLDDT) is the best indicator of model docking quality.</title><p>(<bold>A</bold>) Receiver-operator characteristics (ROC) curve as a function of different metrics for the docking dataset (n = 254). Interface residues are defined based on whether atoms of residues on one partner are within 8 Å from atom/s on another partner. Interface-pLDDT is the average pLDDT of interface residues. Avg-pLDDT corresponds to the average pLDDT across all the residues in the predicted model. Interface contacts and interface residues are the counts of the interface contacts and interface residues respectively. Interface-pLDDT has the highest area under curve (AUC) score of 0.86. (<bold>B</bold>) Confusion matrix with an interface-pLDDT threshold between labels predicted false (&lt;85) and true (≥85) and an interface-RMSD threshold between labels actually true (≤4 Å) and false(&gt;4 Å) actual labels. (<bold>C</bold>) Interface-pLDDT versus DockQ for all protein targets in the benchmark set. DockQ is calculated from the predicted AlphaFold structure and the experimental bound structure in the PDB. We fit a sigmoidal curve to this available data.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig3-v1.tif"/></fig></sec><sec id="s2-4"><title>Docking benchmark targets initiated from AlphaFold models improves performance</title><p>With metrics to identify the flexible regions in the protein and the docking accuracy of generated docked models, we next fused AFm with our docking protocol, ReplicaDock 2.0 (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>), to build a protocol for (1) improving on incorrect AF docking predictions and producing alternate, near-native binding models and (2) capturing backbone conformational changes with our induced-fit protocol ReplicaDock2.0 (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>). We named the protocol AlphaRED. AlphaRED uses AFm predicted structures as the primary template, estimates docking accuracy metrics, and initiates global docking or refinement protocols as required.</p><p><xref ref-type="fig" rid="fig4">Figure 4</xref> illustrates this docking pipeline. After AFm predicts a model from the protein sequences, we calculate the interface-pLDDT to determine the docking scheme to follow. If the AFm model is likely to be inaccurate (interface pLDDT &lt; 85), we initiate a global replica exchange docking simulation to explore the protein conformational landscape and identify putative binding sites. On the other hand, if the interface-pLDDT &gt; 85 for the AFm predicted model, the docked complex is likely in the correct binding orientation. This implies the global docking stage of the protocol can be skipped and local docking simulations can be directly initiated from the complex coordinates. Global docking follows an exhaustive, rigid-body search (no backbone moves) between the protein partners to sample putative landscapes in the energy landscape. An unbiased global docking simulation is initiated by randomizing the spatial orientation of protein partners from the input structure. The replica exchange MC routine ReplicaDock 2.0 performs rigid-body rotations (8°) and translations (4 Å). Sampled decoys are clustered from all replicas (based on energies and structural similarity) and the five top clusters are passed along for flexible local docking.</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>AlphaFold-initiated Replica Exchange Docking (AlphaRED) protein docking pipeline.</title><p>Starting with protein sequences of putative complexes, we obtain predicted models from AlphaFold. Each model is accompanied with predicted local-distance difference test (pLDDT) scores, and based on the interface pLDDT we either initiate global rigid-body docking (interface pLDDT &lt; 85), or flexible local docking refinement(interface pLDDT ≥ 85). For global rigid-body docking, the protein partners are first randomized in Cartesian coordinates and then docked with rigid-backbones using temperature replica exchange docking within ReplicaDock2 (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>). Decoy structures are clustered based on energy before flexible local docking refinement. In flexible local docking, we use the directed induced-fit strategy in ReplicaDock2. With mobile residues selected by the AlphaFold residue-wise pLDDT scores (threshold of 80). The protocol moves the backbones with Rosetta’s Backrub or Balanced Kinematic Closure movers. Output structures are refined and top-scoring structures are selected based on interface energy.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig4-v1.tif"/></fig><p>For flexible local docking, we perform aggressive backbone moves (backrub + kinematic closure, ‘Methods’) on candidate encounter complexes (clustered decoys), with fine rigid-body rotations and translations. To narrow conformational sampling, backbone moves are explicitly performed over residues identified as ‘mobile’ based on the per-residue pLDDT metric (residue pLDDT &lt; 80). Unlike ReplicaDock 2.0 that performs induced-fit over putative interfaces, this approach targets backbone motions over these predicted mobile residues, reducing the sampling space. Local docking decoys are further refined for side-chain packing and minimization to obtain the final docked structures (details in ‘Methods’). The methodological advancements and Rosetta movers in AlphaRED are further detailed in the ‘Methods<italic>’</italic> section.</p><p>We investigated AlphaRED’s performance on all 254 benchmark targets (<xref ref-type="fig" rid="fig5">Figure 5</xref>). Ninety-seven targets under the threshold of interface-pLDDT (≤85) were passed to the global docking branch. Targets with interface-pLDDT over 85 proceeded directly to local docking refinement. For all benchmark targets, we compared AlphaRED performance of the top-scoring decoys against initial AFm-predicted complex structures. <xref ref-type="fig" rid="fig5">Figure 5A</xref>. shows the interface-RMSD (Irms) of the AFm and AlphaRED predictions from the bound structure, respectively. The lower Irms values indicate that AlphaRED improves on existing predictions for almost all targets. For targets where AFm prediction is determined to be a failure (interface-pLDDT ≤ 85, <italic>red</italic>), AlphaRED demonstrates a vast improvement in Irms for 93 out of 97 targets. Additionally, for targets where AFm prediction is considered acceptable (interface-pLDDT &gt; 85), local docking slightly improves performance. AlphaRED captures lower interface-RMSDs (under 10 Å) for targets where AFm models dock at binding sites ∼40 Å away. <xref ref-type="fig" rid="fig5">Figure 5B</xref> demonstrates the improvement in recapitulating native-like contacts (<inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mtext>nat</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>) with AlphaRED.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Docking performance.</title><p>Targets with interface-predicted local-distance difference test (Interface-pLDDT) ≤ 85 passed first to global rigid docking (<italic>red</italic>) where targets with interface-pLDDT &gt; 85 proceeded directly to local flexible backbone docking refinement (colored based on their interface-pLDDT scores; in shades of <italic>blue</italic>). (<bold>A</bold>) Interface-RMSD from AlphaFold-multimer (AFm) predicted models (y-axis) in comparison with AlphaFold-initiated Replica Exchange Docking (AlphaRED) models (x-axis). (<bold>B</bold>) Fraction of native-like contacts for models from AFm and AlphaRED, respectively. (<bold>a</bold>) and (<bold>b</bold>) indicate two targets (global and local docking) highlighted in <xref ref-type="fig" rid="fig6">Figure 6</xref>. (<bold>C</bold>) Performance on the subset of antigen–antibody targets in DB5.5. (<bold>D</bold>) DockQ scores for the benchmark targets (DB5) and antibody–antigen complexes.</p><p><supplementary-material id="fig5sdata1"><label>Figure 5—source data 1.</label><caption><title>Performance of AlphaFold-initiated Replica Exchange Docking (AlphaRED) and AlphaFold-multimer (AFm) on Docking Benchmark Set 5.5.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-94029-fig5-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig5-v1.tif"/></fig><p><xref ref-type="fig" rid="fig5">Figure 5C</xref> shows the performance of the subset of antibody–antigen targets in the benchmark. Antibody targets are critical for understanding adaptive immune responses and for the design and engineering of antibody therapeutics (<xref ref-type="bibr" rid="bib9">Chungyoun and Gray, 2023</xref>). However, antibodies have proven challenging for deep learning methods, especially those reliant on multiple sequence alignments, as each antibody evolves in a different organism, and their antigens evolve on a different timescale altogether. Finally, to evaluate docking success rates, we calculate DockQ for top predictions from AFm and AlphaRED, respectively (<xref ref-type="fig" rid="fig5">Figure 5D</xref>). AlphaRED demonstrates a success rate (DockQ &gt; 0.23) of 63% for the benchmark targets. Particularly for Ab-Ag complexes, AFm predicted acceptable or better quality docked structures in only 20% of the 67 targets. In contrast, the AlphaRED pipeline succeeds in 43% of the targets, a significant improvement. Most of the improvements in the success rates are for cases where AFm predictions are worse. For targets with good AFm predictions, AlphaRED refinement results in minimal improvements in docking accuracy.</p><p><xref ref-type="fig" rid="fig6">Figure 6</xref> highlights a global docking (a) and local docking (b) example for targets 2FJU and 5C7X, respectively. Starting from the incorrect AFm prediction (<italic>orange</italic>), AlphaRED samples over the conformational landscape to identify a top-scoring decoy (<italic>blue</italic>) with 2.6 Å Irms from the native (<italic>gray</italic>). <xref ref-type="fig" rid="fig6">Figure 6b</xref> shows the extent of backbone sampling with ReplicaDock 2.0 local docking. The top-scoring decoy (<italic>blue</italic>) samples backbone closer to the bound form, improving model quality and docking accuracy.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Global and local docking performance.</title><p>Docking performance for targets (<bold>a</bold>) activated Rac1 bound to phospholipase Cβ2 (2FJU), and (<bold>b</bold>) neutralizing anti-human antibody Fab fragment in complex with human GM-CSF (5C7X).Starting from the AlphaFold-multimer (AFm) model (<italic>orange</italic>), global docking performance on 2FJU shows native-like binding site (<italic>gray</italic>) and sampled AlphaFold-initiated Replica Exchange Docking (AlphaRED) decoy (<italic>blue</italic>). For local docking, backbone sampling on mobile residues predicted by residue pLDDT (<italic>outlined cartoon</italic>) shows AlphaRED decoy (<italic>blue</italic>) moves backbone toward the bound form (<italic>gray</italic>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig6-v1.tif"/></fig></sec><sec id="s2-5"><title>Evaluation on blind CASP15 targets</title><p>All results presented thus far may be biased by the fact that these benchmark target structures were used in the AFm training. The ultimate challenge for protein structure prediction protocols is to perform successfully over blind targets such as those in CASP (Critical Assessment of protein Structure Prediction) or CAPRI (Critical Assessment of PRotein Interactions) competitions (<xref ref-type="bibr" rid="bib23">Kryshtafovych et al., 2021</xref>, <xref ref-type="bibr" rid="bib25">Lensink et al., 2021</xref>). CASP15 (Summer 2022) provided multiple protein docking targets (<xref ref-type="bibr" rid="bib7">CASP15, 2022</xref>) that were not included in AFm training, allowing an unbiased evaluation of our AlphaRED pipeline (<xref ref-type="bibr" rid="bib26">Lensink et al., 2023</xref>). Thus, we tested the protocol on the five heterodimeric nanobody–antigen complexes where most of the groups performed poorly (<xref ref-type="fig" rid="fig7">Figure 7</xref>).</p><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>AlphaFold-multimer (AFm) and AlphaFold-initiated Replica Exchange Docking (AlphaRED) performance on CASP15 targets.</title><p>Docking performance for CASP targets T205-T209. (Top) T205. Interface score (Rosetta Energy Units, REU) vs. Interface root mean square deviation (RMSD) (Å) for candidate docking structures generated by the AlphaRED docking pipeline. (Top-right) The top-scoring AlphaRED model (<italic>green-blue</italic>) recapitulates the native interface (<italic>gray</italic>) and has an interface RMSD of 2.84 Å. The distinction between the predicted model with respect to the AFm model (<italic>orange</italic>) is evident (bottom) Top-scoring AlphaRED predictions for targets T206, T207, T208, and T209, respectively.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig7-v1.tif"/></fig><p>Since the nanobody–antigen complexes were CASP targets, we did not have unbound structures, rather only the sequences of individual chains. Therefore, for each target, we employed the AlphaRED strategy as described in <xref ref-type="fig" rid="fig4">Figure 4</xref>. All targets predicted with AFm had low interface-pLDDT, thereby demanding global docking. This is unsurprising since the targets were nanobody–antigen targets and their CDRs, particularly CDR H3, are not conserved with a scarcity of co-evolution data with the antigen (<xref ref-type="bibr" rid="bib2">Adolf-Bryfogle et al., 2015</xref>). For representative target T205, our docking strategy improves the performance drastically (interface RMSD 11.4 Å for AFm model to 2.84 Å for AlphaRED) and binds in the correct site. The interface scores versus interface-RMSD plot shows a distinct funnel with low-energy medium-quality structures (<xref ref-type="fig" rid="fig6">Figure 6</xref><italic>,</italic> top). Since the crystal structures are not yet released, the reference structure here is the top model predicted for each category in CASP15 (<xref ref-type="bibr" rid="bib26">Lensink et al., 2023</xref>). For all the targets, <xref ref-type="fig" rid="fig6">Figure 6</xref><italic>,</italic> bottom shows similar improvements for other nanobody–antigen complexes. These cases validate our strategy for blind targets and demonstrate the ability of AlphaRED to serve as a robust pipeline, integrating AF with biophysical attributes to better predict protein complex structures.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>AF has dramatically transformed the field of structural biology and is currently the state-of-the-art method to predict protein structures from sequences, not just for monomers but also for complexes and higher assemblies (<xref ref-type="bibr" rid="bib6">Bryant et al., 2022</xref>). One of the key elements of its success was the ability to mine evolutionary links between amino acids across protein families and determine structural templates. This approach dramatically improves prediction accuracy for monomers as reflected from prior CASP rounds. However, across protein interfaces, the evolutionary constraints can be weak and often skew predictions to inaccurate binding sites. Here, we demonstrated how augmenting the predictions of AF with an energy function-dependent sampling approach reveals better backbone conformational diversity and accurate prediction of protein complex structures. By utilizing the AlphaRED strategy, we show that failure cases in AFm predicted models are improved for all targets (lower Irms for 97 of 254 failed targets) with CAPRI acceptable quality or better models generated for 62% of targets overall (<xref ref-type="fig" rid="fig8">Figure 8</xref>).</p><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Docking prediction success with AlphaFold-multimer (AFm) and AlphaFold-initiated Replica Exchange Docking (AlphaRED).</title><p>Comparison of AFm (<italic>hashed</italic>) and AlphaRED performance for DB5.5 benchmark set. Success rates evaluated based on DockQ criteria: <italic>incorrect</italic>: DockQ &lt; 0.23; <italic>acceptable</italic>: DockQ ∈(0.23,0.49]; <italic>medium</italic>: DockQ ∈(0.49,0.8]; <italic>high</italic>: DockQ ≥0.8. (<bold>A</bold>) Classification based on the scale of flexibility: difficult (35 targets); medium (60 targets); rigid (159 targets). (<bold>B</bold>) Performance on the antibody–antigen complexes (67 targets) and other (non-antibody targets).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-fig8-v1.tif"/></fig><p>First, we showed that AF confidence measures can be repurposed for estimating flexibility and docking accuracy. Interface-pLDDT, an average of the per-residue pLDDT only for the interfacial residues, is a robust metric to determine whether AFm predicted binding interfaces are correct. Additionally, thresholds of per-residue pLDDT can ascertain regions of backbone flexibility upon binding. Thus, AFm predicted models can be used as input structures for ReplicaDock 2.0 guiding the choices of global or local sampling and identifying the mobile protein segments. With DL methods for structure prediction and downstream sampling with a physics-based energy function, one can efficiently explore the protein energy landscape as demonstrated with AlphaRED’s performance on DB5.5. Finally, we evaluated recent CASP15 targets to investigate the extrapolation of this strategy over blind protein targets. CASP15 targets were absent from the training routine of AF and served as blind challenges to determine the efficacy of the protocol. With AlphaRED, we obtained DockQ scores over 0.23 for all five targets, with medium-quality models (DockQ &gt; 0.49) for targets T205, T207, and T208, respectively. AFSample, a top-performing group in CASP15, employed stochastic perturbation with dropout and increased sampling to obtain medium- and high-quality models for these targets. However, AFSample requires GPU simulations to produce ∼240× models with compute time ∼1000× more than the baseline AFm (<xref ref-type="bibr" rid="bib45">Wallner, 2022</xref>). On other hand, we utilized ColabFold (<xref ref-type="bibr" rid="bib32">Mirdita et al., 2022</xref>) to generate 1–5 structures for our docking routine with the baseline version. As opposed to a couple of days on GPU (each GPU node contains up to 48 cores) utilized by AFSample, our docking routine fused with ColabFold uses 5–7 h on our CPU cluster (runs on 1 node, with 24 cores, approximating to ∼100 h of CPU-hours per target). The AlphaRED docking strategy demonstrates a new and better way to predict protein complex structures within feasible compute times.</p><p>This work is particularly impactful for its success rate on antibody–antigen targets. DL promises accurate design and optimization of antibody therapeutics (<xref ref-type="bibr" rid="bib9">Chungyoun and Gray, 2023</xref>), but a lack of fast and accurate docking methods for antibodies prevent high-throughput computational screening. Additionally, this work is impactful because by integrating a physics-based method for refinement, the pipeline can potentially handle post-translationally modified proteins or non-canonical residue types that are not defined in ML approaches like AF.</p><p>With this work, we have built upon the recent advances in structural biology to develop a robust tool for protein docking. We fused DL tools with conventional physics-based sampling tools to develop a pipeline that extracts the best outcomes of each methodology, where DL methods generate accurate, static structures, and physics-based sampling provides diversity and better discrimination. The protein conformational landscape is vast and DL tools such as AF provide a snapshot of relevant local minima that can aid in narrowing down the degrees of freedom in sampling (<xref ref-type="bibr" rid="bib35">Roney and Ovchinnikov, 2022</xref>). With the paradigm shift in computational structural biology toward DL approaches, integrating physics within these models has tremendous potential toward understanding protein dynamics, modulating protein–protein interactions, and downstream applications to protein design.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><sec id="s4-1"><title>Prediction of structures</title><p>For each target in the DB5.5 dataset, we first extracted the corresponding FASTA sequence for the bound complex and then obtained AF predicted models with the ColabFold v1.5.2 (<xref ref-type="bibr" rid="bib34">Ovchinnikov, 2021</xref>) implementation of AF (<xref ref-type="bibr" rid="bib21">Jumper et al., 2021</xref>) and AlphaFold-multimer (v.2.3.0) (<xref ref-type="bibr" rid="bib11">Evans et al., 2021</xref>). Each prediction run was performed without templates, with automatic alignments and the default number of recycles to generate five relaxed predictions. Each AF prediction includes a per-residue pLDDT (predicted LDDT) measurement (<xref ref-type="bibr" rid="bib28">Mariani et al., 2013</xref>), a confidence measure in prediction accuracy, and predicted template alignment (pTM) score (<xref ref-type="bibr" rid="bib51">Zhang and Skolnick, 2004</xref>). The models were structurally compared with the unbound and bound structures (deposited in the PDB) for measuring flexibility, similarity, and accuracy of docking prediction.</p></sec><sec id="s4-2"><title>Metrics for backbone flexibility: RMSD and LDDT</title><p>Structures of proteins deposited in the PDB (<xref ref-type="bibr" rid="bib5">Berman et al., 2000</xref>) provide a static representation of the native state of the protein. However, structural diversity has been captured by experimental techniques to identify different states of a protein in diverse physiological or chemical states, for example, catalysis (<xref ref-type="bibr" rid="bib22">Kingsley and Lill, 2015</xref>), transport (<xref ref-type="bibr" rid="bib14">Gora et al., 2013</xref>), and ligand binding (<xref ref-type="bibr" rid="bib16">Gunasekaran and Nussinov, 2007</xref>). For protein docking challenges in particular, conformational changes are binding-induced, leading to structural differences between unbound and bound structures of protein targets.</p><p>To measure the conformational change in protein structures, we calculated two metrics: Cα RMSD and LDDT (<xref ref-type="bibr" rid="bib28">Mariani et al., 2013</xref>). To get a detailed representation of the intrinsic motion of a protein, we calculated RMSDs at a residue level, that is, per-residue Cα RMSD for each residue of a protein target. The sequences + structures of unbound and bound proteins were aligned and the RMSDs were calculated for the aligned residues. The total sequence lengths were also matched and lingering end-termini residues were trimmed to ensure structural and sequential similarity.</p><p>LDDT is a superimposition-free score that estimates local distance differences in a model relative to a reference structure (<xref ref-type="bibr" rid="bib28">Mariani et al., 2013</xref>). Unlike the global distance test (<xref ref-type="bibr" rid="bib50">Zemla, 2003</xref>) score based on rigid-body superimposition, the LDDT score measures the conserved local interactions in the protein model to the reference. For every residue, it computes the distance between all pair of atoms <inline-formula><mml:math id="inf3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> in both the model and the reference structure (bound) within a threshold (defined as the inclusion radius, generally set to 10 Å). For each pairwise distance in both distance vectors, if the distance is within the threshold, the distance is considered conserved and the fraction of conserved distances is calculated. The final LDDT score is the average of this fraction for the tolerances of 0.5, 1, 2, and 4 Å.</p><p>For a protein structure with <inline-formula><mml:math id="inf4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> number of residues, the overall LDDT score can be given as follows:<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where norm is the normalization factor<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>and <inline-formula><mml:math id="inf5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mtext>score</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the LDDT score for the residue <inline-formula><mml:math id="inf6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> with respect to every other residue <inline-formula><mml:math id="inf7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula><disp-formula id="equ3"><mml:math id="m3"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mtext>score</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>0.25</mml:mn><mml:mo>⋅</mml:mo><mml:mrow><mml:mo maxsize="2.470em" minsize="2.470em">{</mml:mo></mml:mrow><mml:mtext> </mml:mtext><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&lt;</mml:mo><mml:mn>0.5</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mtext> </mml:mtext><mml:mo>+</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&lt;</mml:mo><mml:mn>1.0</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mtext> </mml:mtext><mml:mo>+</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&lt;</mml:mo><mml:mn>2.0</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mtext> </mml:mtext><mml:mo>+</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&lt;</mml:mo><mml:mn>4.0</mml:mn><mml:mo stretchy="false">]</mml:mo><mml:mtext> </mml:mtext><mml:mrow><mml:mo maxsize="2.470em" minsize="2.470em">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p><inline-formula><mml:math id="inf8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> denotes the absolute difference between <inline-formula><mml:math id="inf9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>true</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf10"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>predicted</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> calculated as follows:<disp-formula id="equ4"><label>(3)</label><mml:math id="m4"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>true</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>predicted</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> denote the distances between the Cα coordinates of the <italic>i</italic>th residue and the <italic>j</italic>th residue for the true (reference) and predicted (model) structures, respectively. Let <inline-formula><mml:math id="inf13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:math></inline-formula> represent the <italic>k</italic>th coordinate of the Cα atom in the <italic>i</italic>th residue in the reference (true) structure and predicted structure respectively, such that<disp-formula id="equ5"><label>(4)</label><mml:math id="m5"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:msqrt><mml:mtext> </mml:mtext><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:msqrt></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Finally, the distances to score (<inline-formula><mml:math id="inf15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>) are computed as those pairwise distances within an inclusion radius (cutoff = 10 Å). <inline-formula><mml:math id="inf16"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>m</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:math></inline-formula> is the mask value (1 or 0) indicating if the <underline><italic>j</italic></underline>th coordinate of the Cα atom in the <italic>i</italic>th residue exists in the true structure.<disp-formula id="equ6"><label>(5)</label><mml:math id="m6"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">_</mml:mi></mml:msub><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mrow><mml:mn>1</mml:mn><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&lt;</mml:mo><mml:mrow><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">f</mml:mi></mml:mrow><mml:mo>⋅</mml:mo><mml:msubsup><mml:mi>m</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>⋅</mml:mo><mml:msubsup><mml:mi>m</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>⋅</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>δ</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mi>N</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace linebreak="newline"/><mml:mn>0</mml:mn><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"/></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf17"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>δ</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="normal">K</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">k</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">D</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><p>The advantage of the LDDT measurement lies in the estimation of relative domain orientations in multi-domain proteins or concerted motions (e.g., hinge-like moves in closed and apo proteins). In these cases, the RMSDs would be relatively high for all residues in the mobile domain; however, since the inter-residue distances within the domains are conserved, they would provide an inaccurate depiction of flexibility for the protein. Estimating both RMSDs and LDDT scores allows us to obtain a nuanced perspective of flexibility during protein association based on experimental structures.</p></sec><sec id="s4-3"><title>Developing a pipeline for protein docking</title><p>Using AlphaFold2 as a structural module, we built a pipeline for protein–protein docking to better predict protein complex structures with relatively higher accuracy. As illustrated in <xref ref-type="fig" rid="fig4">Figure 4</xref>, given a sequence of a protein complex, we use the ColabFold implementation of AF2-multimer to obtain a predictive template. An interface-pLDDT filter determines the accuracy of the docking prediction of the top-ranked model from AFm. If the interface-pLDDT ≤ 85, the prediction has lower confidence in the docking orientation, and the protocol initiates a rigid, global docking search with ReplicaDock 2.0. Implementation of ReplicaDock 2.0 (global docking) is similar to the version reported in prior work (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>). Each simulation initiates eight trajectories across three temperature replicas with inverse temperatures set to 1.5<sup>–1</sup> kcal<sup>–1</sup>.mol, 3<sup>–1</sup> kcal<sup>-1</sup>.mol, and 5<sup>–1</sup> kcal<sup>–1</sup>.mol, respectively. Across each replica within each trajectory, rigid-body perturbations (4 Å translations and 8° rotations) are performed for an exhaustive global search. Next, we perform an energy-based clustering of the models to obtain diverse and energetically favorable clusters. Five cluster centers (decoys) are selected and passed to the flexible local docking stage to sample conformational changes.</p><p>On the other hand, if the interface-pLDDT &gt; 85, the binding orientation has higher confidence and the protocol directly performs a flexible local docking simulation skipping the rigid, global docking. In this stage, we perform smaller rigid-body perturbations (1 Å translations and 3° rotations) and aggressive backbone moves using a set of backbone and side-chain movers: Rosetta Backrub (<xref ref-type="bibr" rid="bib38">Smith and Kortemme, 2008</xref>), Balanced Kinematic Closure (BalancedKIC), and Sidechain. The sampling weights are biased such that backbone and side-chain movers are weighted higher than rigid-body moves (3:1 weightage for backbone:rigid-body moves). We perform directed backbone sampling by focusing on predicted mobile residues (per-residue pLDDT &lt; 80). This is automated with the BFactorResidueSelector that selects contiguous sets of residues below the specified pLDDT threshold.</p><p>However, unlike the induced-fit strategy in ReplicaDock (<xref ref-type="bibr" rid="bib17">Harmalkar et al., 2022</xref>), we perform backbone sampling directed only on the mobile residues (with per-residue pLDDT &lt; 80) identified from the AF model. We automate it using the BFactorResidueSelector to select contiguous sets of residues below the specified pLDDT threshold in the prior section. This residue subset is passed along to the backbone movers to sample backbone moves along with small rigid-body moves. Sampled decoyed are then refined, that is, undergo side-chain packing and minimization, to output docked decoys. The best ranked decoys based on interface scores are then identified as the top-scoring structures.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>JJG is an unpaid board member (co-director) of the Rosetta Commons. Under institutional participation agreements between the University of Washington, acting on behalf of the Rosetta Commons, Johns Hopkins University may be entitled to a portion of revenue received on licensing Rosetta software including some methods described in this paper. JJG has a financial interest in Cyrus Biotechnology. Cyrus Biotechnology distributes the Rosetta software, which may include methods described in this paper. These arrangements have been reviewed and approved by the Johns Hopkins University in accordance with its conflict-of-interest policies</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Software, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Resources, Supervision, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-94029-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>AlphaRED utilizes ColabFold for structure prediction with Rosetta-based docking. The source code for AlphaRED is available on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/Graylab/AlphaRED">https://github.com/Graylab/AlphaRED</ext-link>; copy archived at <xref ref-type="bibr" rid="bib27">Lyskov and Harmalkar, 2025</xref>). To ensure ease of availabilty for researchers, we have implemented an online server on the Gray Lab ROSIE server (<ext-link ext-link-type="uri" xlink:href="https://rosie.graylab.jhu.edu/">rosie.graylab.jhu.edu</ext-link>). Users can submit their prediction and docking jobs on <ext-link ext-link-type="uri" xlink:href="https://r2.graylab.jhu.edu/apps/submit/alpha-red">https://r2.graylab.jhu.edu/apps/submit/alpha-red</ext-link>. The server would implement the AlphaRED pipeline and for input sequences would provide docked models with Rosetta energies for further analysis. We expect this implementation to be a great resource for modeling and better understanding protein-protein interactions.</p></sec><ack id="ack"><title>Acknowledgements</title><p>The authors thank Sergey Ovchinnikov and Yoshitaka Moriwaki for ColabFold implementation of AlphaFold.This work was supported by the National Institute of Health through grant R35-GM141881 (all authors).</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abramson</surname><given-names>J</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Dunger</surname><given-names>J</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Willmore</surname><given-names>L</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Bambrick</surname><given-names>J</given-names></name><name><surname>Bodenstein</surname><given-names>SW</given-names></name><name><surname>Evans</surname><given-names>DA</given-names></name><name><surname>Hung</surname><given-names>C-C</given-names></name><name><surname>O’Neill</surname><given-names>M</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Wu</surname><given-names>Z</given-names></name><name><surname>Žemgulytė</surname><given-names>A</given-names></name><name><surname>Arvaniti</surname><given-names>E</given-names></name><name><surname>Beattie</surname><given-names>C</given-names></name><name><surname>Bertolli</surname><given-names>O</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Cherepanov</surname><given-names>A</given-names></name><name><surname>Congreve</surname><given-names>M</given-names></name><name><surname>Cowen-Rivers</surname><given-names>AI</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Fuchs</surname><given-names>FB</given-names></name><name><surname>Gladman</surname><given-names>H</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Khan</surname><given-names>YA</given-names></name><name><surname>Low</surname><given-names>CMR</given-names></name><name><surname>Perlin</surname><given-names>K</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Savy</surname><given-names>P</given-names></name><name><surname>Singh</surname><given-names>S</given-names></name><name><surname>Stecula</surname><given-names>A</given-names></name><name><surname>Thillaisundaram</surname><given-names>A</given-names></name><name><surname>Tong</surname><given-names>C</given-names></name><name><surname>Yakneen</surname><given-names>S</given-names></name><name><surname>Zhong</surname><given-names>ED</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Bapst</surname><given-names>V</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Jaderberg</surname><given-names>M</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name><name><surname>Jumper</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Accurate structure prediction of biomolecular interactions with AlphaFold 3</article-title><source>Nature</source><volume>630</volume><fpage>493</fpage><lpage>500</lpage><pub-id pub-id-type="doi">10.1038/s41586-024-07487-w</pub-id><pub-id pub-id-type="pmid">38718835</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adolf-Bryfogle</surname><given-names>J</given-names></name><name><surname>Xu</surname><given-names>Q</given-names></name><name><surname>North</surname><given-names>B</given-names></name><name><surname>Lehmann</surname><given-names>A</given-names></name><name><surname>Dunbrack</surname><given-names>RL</given-names><suffix>Jr</suffix></name></person-group><year iso-8601-date="2015">2015</year><article-title>PyIgClassify: a database of antibody CDR structural classifications</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>D432</fpage><lpage>D8</lpage><pub-id pub-id-type="doi">10.1093/nar/gku1106</pub-id><pub-id pub-id-type="pmid">25392411</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baek</surname><given-names>M</given-names></name><name><surname>DiMaio</surname><given-names>F</given-names></name><name><surname>Anishchenko</surname><given-names>I</given-names></name><name><surname>Dauparas</surname><given-names>J</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Lee</surname><given-names>GR</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Cong</surname><given-names>Q</given-names></name><name><surname>Kinch</surname><given-names>LN</given-names></name><name><surname>Schaeffer</surname><given-names>RD</given-names></name><name><surname>Millán</surname><given-names>C</given-names></name><name><surname>Park</surname><given-names>H</given-names></name><name><surname>Adams</surname><given-names>C</given-names></name><name><surname>Glassman</surname><given-names>CR</given-names></name><name><surname>DeGiovanni</surname><given-names>A</given-names></name><name><surname>Pereira</surname><given-names>JH</given-names></name><name><surname>Rodrigues</surname><given-names>AV</given-names></name><name><surname>van Dijk</surname><given-names>AA</given-names></name><name><surname>Ebrecht</surname><given-names>AC</given-names></name><name><surname>Opperman</surname><given-names>DJ</given-names></name><name><surname>Sagmeister</surname><given-names>T</given-names></name><name><surname>Buhlheller</surname><given-names>C</given-names></name><name><surname>Pavkov-Keller</surname><given-names>T</given-names></name><name><surname>Rathinaswamy</surname><given-names>MK</given-names></name><name><surname>Dalwadi</surname><given-names>U</given-names></name><name><surname>Yip</surname><given-names>CK</given-names></name><name><surname>Burke</surname><given-names>JE</given-names></name><name><surname>Garcia</surname><given-names>KC</given-names></name><name><surname>Grishin</surname><given-names>NV</given-names></name><name><surname>Adams</surname><given-names>PD</given-names></name><name><surname>Read</surname><given-names>RJ</given-names></name><name><surname>Baker</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Accurate prediction of protein structures and interactions using a three-track neural network</article-title><source>Science</source><volume>373</volume><fpage>871</fpage><lpage>876</lpage><pub-id pub-id-type="doi">10.1126/science.abj8754</pub-id><pub-id pub-id-type="pmid">34282049</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basu</surname><given-names>S</given-names></name><name><surname>Wallner</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>DockQ: a quality measure for protein-protein docking models</article-title><source>PLOS ONE</source><volume>11</volume><elocation-id>e0161879</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0161879</pub-id><pub-id pub-id-type="pmid">27560519</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Berman</surname><given-names>HM</given-names></name><name><surname>Westbrook</surname><given-names>J</given-names></name><name><surname>Feng</surname><given-names>Z</given-names></name><name><surname>Gilliland</surname><given-names>G</given-names></name><name><surname>Bhat</surname><given-names>TN</given-names></name><name><surname>Weissig</surname><given-names>H</given-names></name><name><surname>Shindyalov</surname><given-names>IN</given-names></name><name><surname>Bourne</surname><given-names>PE</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The Protein Data Bank</article-title><source>Nucleic Acids Research</source><volume>28</volume><fpage>235</fpage><lpage>242</lpage><pub-id pub-id-type="doi">10.1093/nar/28.1.235</pub-id><pub-id pub-id-type="pmid">10592235</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bryant</surname><given-names>P</given-names></name><name><surname>Pozzati</surname><given-names>G</given-names></name><name><surname>Elofsson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Improved prediction of protein-protein interactions using AlphaFold2</article-title><source>Nature Communications</source><volume>13</volume><elocation-id>1265</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-022-28865-w</pub-id><pub-id pub-id-type="pmid">35273146</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="web"><person-group person-group-type="author"><collab>CASP15</collab></person-group><year iso-8601-date="2022">2022</year><article-title>15th community wide experiment on the critical assessment of techniques for protein structure prediction</article-title><ext-link ext-link-type="uri" xlink:href="https://predictioncenter.org/casp15/index.cgi">https://predictioncenter.org/casp15/index.cgi</ext-link><date-in-citation iso-8601-date="2022-04-04">April 4, 2022</date-in-citation></element-citation></ref><ref id="bib8"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Chu</surname><given-names>LS</given-names></name><name><surname>Jeffrey</surname><given-names>AR</given-names></name><name><surname>Harmalkar</surname><given-names>A</given-names></name><name><surname>Gray.</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Flexible protein-protein docking with a multi-track iterative transformer</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2023.06.29.547134</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chungyoun</surname><given-names>M</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>AI models for protein design are driving antibody engineering</article-title><source>Current Opinion in Biomedical Engineering</source><volume>28</volume><elocation-id>100473</elocation-id><pub-id pub-id-type="doi">10.1016/j.cobme.2023.100473</pub-id><pub-id pub-id-type="pmid">37484815</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Del Alamo</surname><given-names>D</given-names></name><name><surname>Sala</surname><given-names>D</given-names></name><name><surname>Mchaourab</surname><given-names>HS</given-names></name><name><surname>Meiler</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Sampling alternative conformational states of transporters and receptors with AlphaFold2</article-title><source>eLife</source><volume>11</volume><elocation-id>e75751</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.75751</pub-id><pub-id pub-id-type="pmid">35238773</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>O’Neill</surname><given-names>M</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Antropova</surname><given-names>N</given-names></name><name><surname>Senior</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Blackwell</surname><given-names>S</given-names></name><name><surname>Yim</surname><given-names>J</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Protein complex prediction with AlphaFold-multimer</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.10.04.463034</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuh</surname><given-names>G</given-names></name><name><surname>Wu</surname><given-names>P</given-names></name><name><surname>Liang</surname><given-names>WC</given-names></name><name><surname>Ultsch</surname><given-names>M</given-names></name><name><surname>Lee</surname><given-names>CV</given-names></name><name><surname>Moffat</surname><given-names>B</given-names></name><name><surname>Wiesmann</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Structure-function studies of two synthetic anti-vascular endothelial growth factor fabs and comparison with the avastin fab</article-title><source>Journal of Biological Chemistry</source><volume>281</volume><fpage>6625</fpage><lpage>6631</lpage><pub-id pub-id-type="doi">10.1074/jbc.M507783200</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Ganea</surname><given-names>O</given-names></name><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Bunne</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Independent SE(3)-equivariant models for end-to-end rigid protein docking</article-title><source>arXiv</source><pub-id pub-id-type="doi">10.48550/arXiv.2111.07786</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gora</surname><given-names>A</given-names></name><name><surname>Brezovsky</surname><given-names>J</given-names></name><name><surname>Damborsky</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Gates of enzymes</article-title><source>Chemical Reviews</source><volume>113</volume><fpage>5871</fpage><lpage>5923</lpage><pub-id pub-id-type="doi">10.1021/cr300384w</pub-id><pub-id pub-id-type="pmid">23617803</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guest</surname><given-names>JD</given-names></name><name><surname>Vreven</surname><given-names>T</given-names></name><name><surname>Zhou</surname><given-names>J</given-names></name><name><surname>Moal</surname><given-names>I</given-names></name><name><surname>Jeliazkov</surname><given-names>J</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>An Expanded benchmark for antibody-antigen docking and affinity prediction reveals insights into antibody recognition determinants</article-title><source>SSRN Electronic Journal</source><volume>21</volume><elocation-id>3564997</elocation-id><pub-id pub-id-type="doi">10.2139/ssrn.3564997</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gunasekaran</surname><given-names>K</given-names></name><name><surname>Nussinov</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>How different are structurally flexible and rigid binding sites? sequence and structural features discriminating proteins that do and do not undergo conformational change upon ligand binding</article-title><source>Journal of Molecular Biology</source><volume>365</volume><fpage>257</fpage><lpage>273</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2006.09.062</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harmalkar</surname><given-names>A</given-names></name><name><surname>Mahajan</surname><given-names>SP</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Induced fit with replica exchange improves protein complex structure prediction</article-title><source>PLOS Computational Biology</source><volume>18</volume><elocation-id>e1010124</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1010124</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Hayes</surname><given-names>T</given-names></name><name><surname>Rao</surname><given-names>R</given-names></name><name><surname>Akin</surname><given-names>H</given-names></name><name><surname>Sofroniew</surname><given-names>NJ</given-names></name><name><surname>Oktay</surname><given-names>D</given-names></name><name><surname>Lin</surname><given-names>Z</given-names></name><name><surname>Verkuil</surname><given-names>R</given-names></name><name><surname>Tran</surname><given-names>VQ</given-names></name><name><surname>Deaton</surname><given-names>J</given-names></name><name><surname>Wiggert</surname><given-names>M</given-names></name><name><surname>Badkundri</surname><given-names>R</given-names></name><name><surname>Shafkat</surname><given-names>I</given-names></name><name><surname>Gong</surname><given-names>J</given-names></name><name><surname>Derry</surname><given-names>A</given-names></name><name><surname>Molina</surname><given-names>RS</given-names></name><name><surname>Thomas</surname><given-names>N</given-names></name><name><surname>Khan</surname><given-names>Y</given-names></name><name><surname>Mishra</surname><given-names>C</given-names></name><name><surname>Kim</surname><given-names>C</given-names></name><name><surname>Bartie</surname><given-names>LJ</given-names></name><name><surname>Nemeth</surname><given-names>M</given-names></name><name><surname>Hsu</surname><given-names>PD</given-names></name><name><surname>Sercu</surname><given-names>T</given-names></name><name><surname>Candido</surname><given-names>S</given-names></name><name><surname>Rives</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Simulating 500 Million Years of Evolution with a Language Model</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2024.07.01.600583</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huse</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>YG</given-names></name><name><surname>Massagué</surname><given-names>J</given-names></name><name><surname>Kuriyan</surname><given-names>J</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Crystal structure of the cytoplasmic domain of the type I TGF beta receptor in complex with FKBP12</article-title><source>Cell</source><volume>96</volume><fpage>425</fpage><lpage>436</lpage><pub-id pub-id-type="doi">10.1016/s0092-8674(00)80555-3</pub-id><pub-id pub-id-type="pmid">10025408</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jezyk</surname><given-names>MR</given-names></name><name><surname>Snyder</surname><given-names>JT</given-names></name><name><surname>Gershberg</surname><given-names>S</given-names></name><name><surname>Worthylake</surname><given-names>DK</given-names></name><name><surname>Harden</surname><given-names>TK</given-names></name><name><surname>Sondek</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Crystal structure of Rac1 bound to its effector phospholipase C-β2</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>13</volume><fpage>1135</fpage><lpage>1140</lpage><pub-id pub-id-type="doi">10.1038/nsmb1175</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Back</surname><given-names>T</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Pacholska</surname><given-names>M</given-names></name><name><surname>Berghammer</surname><given-names>T</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Silver</surname><given-names>D</given-names></name><name><surname>Vinyals</surname><given-names>O</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction with AlphaFold</article-title><source>Nature</source><volume>596</volume><fpage>583</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id><pub-id pub-id-type="pmid">34265844</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kingsley</surname><given-names>LJ</given-names></name><name><surname>Lill</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Substrate tunnels in enzymes: structure-function relationships and computational methodology</article-title><source>Proteins</source><volume>83</volume><fpage>599</fpage><lpage>611</lpage><pub-id pub-id-type="doi">10.1002/prot.24772</pub-id><pub-id pub-id-type="pmid">25663659</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kryshtafovych</surname><given-names>A</given-names></name><name><surname>Schwede</surname><given-names>T</given-names></name><name><surname>Topf</surname><given-names>M</given-names></name><name><surname>Fidelis</surname><given-names>K</given-names></name><name><surname>Moult</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Critical assessment of methods of protein structure prediction (CASP)-Round XIV</article-title><source>Proteins</source><volume>89</volume><fpage>1607</fpage><lpage>1617</lpage><pub-id pub-id-type="doi">10.1002/prot.26237</pub-id><pub-id pub-id-type="pmid">34533838</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lensink</surname><given-names>MF</given-names></name><name><surname>Brysbaert</surname><given-names>G</given-names></name><name><surname>Nadzirin</surname><given-names>N</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name><name><surname>Chaleil</surname><given-names>RAG</given-names></name><name><surname>Gerguri</surname><given-names>T</given-names></name><name><surname>Bates</surname><given-names>PA</given-names></name><name><surname>Laine</surname><given-names>E</given-names></name><name><surname>Carbone</surname><given-names>A</given-names></name><name><surname>Grudinin</surname><given-names>S</given-names></name><name><surname>Kong</surname><given-names>R</given-names></name><name><surname>Liu</surname><given-names>R-R</given-names></name><name><surname>Xu</surname><given-names>X-M</given-names></name><name><surname>Shi</surname><given-names>H</given-names></name><name><surname>Chang</surname><given-names>S</given-names></name><name><surname>Eisenstein</surname><given-names>M</given-names></name><name><surname>Karczynska</surname><given-names>A</given-names></name><name><surname>Czaplewski</surname><given-names>C</given-names></name><name><surname>Lubecka</surname><given-names>E</given-names></name><name><surname>Lipska</surname><given-names>A</given-names></name><name><surname>Krupa</surname><given-names>P</given-names></name><name><surname>Mozolewska</surname><given-names>M</given-names></name><name><surname>Golon</surname><given-names>Ł</given-names></name><name><surname>Samsonov</surname><given-names>S</given-names></name><name><surname>Liwo</surname><given-names>A</given-names></name><name><surname>Crivelli</surname><given-names>S</given-names></name><name><surname>Pagès</surname><given-names>G</given-names></name><name><surname>Karasikov</surname><given-names>M</given-names></name><name><surname>Kadukova</surname><given-names>M</given-names></name><name><surname>Yan</surname><given-names>Y</given-names></name><name><surname>Huang</surname><given-names>S-Y</given-names></name><name><surname>Rosell</surname><given-names>M</given-names></name><name><surname>Rodríguez-Lumbreras</surname><given-names>LA</given-names></name><name><surname>Romero-Durana</surname><given-names>M</given-names></name><name><surname>Díaz-Bueno</surname><given-names>L</given-names></name><name><surname>Fernandez-Recio</surname><given-names>J</given-names></name><name><surname>Christoffer</surname><given-names>C</given-names></name><name><surname>Terashi</surname><given-names>G</given-names></name><name><surname>Shin</surname><given-names>W-H</given-names></name><name><surname>Aderinwale</surname><given-names>T</given-names></name><name><surname>Maddhuri Venkata Subraman</surname><given-names>SR</given-names></name><name><surname>Kihara</surname><given-names>D</given-names></name><name><surname>Kozakov</surname><given-names>D</given-names></name><name><surname>Vajda</surname><given-names>S</given-names></name><name><surname>Porter</surname><given-names>K</given-names></name><name><surname>Padhorny</surname><given-names>D</given-names></name><name><surname>Desta</surname><given-names>I</given-names></name><name><surname>Beglov</surname><given-names>D</given-names></name><name><surname>Ignatov</surname><given-names>M</given-names></name><name><surname>Kotelnikov</surname><given-names>S</given-names></name><name><surname>Moal</surname><given-names>IH</given-names></name><name><surname>Ritchie</surname><given-names>DW</given-names></name><name><surname>Chauvot de Beauchêne</surname><given-names>I</given-names></name><name><surname>Maigret</surname><given-names>B</given-names></name><name><surname>Devignes</surname><given-names>M-D</given-names></name><name><surname>Ruiz Echartea</surname><given-names>ME</given-names></name><name><surname>Barradas-Bautista</surname><given-names>D</given-names></name><name><surname>Cao</surname><given-names>Z</given-names></name><name><surname>Cavallo</surname><given-names>L</given-names></name><name><surname>Oliva</surname><given-names>R</given-names></name><name><surname>Cao</surname><given-names>Y</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Baek</surname><given-names>M</given-names></name><name><surname>Park</surname><given-names>T</given-names></name><name><surname>Woo</surname><given-names>H</given-names></name><name><surname>Seok</surname><given-names>C</given-names></name><name><surname>Braitbard</surname><given-names>M</given-names></name><name><surname>Bitton</surname><given-names>L</given-names></name><name><surname>Scheidman-Duhovny</surname><given-names>D</given-names></name><name><surname>Dapkūnas</surname><given-names>J</given-names></name><name><surname>Olechnovič</surname><given-names>K</given-names></name><name><surname>Venclovas</surname><given-names>Č</given-names></name><name><surname>Kundrotas</surname><given-names>PJ</given-names></name><name><surname>Belkin</surname><given-names>S</given-names></name><name><surname>Chakravarty</surname><given-names>D</given-names></name><name><surname>Badal</surname><given-names>VD</given-names></name><name><surname>Vakser</surname><given-names>IA</given-names></name><name><surname>Vreven</surname><given-names>T</given-names></name><name><surname>Vangaveti</surname><given-names>S</given-names></name><name><surname>Borrman</surname><given-names>T</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name><name><surname>Guest</surname><given-names>JD</given-names></name><name><surname>Gowthaman</surname><given-names>R</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Duan</surname><given-names>R</given-names></name><name><surname>Qiu</surname><given-names>L</given-names></name><name><surname>Hou</surname><given-names>J</given-names></name><name><surname>Ryan Merideth</surname><given-names>B</given-names></name><name><surname>Ma</surname><given-names>Z</given-names></name><name><surname>Cheng</surname><given-names>J</given-names></name><name><surname>Zou</surname><given-names>X</given-names></name><name><surname>Koukos</surname><given-names>PI</given-names></name><name><surname>Roel-Touris</surname><given-names>J</given-names></name><name><surname>Ambrosetti</surname><given-names>F</given-names></name><name><surname>Geng</surname><given-names>C</given-names></name><name><surname>Schaarschmidt</surname><given-names>J</given-names></name><name><surname>Trellet</surname><given-names>ME</given-names></name><name><surname>Melquiond</surname><given-names>ASJ</given-names></name><name><surname>Xue</surname><given-names>L</given-names></name><name><surname>Jiménez-García</surname><given-names>B</given-names></name><name><surname>van Noort</surname><given-names>CW</given-names></name><name><surname>Honorato</surname><given-names>RV</given-names></name><name><surname>Bonvin</surname><given-names>AMJJ</given-names></name><name><surname>Wodak</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Blind prediction of homo- and hetero-protein complexes: The CASP13-CAPRI experiment</article-title><source>Proteins</source><volume>87</volume><fpage>1200</fpage><lpage>1221</lpage><pub-id pub-id-type="doi">10.1002/prot.25838</pub-id><pub-id pub-id-type="pmid">31612567</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lensink</surname><given-names>MF</given-names></name><name><surname>Brysbaert</surname><given-names>G</given-names></name><name><surname>Mauri</surname><given-names>T</given-names></name><name><surname>Nadzirin</surname><given-names>N</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name><name><surname>Chaleil</surname><given-names>RAG</given-names></name><name><surname>Clarence</surname><given-names>T</given-names></name><name><surname>Bates</surname><given-names>PA</given-names></name><name><surname>Kong</surname><given-names>R</given-names></name><name><surname>Liu</surname><given-names>B</given-names></name><name><surname>Yang</surname><given-names>G</given-names></name><name><surname>Liu</surname><given-names>M</given-names></name><name><surname>Shi</surname><given-names>H</given-names></name><name><surname>Lu</surname><given-names>X</given-names></name><name><surname>Chang</surname><given-names>S</given-names></name><name><surname>Roy</surname><given-names>RS</given-names></name><name><surname>Quadir</surname><given-names>F</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Cheng</surname><given-names>J</given-names></name><name><surname>Antoniak</surname><given-names>A</given-names></name><name><surname>Czaplewski</surname><given-names>C</given-names></name><name><surname>Giełdoń</surname><given-names>A</given-names></name><name><surname>Kogut</surname><given-names>M</given-names></name><name><surname>Lipska</surname><given-names>AG</given-names></name><name><surname>Liwo</surname><given-names>A</given-names></name><name><surname>Lubecka</surname><given-names>EA</given-names></name><name><surname>Maszota-Zieleniak</surname><given-names>M</given-names></name><name><surname>Sieradzan</surname><given-names>AK</given-names></name><name><surname>Ślusarz</surname><given-names>R</given-names></name><name><surname>Wesołowski</surname><given-names>PA</given-names></name><name><surname>Zięba</surname><given-names>K</given-names></name><name><surname>Del Carpio Muñoz</surname><given-names>CA</given-names></name><name><surname>Ichiishi</surname><given-names>E</given-names></name><name><surname>Harmalkar</surname><given-names>A</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name><name><surname>Bonvin</surname><given-names>AMJJ</given-names></name><name><surname>Ambrosetti</surname><given-names>F</given-names></name><name><surname>Vargas Honorato</surname><given-names>R</given-names></name><name><surname>Jandova</surname><given-names>Z</given-names></name><name><surname>Jiménez-García</surname><given-names>B</given-names></name><name><surname>Koukos</surname><given-names>PI</given-names></name><name><surname>Van Keulen</surname><given-names>S</given-names></name><name><surname>Van Noort</surname><given-names>CW</given-names></name><name><surname>Réau</surname><given-names>M</given-names></name><name><surname>Roel-Touris</surname><given-names>J</given-names></name><name><surname>Kotelnikov</surname><given-names>S</given-names></name><name><surname>Padhorny</surname><given-names>D</given-names></name><name><surname>Porter</surname><given-names>KA</given-names></name><name><surname>Alekseenko</surname><given-names>A</given-names></name><name><surname>Ignatov</surname><given-names>M</given-names></name><name><surname>Desta</surname><given-names>I</given-names></name><name><surname>Ashizawa</surname><given-names>R</given-names></name><name><surname>Sun</surname><given-names>Z</given-names></name><name><surname>Ghani</surname><given-names>U</given-names></name><name><surname>Hashemi</surname><given-names>N</given-names></name><name><surname>Vajda</surname><given-names>S</given-names></name><name><surname>Kozakov</surname><given-names>D</given-names></name><name><surname>Rosell</surname><given-names>M</given-names></name><name><surname>Rodríguez-Lumbreras</surname><given-names>LA</given-names></name><name><surname>Fernandez-Recio</surname><given-names>J</given-names></name><name><surname>Karczynska</surname><given-names>A</given-names></name><name><surname>Grudinin</surname><given-names>S</given-names></name><name><surname>Yan</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Lin</surname><given-names>P</given-names></name><name><surname>Huang</surname><given-names>S-Y</given-names></name><name><surname>Christoffer</surname><given-names>C</given-names></name><name><surname>Terashi</surname><given-names>G</given-names></name><name><surname>Verburgt</surname><given-names>J</given-names></name><name><surname>Sarkar</surname><given-names>D</given-names></name><name><surname>Aderinwale</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Kihara</surname><given-names>D</given-names></name><name><surname>Nakamura</surname><given-names>T</given-names></name><name><surname>Hanazono</surname><given-names>Y</given-names></name><name><surname>Gowthaman</surname><given-names>R</given-names></name><name><surname>Guest</surname><given-names>JD</given-names></name><name><surname>Yin</surname><given-names>R</given-names></name><name><surname>Taherzadeh</surname><given-names>G</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name><name><surname>Barradas-Bautista</surname><given-names>D</given-names></name><name><surname>Cao</surname><given-names>Z</given-names></name><name><surname>Cavallo</surname><given-names>L</given-names></name><name><surname>Oliva</surname><given-names>R</given-names></name><name><surname>Sun</surname><given-names>Y</given-names></name><name><surname>Zhu</surname><given-names>S</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Park</surname><given-names>T</given-names></name><name><surname>Woo</surname><given-names>H</given-names></name><name><surname>Yang</surname><given-names>J</given-names></name><name><surname>Kwon</surname><given-names>S</given-names></name><name><surname>Won</surname><given-names>J</given-names></name><name><surname>Seok</surname><given-names>C</given-names></name><name><surname>Kiyota</surname><given-names>Y</given-names></name><name><surname>Kobayashi</surname><given-names>S</given-names></name><name><surname>Harada</surname><given-names>Y</given-names></name><name><surname>Takeda-Shitaka</surname><given-names>M</given-names></name><name><surname>Kundrotas</surname><given-names>PJ</given-names></name><name><surname>Singh</surname><given-names>A</given-names></name><name><surname>Vakser</surname><given-names>IA</given-names></name><name><surname>Dapkūnas</surname><given-names>J</given-names></name><name><surname>Olechnovič</surname><given-names>K</given-names></name><name><surname>Venclovas</surname><given-names>Č</given-names></name><name><surname>Duan</surname><given-names>R</given-names></name><name><surname>Qiu</surname><given-names>L</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Zhang</surname><given-names>S</given-names></name><name><surname>Zou</surname><given-names>X</given-names></name><name><surname>Wodak</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Prediction of protein assemblies, the next frontier: The CASP14-CAPRI experiment</article-title><source>Proteins</source><volume>89</volume><fpage>1800</fpage><lpage>1823</lpage><pub-id pub-id-type="doi">10.1002/prot.26222</pub-id><pub-id pub-id-type="pmid">34453465</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lensink</surname><given-names>MF</given-names></name><name><surname>Brysbaert</surname><given-names>G</given-names></name><name><surname>Raouraoua</surname><given-names>N</given-names></name><name><surname>Bates</surname><given-names>PA</given-names></name><name><surname>Giulini</surname><given-names>M</given-names></name><name><surname>Honorato</surname><given-names>RV</given-names></name><name><surname>van Noort</surname><given-names>C</given-names></name><name><surname>Teixeira</surname><given-names>JMC</given-names></name><name><surname>Bonvin</surname><given-names>AMJJ</given-names></name><name><surname>Kong</surname><given-names>R</given-names></name><name><surname>Shi</surname><given-names>H</given-names></name><name><surname>Lu</surname><given-names>X</given-names></name><name><surname>Chang</surname><given-names>S</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Guo</surname><given-names>Z</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Morehead</surname><given-names>A</given-names></name><name><surname>Roy</surname><given-names>RS</given-names></name><name><surname>Wu</surname><given-names>T</given-names></name><name><surname>Giri</surname><given-names>N</given-names></name><name><surname>Quadir</surname><given-names>F</given-names></name><name><surname>Chen</surname><given-names>C</given-names></name><name><surname>Cheng</surname><given-names>J</given-names></name><name><surname>Del Carpio</surname><given-names>CA</given-names></name><name><surname>Ichiishi</surname><given-names>E</given-names></name><name><surname>Rodriguez-Lumbreras</surname><given-names>LA</given-names></name><name><surname>Fernandez-Recio</surname><given-names>J</given-names></name><name><surname>Harmalkar</surname><given-names>A</given-names></name><name><surname>Chu</surname><given-names>L-S</given-names></name><name><surname>Canner</surname><given-names>S</given-names></name><name><surname>Smanta</surname><given-names>R</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Lin</surname><given-names>P</given-names></name><name><surname>He</surname><given-names>J</given-names></name><name><surname>Tao</surname><given-names>H</given-names></name><name><surname>Huang</surname><given-names>S-Y</given-names></name><name><surname>Roel-Touris</surname><given-names>J</given-names></name><name><surname>Jimenez-Garcia</surname><given-names>B</given-names></name><name><surname>Christoffer</surname><given-names>CW</given-names></name><name><surname>Jain</surname><given-names>AJ</given-names></name><name><surname>Kagaya</surname><given-names>Y</given-names></name><name><surname>Kannan</surname><given-names>H</given-names></name><name><surname>Nakamura</surname><given-names>T</given-names></name><name><surname>Terashi</surname><given-names>G</given-names></name><name><surname>Verburgt</surname><given-names>JC</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Fujuta</surname><given-names>H</given-names></name><name><surname>Sekijima</surname><given-names>M</given-names></name><name><surname>Kihara</surname><given-names>D</given-names></name><name><surname>Khan</surname><given-names>O</given-names></name><name><surname>Kotelnikov</surname><given-names>S</given-names></name><name><surname>Ghani</surname><given-names>U</given-names></name><name><surname>Padhorny</surname><given-names>D</given-names></name><name><surname>Beglov</surname><given-names>D</given-names></name><name><surname>Vajda</surname><given-names>S</given-names></name><name><surname>Kozakov</surname><given-names>D</given-names></name><name><surname>Negi</surname><given-names>SS</given-names></name><name><surname>Ricciardelli</surname><given-names>T</given-names></name><name><surname>Barradas-Bautista</surname><given-names>D</given-names></name><name><surname>Cao</surname><given-names>Z</given-names></name><name><surname>Chawla</surname><given-names>M</given-names></name><name><surname>Cavallo</surname><given-names>L</given-names></name><name><surname>Oliva</surname><given-names>R</given-names></name><name><surname>Yin</surname><given-names>R</given-names></name><name><surname>Cheung</surname><given-names>M</given-names></name><name><surname>Guest</surname><given-names>JD</given-names></name><name><surname>Lee</surname><given-names>J</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name><name><surname>Shor</surname><given-names>B</given-names></name><name><surname>Cohen</surname><given-names>T</given-names></name><name><surname>Halfon</surname><given-names>M</given-names></name><name><surname>Schneidman-Duhovny</surname><given-names>D</given-names></name><name><surname>Zhu</surname><given-names>S</given-names></name><name><surname>Yin</surname><given-names>R</given-names></name><name><surname>Sun</surname><given-names>Y</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Maszota-Zieleniak</surname><given-names>M</given-names></name><name><surname>Bojarski</surname><given-names>KK</given-names></name><name><surname>Lubecka</surname><given-names>EA</given-names></name><name><surname>Marcisz</surname><given-names>M</given-names></name><name><surname>Danielsson</surname><given-names>A</given-names></name><name><surname>Dziadek</surname><given-names>L</given-names></name><name><surname>Gaardlos</surname><given-names>M</given-names></name><name><surname>Gieldon</surname><given-names>A</given-names></name><name><surname>Liwo</surname><given-names>A</given-names></name><name><surname>Samsonov</surname><given-names>SA</given-names></name><name><surname>Slusarz</surname><given-names>R</given-names></name><name><surname>Zieba</surname><given-names>K</given-names></name><name><surname>Sieradzan</surname><given-names>AK</given-names></name><name><surname>Czaplewski</surname><given-names>C</given-names></name><name><surname>Kobayashi</surname><given-names>S</given-names></name><name><surname>Miyakawa</surname><given-names>Y</given-names></name><name><surname>Kiyota</surname><given-names>Y</given-names></name><name><surname>Takeda-Shitaka</surname><given-names>M</given-names></name><name><surname>Olechnovic</surname><given-names>K</given-names></name><name><surname>Valancauskas</surname><given-names>L</given-names></name><name><surname>Dapkunas</surname><given-names>J</given-names></name><name><surname>Venclovas</surname><given-names>C</given-names></name><name><surname>Wallner</surname><given-names>B</given-names></name><name><surname>Yang</surname><given-names>L</given-names></name><name><surname>Hou</surname><given-names>C</given-names></name><name><surname>He</surname><given-names>X</given-names></name><name><surname>Guo</surname><given-names>S</given-names></name><name><surname>Jiang</surname><given-names>S</given-names></name><name><surname>Ma</surname><given-names>X</given-names></name><name><surname>Duan</surname><given-names>R</given-names></name><name><surname>Qui</surname><given-names>L</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Zou</surname><given-names>X</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name><name><surname>Wodak</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Impact of AlphaFold on structure prediction of protein complexes: The CASP15-CAPRI experiment</article-title><source>Proteins</source><volume>91</volume><fpage>1658</fpage><lpage>1683</lpage><pub-id pub-id-type="doi">10.1002/prot.26609</pub-id><pub-id pub-id-type="pmid">37905971</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Lyskov</surname><given-names>S</given-names></name><name><surname>Harmalkar</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>AlphaRED</data-title><version designator="swh:1:rev:e685615690b92b50d5cfd525a9e9ca948f581bdc">swh:1:rev:e685615690b92b50d5cfd525a9e9ca948f581bdc</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:0f1e7362a9d63940f3b7bf62830f1f8ff3c803b6;origin=https://github.com/Graylab/AlphaRED;visit=swh:1:snp:6e2987bcf0025e07d34a812923690fd5a1a0ad22;anchor=swh:1:rev:e685615690b92b50d5cfd525a9e9ca948f581bdc">https://archive.softwareheritage.org/swh:1:dir:0f1e7362a9d63940f3b7bf62830f1f8ff3c803b6;origin=https://github.com/Graylab/AlphaRED;visit=swh:1:snp:6e2987bcf0025e07d34a812923690fd5a1a0ad22;anchor=swh:1:rev:e685615690b92b50d5cfd525a9e9ca948f581bdc</ext-link></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mariani</surname><given-names>V</given-names></name><name><surname>Biasini</surname><given-names>M</given-names></name><name><surname>Barbato</surname><given-names>A</given-names></name><name><surname>Schwede</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>lDDT: a local superposition-free score for comparing protein structures and models using distance difference tests</article-title><source>Bioinformatics</source><volume>29</volume><fpage>2722</fpage><lpage>2728</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btt473</pub-id><pub-id pub-id-type="pmid">23986568</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Marze</surname><given-names>NA</given-names></name><name><surname>Roy Burman</surname><given-names>SS</given-names></name><name><surname>Sheffler</surname><given-names>W</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Efficient flexible backbone protein-protein docking for challenging targets</article-title><source>Bioinformatics</source><volume>34</volume><fpage>3461</fpage><lpage>3469</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty355</pub-id><pub-id pub-id-type="pmid">29718115</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McMahon</surname><given-names>C</given-names></name><name><surname>Baier</surname><given-names>AS</given-names></name><name><surname>Pascolutti</surname><given-names>R</given-names></name><name><surname>Wegrecki</surname><given-names>M</given-names></name><name><surname>Zheng</surname><given-names>S</given-names></name><name><surname>Ong</surname><given-names>JX</given-names></name><name><surname>Erlandson</surname><given-names>SC</given-names></name><name><surname>Hilger</surname><given-names>D</given-names></name><name><surname>Rasmussen</surname><given-names>SGF</given-names></name><name><surname>Ring</surname><given-names>AM</given-names></name><name><surname>Manglik</surname><given-names>A</given-names></name><name><surname>Kruse</surname><given-names>AC</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Yeast surface display platform for rapid discovery of conformationally selective nanobodies</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>25</volume><fpage>289</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1038/s41594-018-0028-6</pub-id><pub-id pub-id-type="pmid">29434346</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>McPartlon</surname><given-names>M</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Deep learning for flexible and site-specific protein docking and design</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2023.04.01.535079</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mirdita</surname><given-names>M</given-names></name><name><surname>Schütze</surname><given-names>K</given-names></name><name><surname>Moriwaki</surname><given-names>Y</given-names></name><name><surname>Heo</surname><given-names>L</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ColabFold: making protein folding accessible to all</article-title><source>Nature Methods</source><volume>19</volume><fpage>679</fpage><lpage>682</lpage><pub-id pub-id-type="doi">10.1038/s41592-022-01488-1</pub-id><pub-id pub-id-type="pmid">35637307</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Moriwaki</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>LocalColabFold</data-title><version designator="v1.5.1">v1.5.1</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/YoshitakaMo/localcolabfold">https://github.com/YoshitakaMo/localcolabfold</ext-link></element-citation></ref><ref id="bib34"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Ovchinnikov</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>ColabFold online</data-title><version designator="v1.5.2">v1.5.2</version><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/sokrypton/ColabFold">https://github.com/sokrypton/ColabFold</ext-link></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roney</surname><given-names>JP</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>State-of-the-art estimation of protein model accuracy using AlphaFold</article-title><source>Physical Review Letters</source><volume>129</volume><elocation-id>238101</elocation-id><pub-id pub-id-type="doi">10.1103/PhysRevLett.129.238101</pub-id><pub-id pub-id-type="pmid">36563190</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ruffolo</surname><given-names>JA</given-names></name><name><surname>Chu</surname><given-names>LS</given-names></name><name><surname>Mahajan</surname><given-names>SP</given-names></name><name><surname>Gray</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Fast, accurate antibody structure prediction from deep learning on massive set of natural antibodies</article-title><source>Nature Communications</source><volume>14</volume><elocation-id>2389</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-023-38063-x</pub-id><pub-id pub-id-type="pmid">37185622</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Saldaño</surname><given-names>T</given-names></name><name><surname>Escobedo</surname><given-names>N</given-names></name><name><surname>Marchetti</surname><given-names>J</given-names></name><name><surname>Zea</surname><given-names>DJ</given-names></name><name><surname>Mac Donagh</surname><given-names>J</given-names></name><name><surname>Velez Rueda</surname><given-names>AJ</given-names></name><name><surname>Gonik</surname><given-names>E</given-names></name><name><surname>García Melani</surname><given-names>A</given-names></name><name><surname>Novomisky Nechcoff</surname><given-names>J</given-names></name><name><surname>Salas</surname><given-names>MN</given-names></name><name><surname>Peters</surname><given-names>T</given-names></name><name><surname>Demitroff</surname><given-names>N</given-names></name><name><surname>Fernandez Alberti</surname><given-names>S</given-names></name><name><surname>Palopoli</surname><given-names>N</given-names></name><name><surname>Fornasari</surname><given-names>MS</given-names></name><name><surname>Parisi</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Impact of protein conformational diversity on AlphaFold predictions</article-title><source>Bioinformatics</source><volume>38</volume><fpage>2742</fpage><lpage>2748</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btac202</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>CA</given-names></name><name><surname>Kortemme</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Backrub-like backbone simulation recapitulates natural protein conformational variability and improves mutant side-chain prediction</article-title><source>Journal of Molecular Biology</source><volume>380</volume><fpage>742</fpage><lpage>756</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2008.05.023</pub-id><pub-id pub-id-type="pmid">18547585</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Song</surname><given-names>H</given-names></name><name><surname>Hanlon</surname><given-names>N</given-names></name><name><surname>Brown</surname><given-names>NR</given-names></name><name><surname>Noble</surname><given-names>MEM</given-names></name><name><surname>Johnson</surname><given-names>LN</given-names></name><name><surname>Barford</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Phosphoprotein–protein interactions revealed by the crystal structure of kinase-associated phosphatase in complex with phosphoCDK2</article-title><source>Molecular Cell</source><volume>7</volume><fpage>615</fpage><lpage>626</lpage><pub-id pub-id-type="doi">10.1016/S1097-2765(01)00208-8</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Sverrisson</surname><given-names>F</given-names></name><name><surname>Feydy</surname><given-names>J</given-names></name><name><surname>Correia</surname><given-names>BE</given-names></name><name><surname>Bronstein</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Fast end-to-end learning on protein surfaces</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.12.28.424589</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tsaban</surname><given-names>T</given-names></name><name><surname>Varga</surname><given-names>JK</given-names></name><name><surname>Avraham</surname><given-names>O</given-names></name><name><surname>Ben-Aharon</surname><given-names>Z</given-names></name><name><surname>Khramushin</surname><given-names>A</given-names></name><name><surname>Schueler-Furman</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Harnessing protein folding neural networks for peptide-protein docking</article-title><source>Nature Communications</source><volume>13</volume><elocation-id>176</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-021-27838-9</pub-id><pub-id pub-id-type="pmid">35013344</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Wu</surname><given-names>Z</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Laydon</surname><given-names>A</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name><name><surname>Kleywegt</surname><given-names>GJ</given-names></name><name><surname>Bateman</surname><given-names>A</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction for the human proteome</article-title><source>Nature</source><volume>596</volume><fpage>590</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03828-1</pub-id><pub-id pub-id-type="pmid">34293799</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vetter</surname><given-names>IR</given-names></name><name><surname>Arndt</surname><given-names>A</given-names></name><name><surname>Kutay</surname><given-names>U</given-names></name><name><surname>Görlich</surname><given-names>D</given-names></name><name><surname>Wittinghofer</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Structural view of the Ran-Importin beta interaction at 2.3 A resolution</article-title><source>Cell</source><volume>97</volume><fpage>635</fpage><lpage>646</lpage><pub-id pub-id-type="doi">10.1016/s0092-8674(00)80774-6</pub-id><pub-id pub-id-type="pmid">10367892</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vreven</surname><given-names>T</given-names></name><name><surname>Moal</surname><given-names>IH</given-names></name><name><surname>Vangone</surname><given-names>A</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name><name><surname>Kastritis</surname><given-names>PL</given-names></name><name><surname>Torchala</surname><given-names>M</given-names></name><name><surname>Chaleil</surname><given-names>R</given-names></name><name><surname>Jiménez-García</surname><given-names>B</given-names></name><name><surname>Bates</surname><given-names>PA</given-names></name><name><surname>Fernandez-Recio</surname><given-names>J</given-names></name><name><surname>Bonvin</surname><given-names>A</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Updates to the integrated protein-protein interaction benchmarks: docking benchmark version 5 and affinity Benchmark Version 2</article-title><source>Journal of Molecular Biology</source><volume>427</volume><fpage>3031</fpage><lpage>3041</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2015.07.016</pub-id><pub-id pub-id-type="pmid">26231283</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Wallner</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>AFsample: improving multimer prediction with alphafold using aggressive sampling</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.12.20.521205</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Wayment-Steele</surname><given-names>HK</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Colwell</surname><given-names>L</given-names></name><name><surname>Kern</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Prediction of multiple conformational states by combining sequence clustering with AlphaFold2</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.10.17.512570</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname><given-names>Y</given-names></name><name><surname>Tao</surname><given-names>H</given-names></name><name><surname>He</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>S-Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The HDOCK server for integrated protein-protein docking</article-title><source>Nature Protocols</source><volume>15</volume><fpage>1829</fpage><lpage>1852</lpage><pub-id pub-id-type="doi">10.1038/s41596-020-0312-x</pub-id><pub-id pub-id-type="pmid">32269383</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname><given-names>R</given-names></name><name><surname>Feng</surname><given-names>BY</given-names></name><name><surname>Varshney</surname><given-names>A</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Benchmarking AlphaFold for protein complex modeling reveals accuracy determinants</article-title><source>Protein Science</source><volume>31</volume><elocation-id>e4379</elocation-id><pub-id pub-id-type="doi">10.1002/pro.4379</pub-id><pub-id pub-id-type="pmid">35900023</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Yin</surname><given-names>R</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Evaluation of AlphaFold antibody-antigen modeling with implications for improving predictive accuracy</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2023.07.05.547832</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zemla</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>LGA: A method for finding 3D similarities in protein structures</article-title><source>Nucleic Acids Research</source><volume>31</volume><fpage>3370</fpage><lpage>3374</lpage><pub-id pub-id-type="doi">10.1093/nar/gkg571</pub-id><pub-id pub-id-type="pmid">12824330</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Skolnick</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Scoring function for automated assessment of protein structure template quality</article-title><source>Proteins</source><volume>57</volume><fpage>702</fpage><lpage>710</lpage><pub-id pub-id-type="doi">10.1002/prot.20264</pub-id><pub-id pub-id-type="pmid">15476259</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s8"><title>Supplementary methods</title><sec sec-type="appendix" id="s8-1"><title>Benchmark sets</title><p>The benchmark set was curated from Docking Benchmark 5.5 (<xref ref-type="bibr" rid="bib44">Vreven et al., 2015</xref>), with targets classified based on their extent of flexibility, that is, rigid, medium, and difficult. We also curated a subset of only antigen–antibody/nanobody targets from the overall set. For each target in the benchmark set, a FASTA sequence was obtained with individual chains separated by a colon (:) indicating chain break. This sequence was used for AFm structure prediction. For target 1N2C, AFm could not generate a structural prediction owing to the longer sequence length and is excluded from the benchmark. For each structural prediction that was generated, comparisons were made to its corresponding bound and unbound forms as obtained from the Docking Benchmark 5.5.</p></sec><sec sec-type="appendix" id="s8-2"><title>Metrics and evaluation</title><p>The docking performance was evaluated based on interface RMSD (Irms), fraction of native-like contacts (<inline-formula><mml:math id="inf18"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>), CAPRI quality, and DockQ scores. These metrics are defined as follows:</p><list list-type="simple" id="list1"><list-item><p>Interface RMSD (Irms): The RMSD of all atoms on the interface in a docked protein structure relative to a reference structure (native). Interface residues are defined as all 24 amino acid residues within 10 Å of any residue on the binding partner.</p></list-item><list-item><p>Fraction of native-like contacts (<inline-formula><mml:math id="inf19"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>): The fraction of native-like contacts recovered in the docked structure relative to the reference structure (native).</p></list-item><list-item><p>CAPRI quality: A CAPRI-based rank calculated on the basis of I-rms, <inline-formula><mml:math id="inf20"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mtext>nat</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, and ligand-RMSD to classify a docked model as incorrect, acceptable, medium, or high-quality prediction.</p></list-item><list-item><p>DockQ Score: Similar to CAPRI quality, the DockQ scores estimates a score (∈[0,1]) estimating the accuracy of the docked complexes. We calculated this score based on the methodology described in <xref ref-type="bibr" rid="bib4">Basu and Wallner, 2016</xref>.</p></list-item><list-item><p>Interface score (Isc): The interface score is analogous to thermodynamic binding energy of protein association. This score is estimated by calculating the total score (Gibbs free energy) of protein complex and then by subtracting individual (monomeric) scores of protein partners in absence of its partner. Mathematically, for proteins A and B forming a complex AB, it can be defined as follows:</p></list-item></list><p><disp-formula id="equ7"><mml:math id="m7"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">B</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">A</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">B</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:math></disp-formula></p></sec></sec><sec sec-type="appendix" id="s9"><title>Supplementary results</title><fig id="app1fig1" position="float"><label>Appendix 1—figure 1.</label><caption><title>Root mean square deviations (RMSDs) of AlphaFold-multimer structures from experimental unbound and bound structures.</title><p>Distribution of the RMSD between the AlphaFold-multimer prediction top-ranked model and the experimental unbound and bound structures. For each target, the protein partners are split into receptor and ligand respectively for comparison. Each symbol represents a category of flexibility (rigid, medium, and flexible). (<bold>A</bold>) Dockground Benchmark set 5.5; (<bold>B</bold>) antibody/nanobody–antigen targets from the benchmark.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-app1-fig1-v1.tif"/></fig><fig id="app1fig2" position="float"><label>Appendix 1—figure 2.</label><caption><title>TM-scores of AlphaFold-multimer structures from experimental unbound and bound structures.</title><p>Distribution of the TM-score between the AlphaFold-multimer prediction top-ranked model and the experimental unbound and bound structures. For each target, the protein partners are split into receptor and ligand respectively for comparison. Each symbol represents a category of flexibility (rigid, medium, and flexible). (<bold>A</bold>) Dockground Benchmark set 5.5; (<bold>B</bold>) antibody/nanobody–antigen targets from the benchmark.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-94029-app1-fig2-v1.tif"/></fig></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.94029.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Boston University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Convincing</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Valuable</kwd></kwd-group></front-stub><body><p>The authors report how a previously published method, ReplicaDock, can be used to improve predictions from AlphaFold-multimer (AFm) for protein docking studies. The level of improvement is modest for cases where AFm is successful; for cases where AFm is not as successful, the improvement is more significant, although the accuracy of prediction is also notably lower. The evidence for the ReplicaDock approach being more predictive than AFm is particularly <bold>convincing</bold> for the antibody–antigen test case. Overall, the study makes a <bold>valuable</bold> contribution by combining data- and physics-driven approaches.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.94029.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>The authors wanted to use AlphaFold-multimer (AFm) predictions to reduce the challenge of physics-based protein-protein docking.</p><p>Strengths:</p><p>They found two features of AFm predictions that are very useful. (1) pLLDT is predictive of flexible residues, which they could target for conformational sampling during docking; (2) the interface-pLLDT score is predictive of the quality of AFm predictions, which allows the authors to decide whether to do local or global docking.</p><p>Weaknesses:</p><p>(1) As admitted by the authors, the AFm predictions for the main dataset are undoubtedly biased because these structures were used for AFm training. Could the authors find a way to assess the extent of this bias?</p><p>(2) For the CASP15 targets where this bias is absent, the presentation was very brief. In particular, I'm interested in seeing how AFm helped with the docking? They may even want to do a direct comparison with docking results w/o the help of AFm.</p><p>Comments on revisions:</p><p>This revision has addressed my previous comments.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.94029.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>In short, this paper uses a previously published method, ReplicaDock to improve predictions from AlphaFold-multimer. The method generated about 25% more acceptable predictions than AFm, but more important is improving an Antibody-antigen set, where more than 50% of the models become improved.</p><p>When looking at the results in more detail, it is clear that for the models where the AFm models are good, the improvement is modest (or not at all). See, for instance, the blue dots in Fig 6. However, in the cases where AFm fails, the improvement is substantial (red dots in Fig 6), but no models reach a very high accuracy (Fnat ~0.5 compared to 0.8 for the good AFm models). So the paper could be summarized by claiming, &quot;We apply ReplicaDock when AFm fails&quot;, instead of trying to sell the paper as an utterly novel pipeline. I must also say that I am surprised by the excellent performance of ReplicaDock - it seems to be a significant step ahead of other (not AlphaFold) docking methods, and from reading the original paper, that was unclear. Having a better benchmark of it alone (without AFm) would be very interesting.</p><p>These results also highlight several questions I try to describe in the weakness section below. In short, they boil down to the fact that the authors must show how good/bad ReplicaDock is at all targets not only the ones where AFm fails. In addition, I have several more technical comments.</p><p>Strengths:</p><p>Impressive increase in performance on AB-AG set (although a small set and no proteins).</p><p>Weaknesses:</p><p>The presentation is a bit hard to follow. The authors mix several measures (Fnat, iRMS, RMSDbound, etc). In addition, it is not always clear what is shown. For instance, in Fig 1, is the RMSD calculated for a single chain or the entire protein? I would suggest that the author replace all these measures with two: TM-score when evaluating the quality of a single chain and DockQ when evaluating the results for docking. This would provide a clearer picture of the performance. This applies to most figures and tables. For instance, Fig 9 could be shown as a distribution of DockQ scores.</p><p>The improvements on the models where AFm is good are minimal (if at all), and it is unclear how global docking would perform on these targets, nor exactly why the plDDT&lt;0.85 cutoff was chosen. To better understand the performance of ReplicaDock, the authors should therefore (i) run global and local docking on all targets and report the results, (ii) report the results if AlphaFold (not multimer) models of the chains were used as input to ReplicaDock (I would assume it is similar). These models can be downloaded from AlphaFoldDB.</p><p>Further, it would be interesting to see if ReplicaDock could be combined with AFsample (or any other model to generate structural diversity) to improve performance further.</p><p>The estimates of computing costs for the AFsample are incorrect (check what is presented in their paper). What are the computational costs for RepliaDock global docking?</p><p>It is unclear strictly what sequences were used as input to the modelling. The authors should use full-length UniProt sequences if that were not done.</p><p>The antibody-antigen dataset is small. It could easily be expanded to thousands of proteins. It would be interesting to know the performance of ReplicaDock on a more extensive set of Antibodies and nanobodies.</p><p>Using pLDDT on the interface region to identify good/bas models is likely suboptimal. It was acceptable (as a part of the score) for AlphaFold-2.0 (monomer), but AFm behaves differently. Here, AFm provides a direct score to evaluate the quality of the interaction (ipTM or Ranking Confidence). The authors should use these to separate good/bad models (for global/local docking), or at least show that these scores are less good than the one they used.</p><p>Comments on revisions:</p><p>The inclusion of the DockQ improved the paper. No further comments.</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.94029.3.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Harmalkar</surname><given-names>Ameya</given-names></name><role specific-use="author">Author</role><aff><institution>Johns Hopkins University</institution><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Lyskov</surname><given-names>Sergey</given-names></name><role specific-use="author">Author</role><aff><institution>Johns Hopkins University</institution><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Gray</surname><given-names>Jeffrey J</given-names></name><role specific-use="author">Author</role><aff><institution>Johns Hopkins University</institution><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Public Review)</bold></p><p>Summary:</p><p>The authors wanted to use AlphaFold-multimer (AFm) predictions to reduce the challenge of physics-based protein-protein docking.</p><p>Strengths:</p><p>They found that two features of AFm predictions are very useful. (1) pLLDT is predictive of flexible residues, which they could target for conformational sampling during docking; (2) the interface-pLLDT score is predictive of the quality of AFm predictions, which allows the authors to decide whether to do local or global docking.</p><p>Weaknesses:</p><p>(1) As admitted by the authors, the AFm predictions for the main dataset are undoubtedly biased because these structures were used for AFm training. Could the authors find a way to assess the extent of this bias?</p></disp-quote><p>Indeed, the AFm training included most of the structures in the DB5 benchmark for its training as many structures (either unbound or bound) were deposited before the training cut-off period. One of the challenges of estimating this bias is the availability of new structures - both bound and unbound deposited after the training cut-off. Estimating the extent of training bias is therefore conditional on these factors and difficult. A few studies have attempted to address this bias (Yin et al, 2022, <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/pro.4379">https://doi.org/10.1002/pro.4379</ext-link>).</p><p>In our study, we assess this bias by comparing the AFm structures to the bound and unbound forms and calculating their Ca RMSDs and TM-scores (new addition). We now elaborate in the Results:Dataset curation section and we have added a figure comparing the TM-scores in the supplement.</p><p>We added a clarifying text and a note about the TM-score calculation in the manuscript as follows:</p><p>“Since most of the benchmark targets in DB5.5 were included in AlphaFold training, there would be training bias associated with their predictions (i.e. our measured success rates are an upper bound).”</p><p>“We also calculated the TM-scores of the AFm predicted complex structures with respect to the bound and the unbound crystal structures (Supplementary Figure S2). As TM-scores reflect a global comparison between structures and are less sensitive to local structural deviations, no strong conclusions could be derived. This is in agreement with our intuition that since both unbound and bound states of proteins will share a similar fold, and AlphaFold can predict structures with high TM-scores in most cases, gauging the conformational deviations with TM-scores would be inconclusive.”</p><disp-quote content-type="editor-comment"><p>(2) For the CASP15 targets where this bias is absent, the presentation was very brief. In particular, it would be interesting to see how AFm helped with the docking. The authors may even want to do a direct comparison with docking results without the help of AFm.</p></disp-quote><p>Unfortunately since this was a CASP-CAPRI round, the structure of the unbound Antigen or the nanobodies was unavailable. Thus we cannot perform a comparison without using AF2 at all since we need a structure prediction tool to produce the unbound nanobody and the nanobody-antigen complex template structure to dock. This has been clarified in the main text for better understanding for the readers.</p><p>“Since the nanobody-antigen complexes were CASP targets, we did not have unbound structures, rather only the sequences of individual chains. Therefore, for each target, we employed the AlphaRED strategy as described in Fig 7.”</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Recommendations For The Authors):</bold></p><p>For suggestions for major improvements, see comments under weaknesses. One additional suggestion: the authors found that pLLDT is predictive of flexible residues. Can they try to find AFm features that are predictive of the interface site? Such information may guide their docking to a local site.</p></disp-quote><p>This is a great idea that we and others have been thinking about considerably. Prior work by Burke et al. (Towards a structurally resolved human protein interaction network) examines AlphaFold’s ability to predict PPIs. For high-confidence predicted models of interacting protein complexes, the authors showed that pDockQ correlated reasonably well with correct protein interactions.</p><p>That being said, binding site identification, particularly in a partner-agnostic fashion, i.e. determining binding patches on a given protein, is an area of on-going research . We hope a future study examines AlphaFold3 or ESM3 specifically for this task.</p><p>“Further, we tested multiple thresholds to estimate the optimum cut-off for distinguishing near-native structures (defined as an interface-RMSD &lt; 4 Å) from the predictions. Figure 3.B summarizes the performance with a confusion matrix for the chosen interface-pLDDT cutoff of 85. 79 % of the targets are classified accurately with a precision of 75%, thereby validating the utility of interface-pLDDT as a discriminating metric to rank the docking quality of the AFm complex structure predictions. With AlphaFold3 and ESM3 being released, investigating features that could predict flexible residues or interface site would be valuable, as this information may guide local docking.”</p><disp-quote content-type="editor-comment"><p>Minor:</p><p>Page 3, lines 73-77, state how many targets were curated from DB5.5.</p></disp-quote><p>We have now clarified this in the manuscript. All 254 targets curated from DB5.5 at the time of this benchmark study.</p><p>“For each protein target, we extracted the amino acid sequences from the bound structure and predicted a corresponding three-dimensional complex structure with the ColabFold implementation of the AlphaFold multimer v2.3.0 (released in March 2023) for the 254 benchmark targets from DB5.5.”</p><disp-quote content-type="editor-comment"><p>In Figure 1, the color used for medium is too difficult to distinguish from the grey color used for rigid.</p></disp-quote><p>We thank you for this suggestion. We have updated the color to olive. Further, based on Reviewer 2’s suggestions, we have moved this plot to the Supplementary.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>Summary:</p><p>In short, this paper uses a previously published method, ReplicaDock, to improve predictions from AlphaFold-multimer. The method generated about 25% more acceptable predictions than AFm, but more important is improving an Antibody-antigen set, where more than 50% of the models become improved.</p><p>When looking at the results in more detail, it is clear that for the models where the AFm models are good, the improvement is modest (or not at all). See, for instance, the blue dots in Figure 6. However, in the cases where AFm fails, the improvement is substantial (red dots in Figure 6), but no models reach a very high accuracy (Fnat ~0.5 compared to 0.8 for the good AFm models). So the paper could be summarized by claiming, &quot;We apply ReplicaDock when AFm fails&quot;, instead of trying to sell the paper as an utterly novel pipeline. I must also say that I am surprised by the excellent performance of ReplicaDock - it seems to be a significant step ahead of other (not AlphaFold) docking methods, and from reading the original paper, that was unclear. Having a better benchmark of it alone (without AFm) would be very interesting.</p></disp-quote><p>We thank the reviewer for highlighting the performance of ReplicaDock. ReplicaDock alone is benchmarked in the original paper (10.1371/journal.pcbi.1010124), with full details on the 2022 version of DB5.5 in the supplement. Indeed ReplicaDock2 achieves the highest reported success rates on flexible docking targets reported in the literature (until this AlphaRED paper!).</p><p>Regarding this statement about “the paper could be summarized…” it might be helpful to give more context. ReplicaDock is a replica exchange Monte Carlo sampling approach for protein docking that incorporates flexibility in an induced-fit fashion. However, the choice of which backbone residues to move is solely dependent on contacts made during each docking trajectory. In the last section of the ReplicaDock paper, we introduced “Directed Induced-fit” where we biased the backbone sampling only towards those residues where we knew the backbone is flexible (this information is obtained because for the benchmark set, we had both unbound and bound structures and hence could cherry-pick the specific residues which are mobile). We agree with the reviewers that AlphaRED is essentially a derivative of ReplicaDock, however, the two major claims that we make in this paper are:</p><p>(1) AlphaFold pLDDT is an effective predictor of backbone flexibility for practical use in docking.</p><p>(2) We can automate the Directed InducedFit approach within ReplicaDock by utilizing this pLDDT information per residue for conformational sampling in protein docking; and in doing so, create a pipeline that would allow us to go from sequence-to-structure-to-complex, specifically capturing conformational changes.</p><p>To conclude these claims, we pose the following questions in the Introduction:</p><p>“(1) Do the residue-specific estimates from AF/AFm relate to potential metrics demonstrating conformational flexibility?</p><p>(2) Can AF/AFm metrics deduce information about docking accuracy?</p><p>(3) Can we create a docking pipeline for in-silico complex structure prediction incorporating AFm to convert sequence-to-structure-to-docked complexes?”</p><p>This work requires a pipeline, the center of which lies in ReplicaDock as a docking method, but has functionalities that were absent in prior work. The goal is also to develop a one-stop shop without manual intervention (a prerequisite for biasing backbone sampling in ReplicaDock) that could be utilized by structural biologists efficiently.</p><p>We clarify this points in the abstract and main text as follows:</p><p>Abstract: “In this work, we combine AlphaFold as a structural template generator with a physics-based replica exchange docking algorithm to better sample conformational changes.”</p><p>Introduction:</p><p>“The overarching goal is to create a one-stop, fully-automated pipeline for simple, reproducible, and accurate modeling of protein complexes. We investigate the aforementioned questions and create a protocol to resolve AFm failures and capture binding-induced conformational changes. We first assess the utility of AFm confidence metrics to detect conformational flexibility and binding site confidence.”</p><disp-quote content-type="editor-comment"><p>These results also highlight several questions I try to describe in the weakness section below. In short, they boil down to the fact that the authors must show how good/bad ReplicaDock is at all targets not only the ones where AFm fails. In addition, I have several more technical comments.</p><p>Strengths:</p><p>Impressive increase in performance on AB-AG set (although a small set and no proteins).</p></disp-quote><p>We thank the reviewer for their comments.</p><disp-quote content-type="editor-comment"><p>Weaknesses:</p><p>The presentation is a bit hard to follow. The authors mix several measures (Fnat, iRMS, RMSDbound, etc). In addition, it is not always clear what is shown. For instance, in Figure 1, is the RMSD calculated for a single chain or the entire protein? I would suggest that the author replace all these measures with two: TM-score when evaluating the quality of a single chain and DockQ when evaluating the results for docking. This would provide a clearer picture of the performance. This applies to most figures and tables.</p></disp-quote><p>We apologize for the lack of clarity owing to different metrics. Irms and fnat are standard performance metrics in the docking field, but we agree that DockQ would be simpler when the detail of the other metrics are not required. We have updated the figures Figure 5 and Figure 8 to also show DockQ comparisons.</p><p>Regarding Figure 1, as highlighted in Line 90 of the main-text, “Figure 1 shows the Ca-RMSD of all protein partners of the AFm predicted complex structures with respect to the bound and the unbound.” As suggested by the reviewer in their further comments, we have moved this FIgure to the Supplementary. We have also included TM-score comparison in the Supplementary (SupFig S2) and included clarifying statements in the main text:</p><p>“We also tested TM-scores to measure the structural deviations of the AFm predicted complex structures with respect to the bound and unbound structures (Supplementary Figure S2). However, this metric is not sensitive enough to detect the subtle, local conformational changes upon binding.”</p><disp-quote content-type="editor-comment"><p>For instance, Figure 9 could be shown as a distribution of DockQ scores.</p></disp-quote><p>We have now updated Figure 5 to include DockQ scores in Panel D. Since DockQ is a function of iRMSD, fnat and L-RMSD, it shows cumulative improvement in performance. Some of the nuanced details, such as, the protocol improves i-RMSD considerably but fnat improvement is lacking, and can highlight whether backbone sampling is the challenge or is it sidechain refinement.Therefore, we need to retain the iRMSD and fnat metrics in panel A-C . But We have incorporated this in the main text as follows:</p><p>“Finally, to evaluate docking success rates, we calculate DockQ for top predictions from AFm and AlphaRED respectively (Figure 5D). AlphaRED demonstrates a success rate (DockQ&gt;0.23) for 63% of the benchmark targets. Particularly for Ab-Ag complexes, AFm predicted acceptable or better quality docked structures in only 20% of the 67 targets. In contrast, the AlphaRED pipeline succeeds in 43% of the targets, a significant improvement.”</p><p>Further, we have reevaluated success rates in Figure 8 (previously Figure 9) and have updated the manuscript to report these updated success rates.</p><p>“By utilizing the AlphaRED strategy, we show that failure cases in AFm predicted models are improved for all targets (lower Irms for 97 of 254 failed targets) with CAPRI acceptable-quality or better models generated for 62% of targets overall (Fig 8)”.</p><disp-quote content-type="editor-comment"><p>The improvements on the models where AFm is good are minimal (if at all), and it is unclear how global docking would perform on these targets, nor exactly why the plDDT&lt;0.85 cutoff was chosen.</p></disp-quote><p>We agree with the reviewers that the improvement on the models with good AFm predictions is minimal. We acknowledge this in the text now as follows:</p><p>“Most of the improvements in the success rates are for cases where AFm predictions are worse. For targets with good AFm predictions, AlphaRED refinement results in minimal improvements in docking accuracy.”</p><p>The choice of pLDDT cutoff = 85 is elaborated in the “Interface-pLDDT correlates with DockQ and discriminates poorly docked structures” section, paragraph 3. Briefly, we tested multiple metrics and the interface pLDDT had the highest AUC, indicating that it is the best metric for this task. For interface-pLDDT we tested multiple thresholds, and the cutoff of 85 resulted in the highest percentage of true-positive and true-negative rates. This is illustrated with the confusion matrix in Figure 3.B with the precision scores. We now clarify this in the text as follows:</p><p>“With interface-pLDDT as a discriminating metric, we tested multiple thresholds to estimate the optimum cut-off for distinguishing near-native structures (defined as an interface-RMSD &lt; 4 Å) from the predictions. Figure 3B summarizes the performance with a confusion matrix for the chosen interface-pLDDT cutoff of 85. 79% of the targets are classified accurately with a precision of 75%, thereby validating the utility of interface-pLDDT as a discriminating metric to rank the docking quality of the AFm complex structure predictions.”</p><disp-quote content-type="editor-comment"><p>To better understand the performance of ReplicaDock, the authors should therefore (i) run global and local docking on all targets and report the results, (ii) report the results if AlphaFold (not multimer) models of the chains were used as input to ReplicaDock (I would assume it is similar). These models can be downloaded from AlphaFoldDB.</p></disp-quote><p>The performance of ReplicaDock on DB5.5 is tabulated in our prior work (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1010124">https://doi.org/10.1371/journal.pcbi.1010124</ext-link>) and we direct the reviewers there for the detailed performance and results. In our opinion, the benchmark suggested by the reviewer would be redundant and not worth the computational expense.</p><p>The scope of this paper is to highlight a structure prediction + physics-based modeling pipeline for docking to adapt to the accuracy of up-and-coming structure prediction tools.</p><p>Using AlphaFold monomer chains as input and benchmarking on that, albeit interesting scientifically, will not be useful for either the pipeline or biologists who would want a complex structure prediction. We thank the authors for their comments but want to reemphasize that the end goal of this work is to increase the accuracy of complex structure predictions and PPIs obtained from computational tools.</p><disp-quote content-type="editor-comment"><p>Further, it would be interesting to see if ReplicaDock could be combined with AFsample (or any other model to generate structural diversity) to improve performance further.</p></disp-quote><p>We would like to highlight that ReplicaDock is a stand-alone tool for protein docking and here we demonstrate the ability of adapting it with metrics derived from AlphaFold or other structure prediction tools (say ESMFold) such as pLDDT for conformational sampling and improving docking accuracy. We definitely agree that adapting it to use with tools such as AFSample will be interesting but it is out of scope of this work.</p><disp-quote content-type="editor-comment"><p>The estimates of computing costs for the AFsample are incorrect (check what is presented in their paper). What are the computational costs for RepliaDock global docking?</p></disp-quote><p>The authors of the AFSample paper report that “AFsample requires more computational time than AF2, as it generates 240 models, and including the extra recycles, the overall timing is 1000 more costly than the baseline.” We have reported these exact numbers in our manuscript.</p><p>The computational costs of ReplicaDock are 8-72 CPU hours on a single node with 24 processors as reported in our prior work.</p><p>For AlphaRED, the costs are slightly higher owing to the structure prediction module in the beginning and are up to 100 CPU hrs for our largest (max Nres) target.</p><disp-quote content-type="editor-comment"><p>It is unclear strictly what sequences were used as input to the modelling. The authors should use full-length UniProt sequences if they were not done.</p></disp-quote><p>We report this in the methods section of the manuscript as well as in Figure 5. Full length complex sequences were used for the models that we extracted from DB5.5.</p><p>“As illustrated in Fig. 5, given a sequence of a protein complex, we use the ColabFold implementation of AF2-multimer to obtain a predictive template.”</p><p>We clarify this in the methods section as:</p><p>“For each target in the DB5.5 dataset, we first extracted the corresponding FASTA sequence for the bound complex and then obtained AlphaFold predicted models with the ColabFold v1.5.2 implementation of AlphaFold and AlphaFold-multimer (v.2.3.0).”</p><disp-quote content-type="editor-comment"><p>The antibody-antigen dataset is small. It could easily be expanded to thousands of proteins. It would be interesting to know the performance of ReplicaDock on a more extensive set of Antibodies and nanobodies.</p></disp-quote><p>This work demonstrates the performance on the docking benchmark, i.e. given unbound structure can you predict the bound complexes. With this regard, our analysis has been focussed on targets where both the unbound and bound structures are available so that we could evaluate the ability of AlphaRED on modeling protein flexibility and docking accuracy. For antibody-antigen complexes, there are <italic>only 67 structures</italic> with both unbound and bound complexes available and they constituted our dataset. Benchmarking AlphaRED on all antibody-antigen targets can give biased results as most Ab-Ag complexes are in AlphaFold training set. Further, our work is more aimed towards predicting conformational flexibility in docking and not rigid-body docked complexes, so benchmarking on existing bound Ab-Ag structures is out of scope for this work.</p><disp-quote content-type="editor-comment"><p>Using pLDDT on the interface region to identify good/bas models is likely suboptimal. It was acceptable (as a part of the score) for AlphaFold-2.0 (monomer), but AFm behaves differently. Here, AFm provides a direct score to evaluate the quality of the interaction (ipTM or Ranking Confidence). The authors should use these to separate good/bad models (for global/local docking), or at least show that these scores are less good than the one they used.</p></disp-quote><p>We thank the reviewers for this suggestion.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Recommendations For The Authors):</bold></p><p>Some Figures could be skipped/improved</p><p>Fig 1: Use TM-score instead a much better measure (and the figure is not necessary).</p></disp-quote><p>Figure 1 compares the bias of AlphaFold towards unbound or bound forms of the proteins. We believe that this figure highlights the slight inherent bias of AlphaFold towards bound structures over unbound.</p><p>As the reviewers have suggested we have included a plot comparing the TM-scores for the structures. Further, we have moved this figure to the Supplementary.</p><disp-quote content-type="editor-comment"><p>Fig 2. Skip B (why compare RMSD with pLDDT?). Add a figure to see how this correlates over all targets not just two.</p></disp-quote><p>RMSD and LDDT both represent metrics to evaluate conformational variability between two structures, such as the bound and unbound forms of the same protein structure. On one hand where RMSD measures overall deviation of residues, LDDT allows the estimation of relative domain orientations and concerted proteins. We have elaborated this in Methods as well as in the Results section titled “AlphaFold pLDDT provides a predictive confidence measure for backbone flexibility”.</p><p>The data for the benchmark targets is now included in the Supplementary (Supplementary Figures S3-S4).</p><disp-quote content-type="editor-comment"><p>Fig 3. Color the different chains of a protein differently. Thereby the Receptor/Ligand/Bound labels can be omitted.</p></disp-quote><p>We thank the reviewers for this suggestion. However, the color scheme is chosen to highlight (1) the relative orientation of protein partners relative to each other. We have ensured that the alignment is over one partner (Receptor) so that you could see the relative orientation of the other partner (Ligand) in the modeled protein over the bound structure (in one color). (2) The coloring of the receptor and ligand chain is by pLDDT (from red to blue) to highlight that for decoys with incorrectly predicted interfaces, the pLDDT scores of the interface residues are indeed lower and can be a discriminating metric. We elaborate this in the caption of Figure 3 as well as in the section “Interface-pLDDT correlates with DockQ and discriminates poorly docked structures”. Coloring the chains of a protein differently will obfuscate the point that we are aiming to make and will be inconclusive for the readers as they would need to rely only on quantitative metrics (Irms and DockQ) reported but won’t be able to visualize the interface pLDDT of the incorrectly bound structures. We hope that this justifies the choice of our color scheme.</p><disp-quote content-type="editor-comment"><p>Fig 4. Include RankConf, ipTM, pDockQ, and other measures in the plos (they are likely better). Include DockQ for the top targets. It is difficult to estimate for multi chain complexes.</p></disp-quote><p>We thank the reviewer for this suggestion. We have now included the DockQ performances for all targets in Figure 5 (previously Figure 6) as well as re-evaluated our final success rates based on the DockQ calculations in Figure 8 (previously Figure 9).</p><disp-quote content-type="editor-comment"><p>Fig 5. use a better measure to split (see above).</p></disp-quote><p>We have elaborated on the choice of the split for the comments above and the interface pLDDT threshold of 85 is a decision made post observation on the docking benchmark. We do want to highlight that the cut-off is arbitrary and in our online server (ROSIE) as well as in custom scripts, this cut-off can be tuned by the user as required. We would suggest a cut-off of 85 based on our observations but the users are welcome to tune this as per their needs.</p><disp-quote content-type="editor-comment"><p>Fig 6. Replace lrms/fnat with DockQ.</p></disp-quote><p>We have now included DockQ scores in our manuscript.</p><disp-quote content-type="editor-comment"><p>Fig 7. Color the different chains of a protein differently.</p></disp-quote><p>We have colored the protein chains differently. AlphaFold models are in Orange, Bound complexes are in Gray, and predicted proteins from AlphaRED are in Blue-Green indicating the two partners. All models are aligned over the receptor so relative orientations of the ligand protein can be observed.</p><disp-quote content-type="editor-comment"><p>Fig 8 Color the different chains of a protein differently.</p></disp-quote><p>The chains are colored differently. We would like the reviewer to elaborate more on what they would like to observe as we believe our color scheme makes intuitive sense for readers.</p><disp-quote content-type="editor-comment"><p>Fig 9. Use DockQ instead of CAPRI criteria.</p></disp-quote><p>The figure has been updated based on DockQ. To elaborate, the CAPRI criteria is set based on DockQ scores as elaborated in the figure caption.</p></body></sub-article></article>