<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">67790</article-id><article-id pub-id-type="doi">10.7554/eLife.67790</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Most cancers carry a substantial deleterious load due to Hill-Robertson interference</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-230432"><name><surname>Tilk</surname><given-names>Susanne</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9156-9360</contrib-id><email>tilk@stanford.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-289826"><name><surname>Tkachenko</surname><given-names>Svyatoslav</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-157037"><name><surname>Curtis</surname><given-names>Christina</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0166-3802</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-7314"><name><surname>Petrov</surname><given-names>Dmitri A</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3664-9130</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund4"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-289827"><name><surname>McFarland</surname><given-names>Christopher D</given-names></name><email>cdm113@case.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund7"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Biology, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/051fd9666</institution-id><institution>Department of Genetics and Genome Sciences, Case Western Reserve University</institution></institution-wrap><addr-line><named-content content-type="city">Cleveland</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Medicine, Division of Oncology, Stanford University School of Medicine</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Genetics, Stanford University School of Medicine</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Stanford Cancer Institute, Stanford University School of Medicine</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Taylor</surname><given-names>Martin</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01nrxwf90</institution-id><institution>University of Edinburgh</institution></institution-wrap><country>United Kingdom</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Przeworski</surname><given-names>Molly</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hj8s172</institution-id><institution>Columbia University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>01</day><month>09</month><year>2022</year></pub-date><pub-date pub-type="collection"><year>2022</year></pub-date><volume>11</volume><elocation-id>e67790</elocation-id><history><date date-type="received" iso-8601-date="2021-02-23"><day>23</day><month>02</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2022-08-31"><day>31</day><month>08</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2019-09-14"><day>14</day><month>09</month><year>2019</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/764340"/></event></pub-history><permissions><copyright-statement>© 2022, Tilk et al</copyright-statement><copyright-year>2022</copyright-year><copyright-holder>Tilk et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-67790-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-67790-figures-v2.pdf"/><abstract><p>Cancer genomes exhibit surprisingly weak signatures of negative selection (Martincorena et al., 2017; Weghorn, 2017). This may be because selective pressures are relaxed or because genome-wide linkage prevents deleterious mutations from being removed (Hill-Robertson interference; Hill and Robertson, 1966). By stratifying tumors by their genome-wide mutational burden, we observe negative selection (<italic>dN</italic>/<italic>dS</italic> ~ 0.56) in low mutational burden tumors, while remaining cancers exhibit <italic>dN</italic>/<italic>dS</italic> ratios ~1. This suggests that most tumors do not remove deleterious passengers. To buffer against deleterious passengers, tumors upregulate heat shock pathways as their mutational burden increases. Finally, evolutionary modeling finds that Hill-Robertson interference alone can reproduce patterns of attenuated selection and estimates the total fitness cost of passengers to be 46% per cell on average. Collectively, our findings suggest that the lack of observed negative selection in most tumors is not due to relaxed selective pressures, but rather the inability of selection to remove deleterious mutations in the presence of genome-wide linkage.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>cancer evolution</kwd><kwd>mutation load</kwd><kwd>Hill-Robertson interference</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000051</institution-id><institution>National Human Genome Research Institute</institution></institution-wrap></funding-source><award-id>T32-HG000044-21</award-id><principal-award-recipient><name><surname>McFarland</surname><given-names>Christopher D</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>E25-CA180993</award-id><principal-award-recipient><name><surname>McFarland</surname><given-names>Christopher D</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>DP1-CA238296</award-id><principal-award-recipient><name><surname>Curtis</surname><given-names>Christina</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>R01-CA207133</award-id><principal-award-recipient><name><surname>Petrov</surname><given-names>Dmitri A</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000057</institution-id><institution>National Institute of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R35-GM118165</award-id><principal-award-recipient><name><surname>Petrov</surname><given-names>Dmitri A</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01-CA231253</award-id><principal-award-recipient><name><surname>Petrov</surname><given-names>Dmitri A</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>K99-CA226506</award-id><principal-award-recipient><name><surname>McFarland</surname><given-names>Christopher D</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The absence of negative selection observed in most cancer genomes can be explained by the intrinsic genome-wide linkage in somatic evolution and creates a substantial proteotoxic load.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Tumor progression is an evolutionary process acting on somatic cells within the body. These cells acquire mutations over time that can alter cellular fitness by either increasing or decreasing the rates of cell division and/or cell death. Mutations which increase cellular fitness (drivers) are observed in cancer genomes more frequently because natural selection enriches their prevalence within the tumor population (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="bibr" rid="bib62">Weghorn and Sunyaev, 2017</xref>). This increased prevalence of mutations across patients within specific genes is used to identify driver genes. Conversely, mutations that decrease cellular fitness (deleterious passengers) are expected to be observed less frequently. This enrichment or depletion is often measured by comparing the expected rate of nonsynonymous mutations (<italic>dN</italic>) accruing within a region of the genome to the expected rate of synonymous mutations (<italic>dS</italic>), which are presumed to be neutral. This ratio, <italic>dN</italic>/<italic>dS</italic>, is expected to be below 1 when the majority of nonsynonymous mutations are deleterious and removed by natural selection, be ~1 when all nonsynonymous mutations are neutral, and can be &gt;1 when a substantial proportion of nonsynonymous mutations are advantageous.</p><p>Two recent analyses of <italic>dN</italic>/<italic>dS</italic> patterns in cancer genomes found that for most nondriver genes <italic>dN</italic>/<italic>dS</italic> is ~1 and that only 0.1–0.4% of genes exhibit detectable negative selection (<italic>dN</italic>/<italic>dS</italic> &lt; 1) (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="bibr" rid="bib62">Weghorn and Sunyaev, 2017</xref>).This differs substantially from patterns in human germline evolution where most genes show signatures of negative selection (<italic>dN</italic>/<italic>dS</italic> ~ 0.4) (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>). Two explanations for this difference have been posited. First, the vast majority of nonsynonymous mutations may not be deleterious in somatic cellular evolution despite their deleterious effects on the organism. While most genes may be critical for proper organismal development and multicellular functioning, they may not be essential for clonal tumor growth. In this hypothesis, negative selection (<italic>dN</italic>/<italic>dS</italic> &lt; 1) should be observed only within essential genes and absent elsewhere (<italic>dN</italic>/<italic>dS</italic> ~ 1). While appealing in principle, most germline selection against nonsynonymous variants appears to be driven by protein misfolding toxicity (<xref ref-type="bibr" rid="bib15">Drummond and Wilke, 2008</xref>; <xref ref-type="bibr" rid="bib39">Lobkovsky et al., 2010</xref>), in addition to gene essentiality. These damaging folding effects ought to persist in somatic evolution.</p><p>A second hypothesis is that even though many nonsynonymous mutations are deleterious in somatic cells, natural selection fails to remove them. One possible reason for this inefficiency is the unique challenge of evolving without recombination. Unlike sexually recombining germline evolution, tumors must evolve under genome-wide linkage that creates interference between mutations, known as Hill-Robertson interference, which reduces the efficiency of natural selection (<xref ref-type="bibr" rid="bib33">Hill and Robertson, 1966</xref>). Without recombination to link and unlink combinations of mutations, natural selection must act on entire genomes – not individual mutations – and select for clones with combinations of mutations of better aggregate fitness. Thus, advantageous drivers may not fix in the population, if they arise on an unfit background, and conversely, deleterious passengers can fix, if they arise on fit backgrounds.</p><p>The inability of asexuals to eliminate deleterious passengers is driven by two Hill-Robertson interference processes: <italic>hitchhiking</italic> and <italic>Muller’s ratchet</italic> (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Hitchhiking occurs when a strong driver arises within a clone already harboring several passengers. Because these passengers cannot be unlinked from the driver under selection, they are carried with the driver to a greater frequency in the population. Muller’s ratchet is a process where deleterious mutations continually accrue within different clones in the population until natural selection is overwhelmed. Whenever the fittest clone in an asexual population is lost through genetic drift, the maximum fitness of the population declines to the next most fit clone (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). The rate of hitchhiking and Muller’s ratchet both increase with the genome-wide mutation rate (<xref ref-type="bibr" rid="bib35">Johnson, 1999</xref>; <xref ref-type="bibr" rid="bib49">Neher and Shraiman, 2012</xref>). Therefore, the second hypothesis predicts that selection against deleterious passengers should be more efficient (<italic>dN</italic>/<italic>dS</italic> &lt; 1) in tumors with lower mutational burdens.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Two Hill-Robertson interference processes that accumulate deleterious mutations at high mutation rates.</title><p>(<bold>A</bold>) Genetic hitchhiking. Each number identifies a different segment of a clone genome within a tumor. De novo beneficial driver mutations that arise in a clone can drive other mutations (passengers) in the clone to high frequencies (black dotted column). If the passenger is deleterious, both beneficial drivers and deleterious passengers can accumulate. (<bold>B</bold>) Muller’s ratchet. As the mutation rate within a tumor increases, deleterious passengers accumulate on more clones. If the fittest clone within the tumor is lost through genetic drift (black dotted row), the overall fitness of the population will decline.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig1-v2.tif"/></fig><p>Here, we leverage the 10,000-fold variation in tumor mutational burden across 33 cancer types to quantify the extent that selection attenuates, and thus becomes more inefficient, as the mutational burden increases. Using <italic>dN</italic>/<italic>dS</italic>, we find that selection against deleterious passengers and in favor of advantageous drivers is most efficient in low mutational burden cancers. Furthermore, low mutational burden cancers exhibit efficient selection across cancer subtypes, as well as within subclonal mutations, homozygous mutations, somatic copy number alterations (CNAs), and essential genes. Additionally, high mutational burden tumors appear to mitigate this deleterious load by upregulating protein folding and degradation machinery. Finally, using evolutionary modeling, we find that Hill-Robertson interference alone can in principle explain these observed patterns of selection. Modeling predicts that most cancers carry a substantial deleterious burden (~46%) that necessitates the acquisition of multiple strong drivers (~5) in malignancies that together provide a benefit of ~119%. Collectively, these results explain why signatures of selection are largely absent in cancers with elevated mutational burdens and indicate that the vast majority of tumors harbor a large mutational load.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Null models of mutagenesis in cancer</title><p>Mutational processes in cancer are heterogeneous, which can bias <italic>dN</italic>/<italic>dS</italic> estimates of selective pressures. <italic>dN</italic>/<italic>dS</italic> overcomes this issue by dividing observed mutation counts by what is expected under neutral evolution using null models. These null models must account for mutational biases that are often specific to cancer types and genomic regions.</p><p>To ensure our <italic>dN</italic>/<italic>dS</italic> calculations are robust and reproducible, we applied two different methods to account for mutational biases. The first approach uses a previously established parametric mutational model (<italic>dNdScv</italic>) that explicitly estimates the background mutational bias of each gene in its calculation of <italic>dN</italic>/<italic>dS</italic> (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>). The second approach uses a permutation-based, non-parametric (parameter-free) estimation of <italic>dN</italic>/<italic>dS</italic>. In this approach, every observed mutation is permuted while preserving the gene, patient samples, specific base change (e.g. A&gt;T) and its tri-nucleotide context. Note that permutations do not preserve the codon position of a mutation and thus can change its protein coding effect (nonsynonymous vs. synonymous). The permutations are then tallied for both nonsynonymous <italic>d</italic><sub><italic>N</italic></sub><sup>(permuted)</sup> and synonymous <italic>d</italic><sub><italic>S</italic></sub><sup>(permuted)</sup> substitutions (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>) and used as expected proportional values for the observed number of nonsynonymous <italic>d</italic><sub><italic>N</italic></sub><sup>(observed)</sup> (or simply <italic>d</italic><sub><italic>N</italic></sub>) and synonymous <italic>d</italic><sub><italic>S</italic></sub><sup>(observed)</sup> (<italic>d</italic><sub><italic>S</italic></sub>) mutations in the absence of selection. The unbiased effects of selection on a gene, <italic>dN</italic>/<italic>dS</italic>, is then:<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi mathvariant="normal">N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi mathvariant="normal">N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi mathvariant="normal">S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi mathvariant="normal">S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>For all cancer types and patient samples, p-values and confidence intervals are determined by bootstrapping patient samples. Note that this permutation procedure will account for gene and tumor-level mutational biases (e.g. neighboring bases [<xref ref-type="bibr" rid="bib3">Alexandrov and Stratton, 2014</xref>], transcription-coupled repair, S phase timing [<xref ref-type="bibr" rid="bib31">Haradhvala et al., 2016</xref>], mutator phenotypes) and their covariation. We confirmed that this approach accurately measures selection even in the presence of simulated mutational biases (Materials and methods, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2A</xref>). In addition, this approach also reliably measures the absence of selection (<italic>dN</italic>/<italic>dS</italic> = 1) in weakly expressed genes (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2C</xref>).</p><p>We find that both the parametric and non-parametric approaches identify similar patterns of selection (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). Since parametric mutational models can become very complex in cancer (exceeding 5000 parameters in some cases; <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="bibr" rid="bib65">Zapata et al., 2018</xref>), we elected to use the non-parametric approach, which makes fewer assumptions about underlying mutational processes, in subsequent calculations of <italic>dN</italic>/<italic>dS</italic>.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Attenuation of selection and increased protein folding stress in high mutation load tumors.</title><p>(<bold>A</bold>) <italic>dN</italic>/<italic>dS</italic> of passenger (red) and driver (green) gene sets within 10,288 tumors in TCGA stratified by total number of substitutions present in the tumor (<italic>d</italic><sub><italic>N</italic></sub><sup>(observed)</sup>+<italic>d</italic><sub><italic>S</italic></sub><sup>(observed)</sup>). <italic>dN</italic>/<italic>dS</italic> is calculated with error bars using a permutation-based null model (left) and <italic>dNdScv</italic> (right). A <italic>dN</italic>/<italic>dS</italic> of 1 (solid black line) is expected under neutrality. Solid gray line denotes pan-cancer genome-wide <italic>dN</italic>/<italic>dS</italic>. (<bold>B</bold>) Fraction of pathogenic missense mutations, annotated by PolyPhen2, in the same driver and passenger gene sets also stratified by total number of substitutions. Black line denotes the pathogenic fraction of missense mutations across the entire human genome. (<bold>C</bold>) Breakpoint frequency of copy number alterations (CNAs) that reside within exonic (<italic>dE</italic>) to intergenic (<italic>dI</italic>) regions within putative driver and passenger gene sets (identified by GISTIC 2.0, Materials and methods) in tumors stratified by the total number of CNAs present in each tumor and separated by CNA length. Solid black line of 1 denotes values expected under neutrality. (<bold>D</bold>) <italic>dN</italic>/<italic>dS</italic> of clonal (variant allele frequency [VAF] &gt; 0.2; darker colors) and subclonal (VAF &lt; 0.2; lighter colors) passenger and driver gene sets in tumors stratified by the total number of substitutions. A <italic>dN</italic>/<italic>dS</italic> of 1 (solid black line) is expected under neutrality. (<bold>A–D</bold>) Histogram counts of tumors within mutational burden bins are shown in the top panels. (<bold>E</bold>) Driver and passenger <italic>dN</italic>/<italic>dS</italic> values of the highest and lowest defined mutational burden bin in broad anatomical sub-categories. (<bold>F</bold>) Same as (<bold>E</bold>), except for all specific cancer subtypes with ≥500 samples. (<bold>G</bold>) Z-scores of median gene expression within all genes, HSP90, Chaperonin, and Proteasome gene sets averaged across patients (relative to an average tumor) stratified by the total number of substitutions. All shaded error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Schematic of our permuted <italic>dN</italic> and <italic>dS</italic> calculation.</title><p>Permuted synonymous and nonsynonymous counts are used to account for mutational biases in <italic>dN</italic>/<italic>dS</italic> calculations. Observed mutations and their tri-nucleotide context is shown in a solid gray bar. Permuted mutations with the same tri-nucleotide context are shown in dashed gray lines. Note that permutations do not preserve the codon position of a mutation and can alter protein coding effect (nonsynonymous vs. synonymous).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Permutation-based null model of mutagenesis corrects for mutational biases in <italic>dN</italic>/<italic>dS</italic> calculations.</title><p>(<bold>A</bold>) Simulations (<italic>N</italic>=100) of negative selection under extreme mutational bias scenarios where all mutations are generated from a single mutational signature (e.g. APOBEC or smoking, COSMIC signatures 1–9, gray titles). Bias-corrected <italic>dN</italic>/<italic>dS</italic> values calculated from these simulations are compared to simulated levels of negative selection. Colors denote bias-corrected <italic>dN</italic>/<italic>dS</italic> before negative selection was simulated, which is expected to be neutral (~1). Negative selection is simulated as the probability of randomly removing nonsynonymous mutations, (e.g. a simulated ‘true’ <italic>dN</italic>/<italic>dS</italic> of 0.1 defines simulations where each nonsynonymous mutation had a 90% probability of removal). Shapes correspond to different numbers of sites simulated. Black line identifies perfect correspondence between bias-correct <italic>dN</italic>/<italic>dS</italic> and simulated (true) <italic>dN</italic>/<italic>dS</italic>. (<bold>B</bold>) 95% confidence intervals of <italic>dN</italic>/<italic>dS</italic> in passenger mutations randomly sampled in blue (<italic>N</italic>=1000) from high mutational burden tumors (&gt;10 substitutions) in the same proportion of sites as binned in <xref ref-type="fig" rid="fig2">Figure 2A</xref>. Red line denotes observed <italic>dN</italic>/<italic>dS</italic> of passengers in TCGA as depicted in <xref ref-type="fig" rid="fig2">Figure 2A</xref>. (<bold>C</bold>) <italic>dN</italic>/<italic>dS</italic> of weakly expressed genes (defined as having &lt;1 TPM across all samples in Genotype-Tissue Expression [GTEx]) in tumors stratified by the total number of substitutions within TCGA. Solid black shows <italic>dN</italic>/<italic>dS</italic> values of 1, expected under neutrality. Error bars are shaded 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp2-v2.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Attenuation of selection with increasing mutational burden in both oncogenes and tumor suppressors.</title><p><italic>dN</italic>/<italic>dS</italic> of passenger and driver gene sets (<xref ref-type="bibr" rid="bib6">Bailey et al., 2018</xref>) within tumors in TCGA stratified by the total number of substitutions present in the tumor (<italic>d</italic><sub><italic>N</italic></sub>+<italic>d</italic><sub><italic>S</italic></sub>). <italic>dN</italic>/<italic>dS</italic> is calculated with error bars using a permutation-based null model (left) and <italic>dNdScv</italic> (right). Tumor suppressors (purple), oncogenes (blue), and pan-cancer driver (green) gene sets are shown. Solid black shows <italic>dN</italic>/<italic>dS</italic> values of 1, expected under neutrality. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp3-v2.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>No common germline polymorphisms observed in low mutation rate cancers.</title><p>(<bold>A</bold>) Fraction of mutations that overlap all germline polymorphisms in the 1000 Genomes Project within tumors stratified by the total number of substitutions. (<bold>B–D</bold>) Fraction of mutations that overlap only common (MAF &gt; 0.05, 0.01, or 0.005) polymorphisms in the 1000 Genomes Project within tumors stratified by the total number of substitutions in TCGA. Colors denote mutations that are synonymous (blue) or nonsynonymous (red). Strong negative germline selection is expected only within common polymorphisms. No mutations within low mutational burden cancers (≤10 substitutions) overlap common polymorphic sites (when MAF &gt; 0.1). Note that there are no synonymous mutations at MAF &gt; 0.05 within low mutational burden cancers that could lower <italic>dN</italic>/<italic>dS</italic> rates through germline contamination.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp4-v2.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Weaker signals of positive selection within cancer-specific drivers.</title><p><italic>dN</italic>/<italic>dS</italic> values of passenger and different driver gene sets within tumors in TCGA stratified by the total number of substitutions present in the tumor (<italic>d</italic><sub><italic>N</italic></sub>+<italic>d</italic><sub><italic>S</italic></sub>). <italic>dN</italic>/<italic>dS</italic> is calculated with error bars using a permutation-based null model (left) and <italic>dNdScv</italic> (right). Pan-cancer driver (lime) and cancer-specific (blue) driver gene sets identified by <xref ref-type="bibr" rid="bib6">Bailey et al., 2018</xref>, are shown. Pan-cancer driver genes identified in this study also exhibited stronger signatures of positive selection than driver genes identified by COSMIC (<xref ref-type="bibr" rid="bib21">Futreal et al., 2004</xref>) (light green) and Intogen (<xref ref-type="bibr" rid="bib25">Gonzalez-Perez et al., 2013</xref>) (forest green). Hence, pan-cancer drivers from <xref ref-type="bibr" rid="bib6">Bailey et al., 2018</xref>, were used throughout this study. Cancer-specific gene sets are defined as the top 100 recurrently mutated genes within the particular cancer type, and used separately for each of the 33 cancer types in TCGA. Solid black shows <italic>dN</italic>/<italic>dS</italic> values of 1, expected under neutrality. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp5-v2.tif"/></fig><fig id="fig2s6" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 6.</label><caption><title>Patterns of attenuated selection persist across tumor purity thresholds.</title><p>(<bold>A</bold>) Correlation between tumor purity (calculated by GDC using the ABSOLUTE [<xref ref-type="bibr" rid="bib35">Johnson, 1999</xref>] algorithm, Materials and methods) and the total number of substitutions in all TCGA samples (<italic>R</italic><sup>2</sup> = −0.01). Blue line denotes a linear regression fit and gray colors denote the 95% confidence intervals for the fit of this linear model. (<bold>B</bold>) Boxplot of tumor purity in TCGA samples stratified into low mutation rate bins (1–3 and 3–10 substitutions) and high mutation rate bins (10–10,000 substitutions). (<bold>C</bold>) <italic>dN</italic>/<italic>dS</italic> in driver (green) and passenger (red) gene sets of tumors in TCGA stratified by the total number of substitutions after removing tumors below various purity thresholds using a permutation-based null model. Values at the top denote the threshold of tumors removed from the analysis (e.g. 0.3 shows <italic>dN</italic>/<italic>dS</italic> of tumors with a purity ≥0.3). (<bold>D</bold>) Number of mutations in each bin within (<bold>C</bold>) after removing tumors at increasing purity thresholds. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp6-v2.tif"/></fig><fig id="fig2s7" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 7.</label><caption><title>Comparison of <italic>dN</italic>/<italic>dS</italic> to results in <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>, for tumors stratified by mutational burden.</title><p>(<bold>A</bold>) <italic>dN</italic>/<italic>dS</italic> in driver (green), passenger (red), and all gene sets (gray) of tumors in TCGA stratified by the total number of substitutions using nine bins of equal width (log-scale TMB), as depicted in <xref ref-type="fig" rid="fig2">Figure 2</xref>. Left panel uses our non-parametric null model of mutagenesis to calculate <italic>dN</italic>/<italic>dS</italic>, while the right panel uses <italic>dNdScv</italic> (from <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>) as a null model of mutagenesis. Error bars are 95% confidence intervals determined by bootstrap sampling. (<bold>B</bold>) <italic>dN</italic>/<italic>dS</italic> of driver (green), passenger (red), and all gene sets (gray) of tumors in TCGA stratified by the total number of substitutions using 20 bins of equal sample sizes, as was done in Figure 5 of <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>. The binning scheme and linear axes compress results at low TMB. To replicate <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>, three tumor types were also excluded in this analysis: UVM, CHOL, and DLBC. <italic>dNdScv</italic> was used as a null model of mutagenesis. <italic>dN</italic>/<italic>dS</italic> for driver and passenger genes sets was not calculated in Figure 5 of <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>. Error bars are 95% confidence intervals derived from <italic>dNdScv</italic>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp7-v2.tif"/></fig><fig id="fig2s8" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 8.</label><caption><title>Random permutations of the positions of observed copy number alterations (CNAs) exhibit neutral values of <italic>dE</italic>/<italic>dI</italic>.</title><p>The stop and start location of each observed CNA was randomly permuted, while preserving its length. <italic>dE</italic>/<italic>dI</italic> was calculated for CNAs (with and without non-focal amplifications) using both metrics: breakpoint frequency and fractional overlap. <italic>dE</italic>/<italic>dI</italic> values of random permutations are ~1, as expected for CNAs not experiencing selection.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp8-v2.tif"/></fig><fig id="fig2s9" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 9.</label><caption><title>Fractional overlap of copy number alterations (CNAs) within exomic regions (<italic>dE</italic>) relative to intergenic regions (<italic>dI</italic>) exhibits similar patterns of selection as fractional overlap.</title><p>Calculations of fractional overlap (<xref ref-type="bibr" rid="bib64">Zack et al., 2013</xref>) of exomic regions (<italic>dE</italic>) to intergenic (<italic>dI</italic>) regions within passenger and GISTIC (<xref ref-type="bibr" rid="bib45">Mermel et al., 2011</xref>) driver gene sets in tumors stratified by the total number of CNAs present. <italic>dE</italic>/<italic>dI</italic> is shown separately for CNAs &gt;100 kb in length (right) and smaller than 100 kb in length (left). Solid black line of 1 denotes values expected under neutrality. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp9-v2.tif"/></fig><fig id="fig2s10" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 10.</label><caption><title>Signal of negative selection in subclonal mutations are robust to variant allele frequency (VAF) threshold.</title><p><italic>dN</italic>/<italic>dS</italic> calculations within clonal and subclonal passenger and driver gene sets within tumors in TCGA stratified by the total number of substitutions using a permutation-based null model of mutagenesis. Title of each graph corresponds to increasing VAF threshold value used to define ‘subclonal’ (e.g. mutations with a VAF &gt; 0.2 are clonal; mutations with a VAF &lt; 0.2 are subclonal). Darker colors denote clonal passengers and drivers, while lighter colors denote subclonal passengers and drivers. ‘All Passengers’ and ‘All Drivers’ contain the entire set of all clonal and subclonal passengers or drivers as a reference. Solid line of 1 is shown of <italic>dN</italic>/<italic>dS</italic> values expected under neutrality. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp10-v2.tif"/></fig><fig id="fig2s11" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 11.</label><caption><title><bold>A</bold>ttenuation of negative selection within different functional gene sets.</title><p><italic>dN</italic>/<italic>dS</italic> of passengers within different functional gene sets in the highest (black) and lowest (gray) mutational burden bin across all tumors using a permutation-based null model of mutagenesis. Dotted line denotes genome-wide <italic>dN</italic>/<italic>dS</italic> of passengers for all mutation rates. Error bars are 95% confidence intervals determined by bootstrap sampling. Patterns of negative selection are not specific to any functional category shown here. Functional gene sets were chosen based on previous literature that supports these categories as being relevant to protein misfolding (i.e. translation, transcription, highly expressed genes, genes with high degrees of protein-protein interaction) (<xref ref-type="bibr" rid="bib16">Drummond and Wilke, 2009</xref>) or has been previously reported to be under constraint (e.g. genes related to chromosome segregation and essential or housekeeping genes) (<xref ref-type="bibr" rid="bib66">Zhang and Li, 2004</xref>; <xref ref-type="bibr" rid="bib51">Potapova and Gorbsky, 2017</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp11-v2.tif"/></fig><fig id="fig2s12" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 12.</label><caption><title>Attenuation of selection in somatic nucleotide variants (SNVs) persists across cancer subtypes and broad cancer group categories.</title><p>(<bold>A</bold>) <italic>dN</italic>/<italic>dS</italic> in passenger and driver gene sets within tumors stratified by the total number of substitutions in broad tumor sub-categories. Error bars are 95% confidence intervals determined by bootstrap sampling. (<bold>B</bold>) Log-scale heatmap of <italic>dN</italic>/<italic>dS</italic> values in passenger and driver gene sets of tumors stratified by the total number of substitutions within all 33 cancer subtypes in TCGA. <italic>dN</italic>/<italic>dS</italic> of the lowest and highest mutational burden bin and the total number of tumors for each cancer subtype are shown.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp12-v2.tif"/></fig><fig id="fig2s13" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 13.</label><caption><title>Attenuation of selection in copy number alterations (CNAs) in cancer subtypes and broad cancer group categories.</title><p><italic>dE</italic>/<italic>dI</italic> in driver (green) and passenger (red) gene sets in tumors stratified by the total number of CNAs for the six most commonly sequenced cancer subtypes (presented in <xref ref-type="fig" rid="fig2">Figure 2</xref>) calculated using (<bold>A</bold>) normalized fractional overlap and (<bold>B</bold>) breakpoint frequency. <italic>dE</italic>/<italic>dI</italic> in driven and passenger gene sets in tumors stratified into broad cancer groups calculated using (<bold>C</bold>) normalized fractional overlap and (<bold>D</bold>) normalized breakpoint frequency. <italic>dE</italic>/<italic>dI</italic> &gt; 1 suggests positive selection, while <italic>dE</italic>/<italic>dI</italic> &lt; 1 suggests negative selection.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp13-v2.tif"/></fig><fig id="fig2s14" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 14.</label><caption><title>Upregulation of heat shock protein pathways in tumors with elevated mutational burdens.</title><p>(<bold>A</bold>) Z-scores of median gene expression of (<bold>i</bold>) all genes, (ii) HSP90, (iii) Chaperonins, and (iv) the Proteasome averaged across tumors stratified by the total number of copy number alterations (CNAs). Expression of HSP90, Chaperonins, and Proteasome gene sets increases with the mutational burden of tumors (weighted <italic>R<sup>2</sup></italic> of 0.81, 0.87, and 0.86, respectively). Error bars are 95% confidence intervals determined by bootstrap sampling. (<bold>B</bold>) Correlation coefficients (<bold><italic>r</italic></bold>) of the expression of each gene in the genome (gray) in tumors stratified by the total number of substitutions. Shown in arrows are the correlation coefficients for HSP90 (blue), Chaperonins (orange), and the Proteasome (purple). Dashed lines in intervals of 0.25 are for viewing purposes only. (<bold>C</bold>) Median correlation coefficients of 10 million randomly sampled gene sets of the same size as HSP90, Chaperonins, and the Proteasome (<italic>n</italic>=28) in gray. Red line denotes the median correlation coefficients of genes in HSP90, Chaperonins, and the Proteasome (0.66). (<bold>D–E</bold>) Log-scale heatmap of changes in the Z-scores of median gene expression values of gene sets in for tumors stratified by the total number of substitutions (<bold>D</bold>) or CNAs (<bold>E</bold>) for cancer subtypes in TCGA. Changes in the mean gene expression of all genes, HSP90, Chaperonins, and Proteasome gene sets in the lowest and highest mutational burden bin for each cancer subtype are shown. Colors denote whether changes in gene expression from low mutational burden bins to high mutational burden bins are positive (green) or negative (red). Expression of HSP90, Chaperonins, and Proteasome gene sets increases with the mutational burden of tumors across cancer types stratified by the number of somatic nucleotide variants (SNVs) (p&lt;0.05, p&lt;1.3 × 10<sup>–4</sup>, p&lt;5.2 × 10<sup>–4</sup>, respectively; Wilcoxon signed-rank test) and CNAs (p&gt;0.05, p&lt;9.3 × 10<sup>–3</sup>, p&lt;1.9 × 10<sup>–2</sup> respectively; Wilcoxon signed-rank test).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp14-v2.tif"/></fig><fig id="fig2s15" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 15.</label><caption><title>The power to detect signals of selection is dependent on the quality of mutation calls.</title><p><italic>dN</italic>/<italic>dS</italic> values in passenger and driver gene sets within tumors in TCGA stratified by the total number of substitutions when using low-quality mutations (bottom panels, ‘Mutect2 SNP Calls’) and high-quality mutations (top panels, ‘MC3 SNP Calls’). ‘MC3 SNP Calls’ are a consensus set of mutations from seven different mutation callers and ‘Mutect2 SNP calls’ are mutations calls from one mutation caller. <italic>dN</italic>/<italic>dS</italic> is calculated with error bars using a permutation-based null model (left) and <italic>dNdScv</italic> (right). A <italic>dN</italic>/<italic>dS</italic> of 1 (solid black line) is expected under neutrality. Error bars are 95% confidence intervals determined by bootstrap sampling.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp15-v2.tif"/></fig><fig id="fig2s16" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 16.</label><caption><title>Quantity of mutations within each mutational burden bin for data depicted in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</title><p>(<bold>A–D</bold>) all report the total number of samples used in their respective figure pane within <xref ref-type="fig" rid="fig2">Figure 2</xref>. (<bold>A</bold>) Counts of mutations in passenger (red) and driver (green) gene sets within tumors stratified by the total number of substitutions in ICGC and TCGA. (<bold>B</bold>) Counts of the fraction of pathogenic missense mutations, annotated by PolyPhen2, in the same driver and passenger gene sets also stratified by total number of substitutions. (<bold>C</bold>) Counts of copy number alterations (CNAs) that reside within putative driver and passenger gene sets (identified by GISTIC 2.0, Materials and methods) in tumors stratified by the total number of CNAs and separated by CNA length. (<bold>D</bold>) Counts of clonal (VAF &gt; 0.2; darker colors) and subclonal (VAF &lt; 0.2; lighter colors) passenger and driver gene sets in tumors stratified by the total number of substitutions.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig2-figsupp16-v2.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Attenuation of selection in drivers and passengers for elevated mutational burden tumors</title><p>We estimated <italic>dN</italic>/<italic>dS</italic> patterns in both driver and passenger gene sets across 10,288 tumors from TCGA aggregated over 33 cancer types (<xref ref-type="bibr" rid="bib18">Ellrott et al., 2018</xref>) (Materials and methods). Since TCGA is composed of whole-exome data, which limits our ability to assess mutations in non-coding regions, we elected to use the total number of protein-coding mutations as our proxy for the mutational burden of tumors. To quantify the extent that selection attenuates as the mutational burden increases, we stratified tumors into bins based on their total number of substitutions on a log-scale. For each bin of tumors, we pooled all of the variants together and estimated <italic>dN</italic>/<italic>dS</italic> jointly. Consistent with the inefficient selection model, whereby selection fails to eliminate deleterious mutations in high mutational burden tumors, we observe pervasive selection against passengers exclusively in tumors with low mutational burdens (<italic>dN</italic>/<italic>dS</italic> ~ 0.56 in tumors with ≤3 substitutions, while <italic>dN</italic>/<italic>dS</italic> ~ 0.93 in tumors with &gt;10 substitutions, <xref ref-type="fig" rid="fig2">Figure 2A</xref>). We observed little negative selection in passenger genes when aggregating tumors across all mutational burdens (<italic>dN</italic>/<italic>dS</italic> ~ 0.93), which is broadly similar to previous estimates (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="bibr" rid="bib62">Weghorn and Sunyaev, 2017</xref>; <xref ref-type="bibr" rid="bib65">Zapata et al., 2018</xref>; <xref ref-type="bibr" rid="bib50">Ostrow et al., 2014</xref>).</p><p>We confirmed that negative selection on passengers is specific to low mutational burden tumors and not biased by small sample sizes (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2B</xref>). We randomly sampled passengers from high mutational burden tumors (&gt;10 substitutions) 1000 times using the same bin sizes in <xref ref-type="fig" rid="fig2">Figure 2A</xref> and calculated <italic>dN</italic>/<italic>dS</italic>. Within the smallest bin size (<italic>N</italic>=168 somatic nucleotide variant [SNVs]), negative selection on passengers sampled from high mutational burden tumors was absent (average <italic>dN</italic>/<italic>dS</italic> ~ 0.96) compared to observed <italic>dN</italic>/<italic>dS</italic> in low mutational burden tumors (<italic>dN</italic>/<italic>dS</italic> ~ 0.56; p&lt;2.2<sup>–16</sup>). In fact, only 1.7% of randomly sampled sets of sites had similar signals of negative selection (<italic>dN</italic>/<italic>dS</italic> &lt; 0.56).</p><p>Also consistent with the inefficient selection model, drivers exhibit a similar but opposing trend of attenuated selection at elevated mutational burdens (<italic>dN</italic>/<italic>dS</italic> ~ 2.7 when in tumors with ≤3 substitutions and <italic>dN</italic>/<italic>dS</italic> gradually declines to ~1.16 in tumors with &gt;100 substitutions). This pattern is not specific to drivers that are oncogenes or tumor suppressors (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>). While the attenuation of selection against passengers in higher mutational burden tumors is a novel discovery, this pattern among drivers has been reported previously (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>). Furthermore, we confirmed that these patterns are robust to the choices that we made in our analysis pipeline. These include the: (i) effects of germline SNP contamination (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>), (ii) choice of driver gene set (<xref ref-type="bibr" rid="bib6">Bailey et al., 2018</xref>, IntOGen <xref ref-type="bibr" rid="bib25">Gonzalez-Perez et al., 2013</xref>, and COSMIC <xref ref-type="bibr" rid="bib56">Tate et al., 2019</xref>; <xref ref-type="bibr" rid="bib19">Forbes et al., 2008</xref>, <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>), (iii) differences in tumor purity and thresholding (<xref ref-type="fig" rid="fig2s6">Figure 2—figure supplement 6</xref>), and (iv) null model of mutagenesis (<italic>dNdScv</italic>, <xref ref-type="fig" rid="fig2">Figure 2A</xref> and <xref ref-type="fig" rid="fig2s7">Figure 2—figure supplement 7</xref>; <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>) (Materials and methods).</p><p>If negative selection is more pronounced in low mutational burden tumors, then the nonsynonymous mutations observed should also be less functionally consequential. By annotating the functional effect of all missense mutations using PolyPhen2 (<xref ref-type="bibr" rid="bib2">Adzhubei et al., 2010</xref>; <xref ref-type="fig" rid="fig2">Figure 2B</xref>), we indeed find that observed nonsynonymous passengers are less damaging in low mutational burden cancers. Similarly, driver mutations become less functionally consequential as mutational burden increases, as expected for mutations experiencing inefficient positive selection (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Together these two trends provide additional and orthogonal evidence that selective forces on nonsynonymous mutations are more efficient in low mutational burden cancers.</p><p>Since all mutational types experience Hill-Robertson interference, attenuated selection should also persist in CNAs. We used two previously published statistics to quantify selection in CNAs: breakpoint frequency (<xref ref-type="bibr" rid="bib37">Korbel et al., 2007</xref>) and fractional overlap (<xref ref-type="bibr" rid="bib64">Zack et al., 2013</xref>). For both measures, we compare the number of CNAs that either terminate (breakpoint frequency) within or partially overlap (fractional overlap) <bold>E</bold>xonic regions of the genome relative to non-coding (<bold>I</bold>ntergenic and <bold>I</bold>ntronic) regions (<italic>dE</italic>/<italic>dI</italic>, see Materials and methods). Like <italic>dN</italic>/<italic>dS</italic>, <italic>dE</italic>/<italic>dI</italic> is expected to be &lt;1 in genomic regions experiencing negative selection, &gt;1 in regions experiencing positive selection (e.g. driver genes), and ~1 when selection is absent or inefficient (<xref ref-type="fig" rid="fig2s8">Figure 2—figure supplement 8</xref>). Using <italic>dE</italic>/<italic>dI</italic>, we observe attenuating selection in both driver and passenger CNAs as the total number of CNAs increases for both breakpoint frequency (<xref ref-type="fig" rid="fig2">Figure 2C</xref>) and fractional overlap (<xref ref-type="fig" rid="fig2s9">Figure 2—figure supplement 9</xref>). While CNAs of all lengths experience attenuated selection, CNAs longer than the average gene length (&gt;100 KB) experience greater selective pressures in drivers. Collectively, these results strongly support the inefficient selection model and argue that the observed patterns must be due to a universal force in tumor evolution. We find that selection consistently attenuates in both drivers and passengers across all cancers as mutational burden increases.</p></sec><sec id="s2-3"><title>Strong selection in low mutational burden tumors cannot be explained by mutational timing, gene function, or tumor type</title><p>We next tested alternative hypotheses to the inefficient selection model. We considered the possibility that selection is strong only during normal tissue development, but absent after cells have transformed to malignancy. This would disproportionately affect low mutational burden tumors, as a greater proportion of their mutations arise prior to tumor transformation. If true, then attenuated selection should be absent in subclonal mutations, which must arise during tumor growth. However, selection clearly attenuates with increasing mutational burden for the subset of likely subclonal mutations with variant allele frequency (VAF) below 20% (<xref ref-type="fig" rid="fig2">Figure 2D</xref> and <xref ref-type="fig" rid="fig2s10">Figure 2—figure supplement 10</xref>). Although selection attenuates in drivers and passengers in both subclonal and clonal mutations, selection is weaker in both drivers and passengers with lower VAFs. Weaker efficiency of selection among less frequent variants is expected under a range of population genetic models (<xref ref-type="bibr" rid="bib46">Messer, 2009</xref>) and especially so in rapidly expanding, spatially constrained cancers (<xref ref-type="bibr" rid="bib53">Sottoriva et al., 2015</xref>). In addition, heterozygous mutations, to the extent they are only partially dominant (<xref ref-type="bibr" rid="bib40">López et al., 2020</xref>), are also expected to exhibit lower VAFs and experience weaker selection.</p><p>Next, we considered and rejected the possibility that attenuated selection is limited to particular types of genes. We first annotated our observed mutations by different functional categories and Gene Ontology (GO) terms (<xref ref-type="bibr" rid="bib32">Harris et al., 2004</xref>) and find that negative selection is not specific to any particular gene functional category expected to be under constraint, and specifically not limited to essential or housekeeping genes – a key prediction of the ‘weak selection’ model (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="fig" rid="fig2s11">Figure 2—figure supplement 11</xref>, p&lt;0.05, Wilcoxon signed-rank test).</p><p>Finally, we found that these patterns of attenuated selection persist across cancer subtypes for both SNVs and CNAs. We calculated <italic>dN</italic>/<italic>dS</italic> in tumors grouped by nine broad anatomical sub-categories (e.g. neuronal) and 33 subtype classifications (<xref ref-type="bibr" rid="bib27">Grossman et al., 2016</xref>; <xref ref-type="fig" rid="fig2">Figure 2E–F</xref>). We find that patterns of attenuated selection in SNVs persists in the broad and specific (drivers p=3.8 × 10<sup>–5</sup>, passengers p=1.7 × 10<sup>–2</sup>, Wilcoxon signed-rank test; <xref ref-type="fig" rid="fig2s12">Figure 2—figure supplement 12</xref>) classification schemes. Furthermore, <italic>dE</italic>/<italic>dI</italic> measurements of CNAs exhibit similar patterns of selection in broad (<xref ref-type="fig" rid="fig2s13">Figure 2—figure supplement 13</xref>) and specific subtypes (<xref ref-type="fig" rid="fig2">Figure 2F</xref>; drivers p&lt;0.05 and passengers p&lt;0.05).</p><p>Collectively, these results suggest that tumors with elevated mutational burdens carry a substantial deleterious load. Since nonsynonymous mutations are thought to be primarily deleterious by inducing protein misfolding (<xref ref-type="bibr" rid="bib15">Drummond and Wilke, 2008</xref>; <xref ref-type="bibr" rid="bib39">Lobkovsky et al., 2010</xref>), we tested whether an increase in the number of passenger mutations in tumors would lead to elevated protein folding stress, and, in turn, drive the upregulation of heat shock and protein degradation (<xref ref-type="bibr" rid="bib44">McGrail et al., 2020</xref>) pathways in cancer (<xref ref-type="bibr" rid="bib52">Santagata et al., 2011</xref>). Indeed, gene expression of HSP90, Chaperonins, and the Proteasome does increase across the whole range of SNV (weighted <italic>R</italic><sup>2</sup> of 0.84, 0.78, and 0.78, respectively) and CNA burdens (weighted <italic>R</italic><sup>2</sup> of 0.83, 0.88, and 0.85, respectively) (<xref ref-type="fig" rid="fig2">Figure 2G</xref> and <xref ref-type="fig" rid="fig2s14">Figure 2—figure supplement 14A</xref>). This trend persists across cancer types for SNVs and CNAs (<xref ref-type="fig" rid="fig2s14">Figure 2—figure supplement 14D-E</xref>). Importantly, expression of these gene sets increases across the whole range of mutational burdens, even after the <italic>dN</italic>/<italic>dS</italic> of passengers approaches 1. This result presents additional evidence that passengers continue to impart a substantial cost to cancer cells, even in high mutational burden tumors.</p></sec><sec id="s2-4"><title>Evolutionary modeling estimates the fitness effects of drivers and passengers, and rate of Hill-Robertson interference processes</title><p>We next tested whether Hill-Robertson interference – a process where selection becomes inefficient due to interference between linked mutations with competing fitness effects – alone can generate these patterns of attenuated selection. Specifically, we modeled tumor progression as a simple evolutionary process with advantageous drivers and deleterious passengers. We then used approximate Bayesian computation (ABC) to compare these simulations to observed data and infer the mean fitness effects of drivers and passengers.</p><p>Our previously developed evolutionary simulations model a well-mixed population of tumor cells that can randomly acquire advantageous drivers and deleterious passengers during cell division (<xref ref-type="bibr" rid="bib42">McFarland et al., 2013</xref>). The product of the individual fitness effects of these mutations determines the relative birth and death rate of each cell, which in turn dictates the population size <italic>N</italic> of the tumor. If the population size of a tumor progresses to malignancy (<italic>N</italic>&gt;1,000,000) within a human lifetime (≤100 years), the accrued mutations and patient age are recorded. The mutation rate of each simulated tumor is randomly sampled from a broad range (10<sup>–12</sup>–10<sup>–7</sup> mutations · nucleotide<sup>–1</sup> · generation<sup>–1</sup>, Materials and methods). Although this model ignores a great deal of known tumor biology, we believe it constitutes the simplest evolutionary model that could possibly recapitulate observed selection for drivers and against passengers. Our question is not whether this model is correct in all details but rather whether even such a simple model can generate quantitatively similar patterns as observed in the data with sensible values of mutation rates and selection coefficients.</p><p><xref ref-type="fig" rid="fig3">Figure 3A</xref> illustrates the ABC procedure. To compare our model to observed data, we simulated an exponential distribution of fitness effects (DFEs) with mean fitness values that spanned a broad range (10<sup>–2</sup>–10<sup>0</sup> for driver and 10<sup>–4</sup>–10<sup>–2</sup> for passengers, Materials and methods). We summarized observed and simulated data using statistics that capture three relationships: (i) the dependence of driver and passenger <italic>dN</italic>/<italic>dS</italic> rates on mutational burden, (ii) the rate of cancer age incidence (SEERs database <xref ref-type="bibr" rid="bib48">National Cancer Institute, 2007</xref>), and (iii) the distribution of mutational burdens (summary statistics of (ii) and (iii) were based on theoretical parametric models <xref ref-type="bibr" rid="bib20">Frank, 2007</xref>, Materials and methods, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplements 1</xref>–<xref ref-type="fig" rid="fig3s2">2</xref>). We then inferred the posterior probability distribution of mean driver fitness benefit and mean passenger fitness cost using a rejection algorithm that we validated using leave-one-out cross validation (CV) (Materials and methods, <xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Approximate Bayesian computation (ABC) procedure estimates the strength of selection in passengers and drivers.</title><p>(<bold>A</bold>) Schematic overview of the ABC procedure used. A model of tumor evolution with genome-wide linkage contains two parameters – <italic>s</italic><sub><italic>drivers</italic></sub> (mean fitness benefit of drivers) and <italic>s</italic><sub><italic>passengers</italic></sub> (mean fitness cost of passengers) – sampled over broad prior distributions of values. Simulations begin with an initiating driver event that establishes the initial population size of the tumor. The birth rate of each individual cell within the tumor is determined by the total accumulated fitness effects of drivers and passengers. If the final population size of the tumor exceeds 1 million cells within a human lifetime (100 years), patient age and accrued mutations are recorded. Summary statistics of four relationships are used to compare simulations to observed data: (<bold>i</bold>) <italic>dN</italic>/<italic>dS</italic> rates of drivers and (ii) passengers across mutational burden, (iii) rates of cancer incidence vs. age, and (iv) the distribution of mutational burdens. Simulations that excessively deviate from observed data are rejected (Materials and methods). (<bold>B–C</bold>) Inferred posterior probability distributions of <italic>s</italic><sub><italic>drivers</italic></sub> and <italic>s</italic><sub><italic>passengers</italic></sub>. The maximum likelihood estimate (MLE) of <italic>s</italic><sub><italic>drivers</italic></sub> is 53.0% (green, 95% CI [16.0, 111.4]), and the MLE of <italic>s</italic><sub><italic>passengers</italic></sub> is 1.03% (green, 95% CI [0.40, 3.98%]). (<bold>D–F</bold>) Comparison of the summary statistics of the best-fitting simulations (MLE parameters, dashed lines) to observed data (solid lines). (<bold>D</bold>) <italic>dN</italic>/<italic>dS</italic> rates of passengers (red) and drivers (light green) for simulated and observed data vs. mutational burden. A model where 6% of synonymous mutations within drivers experience positive selection (dark green) was also considered. (<bold>E</bold>) Cancer incidence rates for patients above 20 years of age. (<bold>F</bold>) Distribution of the mutational burdens of tumors.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title><italic>dN</italic>/<italic>dS</italic> rates of drivers and passengers in simulated cancers with various fitness coefficients.</title><p>Ten-thousand simulated tumors were generated for various combinations of mean driver fitness benefits (<italic>s</italic><sub><italic>drivers</italic></sub>) and mean passenger fitness costs (<italic>s</italic><sub><italic>passengers</italic></sub>, Materials and methods). For some parameter combinations, the combined fitness cost of passengers overwhelmed the fitness benefit of drivers and prevented cancer progression within 100 years (dark gray). <italic>dN</italic>/<italic>dS</italic> values of simulated mutations were calculated for drivers (left) and passengers (right) at various mutational burden (total number of nonsynonymous and synonymous mutations). Top row is a mutational burden of 1–10; middle row is 11–100, and bottom row is 100–1000. Some parameter combinations did not produce any tumors with low mutational burdens (light gray). Across all parameters, positive selection on drivers and negative selection against passengers attenuates with mutational burden. Passengers exhibit minimal negative selection in general, despite a collective burden that often prevented tumor progression, because of strong Hill-Roberston interference in asexual populations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Probability of cancer by age and mutational burdens in simulated cancers at various fitness coefficients.</title><p>Clinical summary statistics of simulated tumors at various combinations of mean driver fitness benefits (<italic>s</italic><sub><italic>drivers</italic></sub>) and mean passenger fitness costs (<italic>s</italic><sub><italic>p</italic></sub>, Materials and methods). (<bold>A</bold>) Initial population size <italic>N</italic><sup>0</sup> of simulated tumors. Initial population size approximates the equilibrium population size of a tumor following an initiating driver. Large population sizes are necessary for tumor progression when passenger deleteriousness is large compared to driver advantageousness – otherwise natural selection cannot drive carcinogenesis. Eventually, tumor progression is not possible for any reasonable initial population size (gray area). (<bold>B</bold>) Maximum likelihood estimate (MLE) of gamma distribution shape parameters describing the cancer age incidence rates of simulated tumors. A gamma distribution of age incidence is expected from the Armitage-Doll multistage model of tumorigenesis and describes human age incidence rates well (Materials and methods) (<xref ref-type="bibr" rid="bib20">Frank, 2007</xref>). Larger values correspond to a steeper increase in rate with age; human patient rates are ~5 pan-cancer. Scale parameter of the parametric fit is not informative because of a Gauge freedom in the model. (<bold>C</bold>) MLE of shape and (<bold>D</bold>) scale parameters of negative binomial distributions describing the mutational burdens of simulated tumors. Smaller values of shape parameter correspond to broader distributions of mutational burden; human tumors exhibit a value of ~2 pan-cancer. Smaller values of scale parameter correspond to a larger mean mutational burden; human tumors exhibit a value of ~1/50 (i.e. 50 passengers per rate-limiting driver).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp2-v2.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Implementation and use of approximate Bayesian computation (ABC) for model selection and parameter estimation.</title><p>(<bold>A</bold>) Leave-one-out cross validation (CV) on the simulated data was used to select an optimal rejection tolerance and optimal rejection method. Observed data can be compared to simulated data using model rejection alone (left), or by comparing observed data to a (middle) local linear regression or (right) feed-forward neural network single-layer model trained on the simulated data. In general, unsupervised training of a neural network on simulated data will often improve prediction accuracy by denoising stochasticity in the simulations (via kernel prediction.) A neural network with a rejection tolerance of 0.5 minimized prediction error of both driver and passenger fitness effects (illustrated by dotted lines) and was used to infer selection coefficients. This CV optimization procedure for ABC is advised (<xref ref-type="bibr" rid="bib13">Csilléry et al., 2012</xref>). (<bold>B</bold>) Posterior probability of models of tumor evolution incorporating synonymous drivers. The prior distribution of synonymous driver fractions (uniform from 0% to 20%) is nearly identical to this posterior distribution. This suggests that nearly all models incorporating synonymous drivers can explain observed <italic>dN</italic>/<italic>dS</italic> patterns with the right combination of fitness parameters. (<bold>C</bold>) Posterior distribution of fitness effect of driver fitness benefits (<italic>s</italic><sub><italic>drivers</italic></sub>) and passenger fitness costs (<italic>s</italic><sub><italic>passengers</italic></sub>) after synonymous drivers are incorporated. MLE (circles) and 95% confidence intervals (lines) are reported. Similar to (<bold>B</bold>), incorporation of synonymous drivers undermines the ability of ABC to accurately infer fitness coefficients.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp3-v2.tif"/></fig><fig id="fig3s4" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 4.</label><caption><title>Evidence of positive selection on synonymous mutations within driver genes at low mutational burdens.</title><p>(<bold>A</bold>) The quantity of synonymous mutations within driver genes was compared to the quantity of synonymous mutations within passenger genes and both were normalized by their expected frequencies using <italic>dNdScv</italic>. Black line denotes the genome-wide ratio of synonymous drivers to synonymous passengers (~2%, i.e. driver genes are ~2% of the human coding genome). At low mutational burdens, a non-significant increase in the quantity of synonymous drivers is observed, suggestive of positive selection for these mutations. (<bold>B</bold>) The change in codon usage imparted by all synonymous mutations was calculated for oncogenes, tumor suppressors, and passenger genes. Bias in codon usage suggests a functional effect of synonymous mutations. Increase in codon usage is expected to increase translational efficiency and increase protein abundance. Oncogenes are expected to exhibit positive selection for increased codon usage and exhibit a non-significant increase as mutational burden declines – consistent with positive selection for synonymous mutations within oncogenic drivers that is attenuated by Hill-Robertson interference. Similarly, tumor suppressors are expected to exhibit a decrease in codon usage at low mutational burdens, which is indeed significant (p=0.03) presumably because there are more annotated tumor suppressor genes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp4-v2.tif"/></fig><fig id="fig3s5" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 5.</label><caption><title>Distribution of mutation rates of simulated tumors.</title><p>(<bold>A</bold>) Mutation rates of all simulated tumors were randomly sampled from a uniform distribution (in log-space) from 10<sup>–12</sup> to 10<sup>–7</sup> nucleotide<sup>–1</sup> · generation<sup>–1</sup>. (<bold>B</bold>) In simulations that best agreed with observed data (maximum likelihood estimate [MLE] of <italic>s</italic><sub><italic>drivers</italic></sub> = 53%, <italic>s</italic><sub><italic>passengers</italic></sub> = 1.03%), only tumors with intermediate mutation rates progressed to cancer within 100 years. Tumors with lower mutation rates do not progress to cancer within the 100-year time constraint of simulations, while tumors with exceptionally high mutation rates collapse via mutational meltdown.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp5-v2.tif"/></fig><fig id="fig3s6" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 6.</label><caption><title>Relative contribution of genetic hitchhiking and Muller’s ratchet to fix deleterious passengers.</title><p>Using analytical theory developed in <xref ref-type="bibr" rid="bib49">Neher and Shraiman, 2012</xref>; <xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>; <xref ref-type="bibr" rid="bib5">Bachtrog and Gordo, 2004</xref>, we can estimate the relative rates of genetic hitchhiking and Muller’s ratchet in our pan-cancer model of tumor evolution. As the relative strength of driver alterations increase (<italic>s</italic><sub><italic>drivers</italic></sub>) relative to the selective cost of passengers (<italic>s</italic><sub><italic>passengers</italic></sub>), more passengers hitchhike with each driver sweep (left). This increases the relative contribution of observed passengers that accumulate via hitchhiking (right). Using the maximum likelihood estimates (MLE) of selection for drivers and against passengers, we estimate that an average of 1.2 deleterious passengers hitchhike with each driver, which account for 2.0% of accumulated passengers (the majority, and remainder, accumulate via Muller’s ratchet).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-fig3-figsupp6-v2.tif"/></fig></fig-group><p>Using this approach, the maximum likelihood estimate (MLE) of mean driver fitness benefit is 53% (<xref ref-type="fig" rid="fig3">Figure 3B</xref>), while the MLE of passenger mean fitness cost is 1.03% (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). Simulations with these MLE values agree well with all observed data (<xref ref-type="fig" rid="fig3">Figure 3D–F</xref>, Pearson’s <italic>r</italic>=0.988 for combined driver/passenger <italic>dN</italic>/<italic>dS</italic>).</p><p>While Hill-Robertson interference alone explains <italic>dN</italic>/<italic>dS</italic> rates in the passengers well, the simulations most consistent with observed data still exhibited consistently higher <italic>dN</italic>/<italic>dS</italic> rates in drivers (<xref ref-type="fig" rid="fig3">Figure 3D</xref>). We tested whether positive selection on synonymous mutations within driver genes could explain this discrepancy. Indeed, we find that a model incorporating synonymous drivers agrees modestly better with observed statistics (3.5-fold relative likelihood, ABC posterior probability). The best-fitting model predicts that ~6% of synonymous mutations within driver genes experience positive selection, which is consistent with previous estimates for human oncogenes (<xref ref-type="bibr" rid="bib54">Supek et al., 2014</xref>) (Materials and methods, <xref ref-type="fig" rid="fig3">Figure 3D</xref> and <xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>). Furthermore, we observe additional evidence of selection and codon bias in synonymous drivers exclusive to low mutational burdens (TCGA samples, Materials and methods, <xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>).</p><p>We note that although deleterious passengers are necessary to explain attenuation of negative selection with mutational burden in passengers, alternative explanations could also contribute to attenuation of positive selection in drivers. Specifically, high mutational burden tumors are more likely to contain mutations in pan-cancer driver gene sets which might not directly contribute to tumorigenesis in specific tumors, and thus might not be under direct positive selection in all tumors. Similarly, additional driver mutations might not directly contribute to tumor fitness beyond a certain number of driver mutations (e.g. 5-hit model). Nonetheless, it’s important to note that Hill-Robertson interference is capable of reproducing all the features of the data (steep attenuation of negative selection in passengers and gradual attenuation of positive selection in drivers).</p><p>Overall, our results indicate that rapid adaptation through natural selection – acting on entire genomes, rather than individual mutations – is pervasive in all tumors, including those with elevated mutational burdens. Given the quantity of drivers and passengers observed in a typical cancer (TCGA), our model implies that cancer cells are in total ~90% fitter than normal tissues (119% total benefit of drivers, 46% total cost of passengers). A median of five drivers each of which has a mean benefit of ~19% accumulate per tumor in these simulations – also consistent with estimates from age incidence curves (<xref ref-type="bibr" rid="bib48">National Cancer Institute, 2007</xref>), known hallmarks of cancer (<xref ref-type="bibr" rid="bib30">Hanahan and Weinberg, 2000</xref>), and estimates of the selective benefit of individual drivers (<xref ref-type="bibr" rid="bib14">Dai et al., 2007</xref>). Lastly, the mutation rates of tumors that could progress to cancer in our model also recapitulate observed mutation rates in human cancer (<xref ref-type="bibr" rid="bib9">Camps et al., 2007</xref>) (median 3.7×10<sup>–9</sup>, 95% interval 1.1×10<sup>–10</sup>–8.2×10<sup>–8</sup>, <xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>).</p><p>Most notably, under our modeling assumptions, all passengers together confer a fitness cost of ~46% per tumor. While this collective burden appears large, the individual fitness effects of accumulated passengers in these simulations (mean 0.8%) are similar to observed fitness costs in cancer cell lines (1–3%) (<xref ref-type="bibr" rid="bib63">Williams et al., 2008</xref>) and the human germline (0.5%) (<xref ref-type="bibr" rid="bib12">Cassa et al., 2017</xref>). Note that in our model, these passengers accumulated primarily via Muller’s ratchet, while only ~5% accumulated via hitchhiking inferred using population genetics theory (<xref ref-type="bibr" rid="bib42">McFarland et al., 2013</xref>) and MLE fitness effects, Materials and methods, <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplement 6</xref>. These results suggest that Hill-Robertson interference is a plausible model for the empirical patterns of attenuated selection with mutational burden observed in the data.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Here, we argue that signals of selection are largely absent in cancer because of the inefficiency of selection and not because of weakened selective pressures. In low mutational burden tumors (≤3 total substitutions per tumor), increased selection for drivers and against passengers is observed and ubiquitous: in SNVs and CNAs; in heterozygous, homozygous, clonal, and subclonal mutations; and in mutations predicted to be functionally consequential. These trends are not specific to essential or housekeeping genes. Importantly, these patterns persist across broad and specific tumor subtypes. Collectively, these results suggest that inefficient selection is generic to tumor evolution and that deleterious load is a nearly universal hallmark of cancer.</p><p>Importantly, these patterns of selection are missed when <italic>dN</italic>/<italic>dS</italic> rates are not stratified by mutational burden. Since &lt;0.1% of mutations in TCGA reside within low mutational burden tumors (~1% of all tumors, <italic>N</italic>=83), <italic>dN</italic>/<italic>dS</italic> in passengers at low mutational burdens (~0.56) does not appreciably alter pan-cancer <italic>dN</italic>/<italic>dS</italic> of passengers (0.97 in our study, 0.82–0.98 in <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>; <xref ref-type="bibr" rid="bib62">Weghorn and Sunyaev, 2017</xref>; <xref ref-type="bibr" rid="bib65">Zapata et al., 2018</xref>; <xref ref-type="bibr" rid="bib50">Ostrow et al., 2014</xref>). In fact, the power to detect negative selection on passengers at low mutational burdens is only possible by aggregating all mutations within these tumors and estimating <italic>dN</italic>/<italic>dS</italic> jointly. Thus, we believe that low mutational burden tumors are uniquely valuable for identifying genes and pathways under positive and negative selection. While only ~1% of tumors exhibit substantial negative selection, selection in drivers, selection on CNAs, and expression patterns of chaperones and proteasome components all show a continuous response to deleterious passenger load across a broad range of mutational burdens. Collectively, this suggests that passengers continue to be deleterious even in high mutational burden tumors.</p><p>Using a simple evolutionary model, we show that Hill-Robertson interference alone can explain this ubiquitous trend of attenuated selection in both drivers and passengers. <italic>dN</italic>/<italic>dS</italic> rates attenuate in drivers because the background fitness of a clone becomes more important than the fitness effects of an additional driver at elevated mutation rates. Furthermore, these simulations indicate that, despite <italic>dN</italic>/<italic>dS</italic> patterns approaching 1 in tumors with elevated mutational burdens, passengers are not effectively neutral (<italic>Ns</italic> &gt; 1). Instead, passengers confer an individually weak, but collectively substantial fitness cost of ~46% that measurably impacts tumor progression. Because this simple evolutionary model does not explicitly incorporate many known aspects of tumor biology (e.g. haploinsufficiency, see <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>), these fitness estimates are highly provisional. Nonetheless, we note that selection’s efficiency in cancer is further reduced when spatial constraints are considered (<xref ref-type="bibr" rid="bib53">Sottoriva et al., 2015</xref>).</p><p>The functional explanation for why passengers in cancer are deleterious is unknown. In germline evolution, mutations are believed to be primarily deleterious because of protein misfolding (<xref ref-type="bibr" rid="bib15">Drummond and Wilke, 2008</xref>; <xref ref-type="bibr" rid="bib39">Lobkovsky et al., 2010</xref>). Deleterious passengers in somatic cells should confer similar effects. Indeed, we find that elevated mutational burden tumors may buffer the cost of deleterious mutations by upregulating multiple heat shock pathways. However, deleterious passengers may carry other costs to cancers or be buffered by additional mechanisms. Understanding and identifying how tumors manage this deleterious burden should identify new cancer vulnerabilities that enable new therapies and better target existing ones (<xref ref-type="bibr" rid="bib26">Gorgoulis et al., 2018</xref>; <xref ref-type="bibr" rid="bib14">Dai et al., 2007</xref>; <xref ref-type="bibr" rid="bib24">Glaire and Church, 2017</xref>).</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Defining mutational burden in SNVs and binning tumors</title><p>Since TCGA is composed of whole-exome data, which limits our ability to accurately assess mutations in non-coding regions, we elected to use the total number of protein-coding mutations (i.e. missense, nonsense, and synonymous mutations) as our proxy for the mutational burden of tumors. This allows us to focus on the highest quality set of mutations that we have, which can impact the power to detect selection (<xref ref-type="fig" rid="fig2s15">Figure 2—figure supplement 15</xref>). We note that this high-quality set mutations does not have evidence of germline contamination by common SNPs (MAF &gt; 5%) from 1000 Genomes Project (<xref ref-type="bibr" rid="bib1">1000 Genomes Project Consortium et al., 2015</xref>) (v2015 Aug) using ANNOVAR (<xref ref-type="bibr" rid="bib60">Wang et al., 2010</xref>) to annotate mutations in TCGA (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). For all analyses calculating <italic>dN</italic>/<italic>dS</italic> in tumors stratified by their mutational burden, all variants within each bin of tumors were pooled together and <italic>dN</italic>/<italic>dS</italic> was calculated jointly on each bin of tumors. Counts of the number of mutations use to estimate <italic>dN</italic>/<italic>dS</italic> in each mutational burden can be found in <xref ref-type="fig" rid="fig2s16">Figure 2—figure supplement 16</xref>.</p></sec><sec id="s4-2"><title>A non-parametric null model of mutagenesis to calculate <italic>dN</italic>/<italic>dS</italic></title><p>We assume that for any particular tumor, mutation rates are constant across a gene for a particular tri-nucleotide context and base change (e.g. C&gt;G). Our procedure is inspired by constrained marginal models (or ‘edge switching’ in network analysis), whereby the marginal distributions of observations aggregated over known confounding variables are preserved under permutation to create a null distribution. In our application of this strategy, the marginal distributions of mutations (across tri-nucleotide context, base change, gene, and tumor) remain preserved – as they would be in a constrained marginal model; however, we exhaustively consider every acceptable permutation of the data. Because our approach is highly constrained, these permutations are exhaustively computable (median 36 alternatives per mutation). Thus, resampling is unnecessary.</p><p>Our null model presumes that all mutations of type <italic>i</italic>, defined by a tri-nucleotide context and base change, arise with probability <italic>M</italic><sub><italic>igt</italic></sub> within each gene <italic>g</italic> and tumor <italic>t</italic>. For each gene, we tally the total quantity of nonsynonymous mutations <italic>N</italic><sub><italic>ig</italic></sub> and synonymous mutations <italic>S</italic><sub><italic>ig</italic></sub>. Suppose selection enriches or depletes nonsynonymous mutations within a gene and tumor by a rate <italic>ω</italic><sub><italic>gt</italic></sub>. The expected number of nonsynonymous and synonymous mutations within a particular tumor and gene are <inline-formula><mml:math id="inf1"><mml:mi>E</mml:mi><mml:mfenced open="[" close="]" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow/></mml:msub><mml:mrow><mml:msub><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf2"><mml:mi>E</mml:mi><mml:mfenced open="[" close="]" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> in the absence of selective pressures on synonymous mutations. As with the main text, <italic>d</italic><sub><italic>N</italic></sub> and <italic>d</italic><sub><italic>N</italic></sub><sup>(observed)</sup> are used interchangeably. Although <italic>M</italic><sub><italic>igt</italic></sub> is unknown, <italic>dN</italic>/<italic>dS</italic> statistics attempt to infer selection nonetheless by noting that:<disp-formula id="equ2"><mml:math id="m2"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>ω</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&gt;</mml:mo></mml:mrow><mml:mrow><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&gt;</mml:mo></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mspace width="thinmathspace"/><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mspace width="thinmathspace"/><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>Note that <inline-formula><mml:math id="inf3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi><mml:mo>&gt;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:mi>A</mml:mi><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:mi>B</mml:mi><mml:mo symmetric="true">‖</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:mi>A</mml:mi><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mo>&lt;</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>A</mml:mi><mml:mo>&gt;</mml:mo></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mstyle></mml:math></inline-formula> is the Pearson product-moment correlation coefficient. When <italic>ρ</italic><sub><italic>MN</italic></sub> ≈ <italic>ρ</italic><sub><italic>MS</italic></sub>,<disp-formula id="equ3"><mml:math id="m3"><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mrow><mml:mfenced open="[" close="]" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mfenced open="‖" close="‖" separators="|"><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mrow><mml:mfenced open="[" close="]" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mfenced open="‖" close="‖" separators="|"><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mo>≈</mml:mo><mml:mi>ω</mml:mi></mml:mrow><mml:mrow/></mml:msub></mml:math></disp-formula></p><p>That is, <italic>dN</italic>/<italic>dS</italic> is approximately equal to the selective pressures on nonsynonymous mutations when the accessible nonsynonymous and synonymous loci are properly accounted and when the correlation between mutational processes and nonsynonymous loci are roughly equivalent to the correlation between mutational processes and synonymous loci. Traditionally, this assumption was used to calculate <italic>dN</italic>/<italic>dS</italic>. To improve resolution of <italic>dN</italic>/<italic>dS</italic>, researchers have attempted to account for these correlations using sophisticated parametric models of <italic>M</italic><sub><italic>igt</italic></sub>. An alternative statistical approach, however, is to treat these correlations as nuisance parameters.</p><p>Constrained marginal models permute observed data in all possible manners that preserve the underlying covariance structure of the data (e.g. <italic>ρ</italic><sub><italic>MN</italic></sub>, <italic>ρ</italic><sub><italic>MS</italic></sub>). In our particular case of this method, we note that by definition, <inline-formula><mml:math id="inf5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> . Thus:<disp-formula id="equ4"><mml:math id="m4"><mml:mrow><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:msubsup><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:msubsup><mml:mi>d</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow></mml:msubsup><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo symmetric="true">‖</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow/></mml:msub><mml:mo symmetric="true">‖</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>Hence, by dividing the observed mutations by all permutations, we eliminate the covariance of mutational processes with available loci and, thus, measure <italic>ω</italic><sub><italic>gt</italic></sub> directly for any particular gene-tumor combination without mutational bias.</p><p>Unfortunately, because of the log-sum inequality, mutational bias can arise once cohorts of genes and cohorts of tumor samples are binned. This problem is common to all <italic>dN</italic>/<italic>dS</italic> measures and is a consequence of the correlation of mutational biases with <italic>selection</italic> (i.e. <inline-formula><mml:math id="inf6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:mo>&gt;</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>) – not the correlation of mutational biases with one another, as these covariances are already accounted for in a constrained marginal model. For example, if tri-nucleotide biases covary linearly with gene-level biases, and are independent of tumor-level biases, then a parametric estimate of <italic>M</italic><sub><italic>igt</italic></sub> may deconstruct <italic>M</italic><sub><italic>igt</italic></sub> into <inline-formula><mml:math id="inf7"><mml:mi>M</mml:mi><mml:msub><mml:mrow/><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>ρ</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:math></inline-formula> , where <inline-formula><mml:math id="inf8"><mml:msub><mml:mrow><mml:mi>ρ</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the covariation of tri-nucleotide mutational biases with gene-level biases. Nonetheless, <inline-formula><mml:math id="inf9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:mo>&gt;∝&lt;</mml:mo><mml:msub><mml:mi>ρ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow/></mml:msub><mml:mo>&gt;</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> will still be ignored. Indeed, this covariation of mutational processes with selective forces is the focus of our current study: selection and genome-wide mutation rate are correlated (i.e. <inline-formula><mml:math id="inf10"><mml:mrow><mml:msub><mml:mo>∑</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow/></mml:msub><mml:mo>≠</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>) because of Hill-Robertson interference. Hence, the level at which observed <italic>d</italic><sub><italic>N</italic></sub> values <italic>d</italic><sub><italic>S</italic></sub> are binned necessarily ignores covariation between mutational processes and selection (in addition to any variation of <italic>ω</italic><sub><italic>gt</italic></sub> within the bin). Another example of this binning challenge arises when positive and negative selection act on different regions of the same gene, which gene-level <italic>dN</italic>/<italic>dS</italic> binning can misinterpret as neutral evolution.</p></sec><sec id="s4-3"><title>Validation of non-parametric null model</title><p>To confirm that our null model can accurately estimate <italic>dN</italic>/<italic>dS</italic> even in the presence of extreme tri-nucleotide mutational biases, we simulated artificial data where different COSMIC signatures (<xref ref-type="bibr" rid="bib56">Tate et al., 2019</xref><xref ref-type="bibr" rid="bib19">Forbes et al., 2008</xref>; <xref ref-type="bibr" rid="bib19">Forbes et al., 2008</xref>) (SBS Signatures 1–9, v3) contribute to all of the mutations. Permuted <italic>d</italic><sub><italic>N</italic></sub> and <italic>d</italic><sub><italic>S</italic></sub> tallies for each mutational context were simulated by randomly sampling 1000 genes with the same mutational context. The fraction of permuted <italic>d</italic><sub><italic>N</italic></sub> and <italic>d</italic><sub><italic>S</italic></sub> tallies for each mutational context was used as weighted probabilities to derive observed <italic>d</italic><sub><italic>N</italic></sub> and <italic>d</italic><sub><italic>S</italic></sub> tallies. To simulate negative selection, <italic>d</italic><sub><italic>N</italic></sub> counts were randomly removed from each context at a rate 1 − <italic>ω</italic><sub><italic>gt</italic></sub> (e.g. a simulated ‘true’ <italic>dN</italic>/<italic>dS</italic> of 0.8 in a cohort of samples indicates a 20% chance of nonsynonymous mutations being removed in the samples). These simulated (true) rates were then compared to observed and permuted <italic>d</italic><sub><italic>N</italic></sub> and <italic>d</italic><sub><italic>S</italic></sub> tallies according to the <italic>dN</italic>/<italic>dS</italic> metric that we used throughout this study:<disp-formula id="equ5"><mml:math id="m5"><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mtext>(observed)</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mtext>(permuted)</mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mtext>(observed)</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mtext>(permuted)</mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>We confirmed that this approach accurately measures selection in the presence of simulated mutational biases (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>).</p><p>Lastly, we note that binning nonsynonymous and synonymous mutations at the genome-wide level (e.g. drivers and passengers) provided the most robust estimates of <italic>dN</italic>/<italic>dS</italic> when bootstrapping observed tumor samples. Statistical power is insufficient when binning at the individual gene level. Bootstrapping also demonstrated that log-transformation of <italic>dN</italic>/<italic>dS</italic> values increases statistical power, and thus was generally applied to <italic>dN</italic>/<italic>dS</italic> analyses in this study.</p></sec><sec id="s4-4"><title>A parametric null model of mutagenesis</title><p>For comparison, we also calculated <italic>dN</italic>/<italic>dS</italic> using <italic>dNdScv</italic> (<xref ref-type="bibr" rid="bib8">Campbell and Martincorena, 2017</xref>) – a previously published parametric null model of mutagenesis in cancer (<xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>). To compare both methods, <italic>dNdScv</italic> ran globally and separately on samples stratified by the total number of substitutions using the following parameters: max_coding_muts_per_sample = Inf max_muts_per_gene_per_sample = Inf.</p><p>Global <italic>dN</italic>/<italic>dS</italic> values of all nonsynonymous mutations (<italic>w</italic><sub><italic>all</italic></sub>, reported by <italic>dNdScv</italic>) were used. This model reproduced our non-parametric <italic>dN</italic>/<italic>dS</italic> trends (<xref ref-type="fig" rid="fig2">Figure 2A</xref>) and was used to infer patterns of selection in synonymous mutations (<xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>). We note that stratifying tumors in TCGA into 20 bins of equal sample size (as was done in <xref ref-type="bibr" rid="bib41">Martincorena et al., 2017</xref>), rather than evenly spaced bins, averages out a significant proportion of the negative selection observed in passengers, since low mutation burden tumors reside within the tail-end of the distribution (<xref ref-type="fig" rid="fig2s7">Figure 2—figure supplement 7</xref>).</p></sec><sec id="s4-5"><title>Identification of driver genes in cancer</title><p>For all analysis using SNVs, unless explicitly stated, a comprehensive list of 299 pan-cancer driver genes derived from 26 computational tools was used to catalog driver genes (<xref ref-type="bibr" rid="bib6">Bailey et al., 2018</xref>). Other pan-cancer driver gene sets tested were derived from COSMIC’s Driver Gene Census (<xref ref-type="bibr" rid="bib56">Tate et al., 2019</xref>; <xref ref-type="bibr" rid="bib19">Forbes et al., 2008</xref>) (downloaded on October 2016) and IntOGen’s Cancer Drivers Database (<xref ref-type="bibr" rid="bib25">Gonzalez-Perez et al., 2013</xref>) (v2014.12) which contained 602 and 459 number of driver genes, respectively.</p><p>Many driver genes are associated with only particular tumor subtypes. To compare patterns of selection across cancer subtypes without increasing or decreasing the size of the list for each subtype, we chose to use a single set of driver genes for most analyses. This may understate the degree of positive selection in driver genes as mutations in these genes may be passengers in some tumor subtypes. In <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>, we investigate patterns of selection using the top 100 driver genes identified for each tumor type and observe decreased signatures of positive selection overall in driver genes. Nevertheless, the patterns of attenuated selection in drivers and passengers remain. While tissue-type specific driver genes certainly exist, our results suggest that our statistical power to detect drivers still remains too limited to justify subdividing analyses by tumor type in many cases.</p><p>For all CNA analysis, GISTIC 2.0 <xref ref-type="bibr" rid="bib45">Mermel et al., 2011</xref> was used to identify a set of genomic regions enriched for copy number gains and copy number losses using recommended settings with a confidence threshold of 0.9. CNAs used to identify these peaks were downloaded from the NIH Genomic Data Commons (GDC) (<xref ref-type="bibr" rid="bib27">Grossman et al., 2016</xref>) in the TCGA cohort. For each amplification peak, the closest gene was annotated as a putative oncogene, and similarly the closest gene to each deletion peak was annotated as a putative tumor suppressor. The top 100 amplification peaks (oncogenes) and deletion peaks (tumor suppressors) were classified as drivers for each of the 32 tumor types. Thirty-four percent of identified driver genes appear in more than one tumor type, while 2.6% of identified driver genes appear in more than five tumor types.</p><p>For both SNV and CNA analysis, passengers were defined as mutations that did not reside within driver genes. The vast majority of mutations are passengers, and their relative totals for both SNVs and CNAs are depicted in <xref ref-type="fig" rid="fig2s16">Figure 2—figure supplement 16</xref>.</p></sec><sec id="s4-6"><title>Annotation of clonal and subclonal mutations</title><p>VAFs were calculated per site as the number of mutant read counts divided by the total number of read counts. VAFs were adjusted for purity using calls made by ABSOLUTE (<xref ref-type="bibr" rid="bib27">Grossman et al., 2016</xref>; <xref ref-type="bibr" rid="bib11">Carter et al., 2012</xref>), collected from GDC. A VAF threshold of 0.2 was used to define ‘subclonal’ (&lt;0.2) vs. ‘clonal’ (&gt;0.2) SNVs. Different VAF thresholds were considered (<xref ref-type="fig" rid="fig2s10">Figure 2—figure supplement 10</xref>) and the choice of ‘clonal’ thresholding did not impact the conclusions of this study.</p></sec><sec id="s4-7"><title>PolyPhen2 analysis</title><p>PolyPhen2 annotations in the MC3 SNP calls were used (<xref ref-type="bibr" rid="bib2">Adzhubei et al., 2010</xref>). Only missense mutations that were categorized as either ‘benign’, ‘probably damaging’, or ‘possibly damaging’ were used. The fraction of pathogenic missense mutations was calculated as the number of pathogenic mutations categorized as either ‘probably damaging’ or ‘possibly damaging’ divided by the total number of categorized mutations.</p></sec><sec id="s4-8"><title>Classification of genes by functional category</title><p>To test for patterns of selection in functionally related genes, we annotated all mutations by different functional categories and GO terms (<xref ref-type="bibr" rid="bib32">Harris et al., 2004</xref>). Oncogenes and tumor suppressors were annotated from a curated set of 99 high confidence cancer genes (<xref ref-type="bibr" rid="bib38">Kumar et al., 2015</xref>). Essential genes were collected from a genome-wide CRISPR screen that identified genes required for proliferation and survival in a human cancer cell line (<xref ref-type="bibr" rid="bib61">Wang et al., 2015</xref>). Housekeeping genes were defined as genes with an exon that is expressed in all tissues at any non-zero level, and exhibits a uniform expression level across tissues (<xref ref-type="bibr" rid="bib17">Eisenberg and Levanon, 2015</xref>). Interacting proteins were downloaded from the mentha database in April 2019 (<xref ref-type="bibr" rid="bib7">Calderone and Cesareni, 2012</xref>).</p><p>To identify highly expressed genes, median transcripts per million (TPM) in 54 tissue types (v7 release) were downloaded from the Genotype-Tissue Expression (GTEx) project (<xref ref-type="bibr" rid="bib28">GTEx Consortium, 2020</xref>; <xref ref-type="bibr" rid="bib10">Carithers and Moore, 2015</xref>). Tissues that contained high expression in most genes, specifically testes, were removed. Only genes that had TPM counts above zero in any of the 53 remaining tissues were used. TPM counts were averaged across all tissues. Highly expressed genes were defined as the top 1000 genes expressed across all tissues.</p><p>To test for signals of negative selection in other functional groups, we annotated mutations by candidate GO terms according to biological processes: Transcription Regulation (GO Term ID: 0140110), Translation Regulation (GO Term ID: 0045182), and Chromosome Segregation (GO Term ID: 0007059).</p></sec><sec id="s4-9"><title>Somatic CNAs</title><p>All CNAs were downloaded from the COSMIC database on June 2015 (<xref ref-type="bibr" rid="bib56">Tate et al., 2019</xref>; <xref ref-type="bibr" rid="bib19">Forbes et al., 2008</xref>). Mitochondrial CNAs were discarded from analysis, as copy number changes are difficult to infer. Gene annotations and the locations of telomeres and centromeres were downloaded from the UCSC Genome Browser (hg19). Telomeric and centromeric regions were masked from all measurements of <italic>dE</italic>/<italic>dI</italic>. Because the selection patterns of non-focal CNAs – alterations with at least one terminus in a telomere or centromeric region – were not noticeably different from long (&gt;100 kb) focal CNAs, these two alteration classes were aggregated for analysis. Notably, we observed positive selection for both amplifications and deletions within oncogenes, and for both deletions and amplifications within tumor suppressors. For this reason, we did not distinguish between gains and losses, nor oncogenes and tumor suppressors in published analyses: any CNA that overlapped an oncogene or tumor suppressor in any region (for any fraction of the CNA) was classified as a driver. Mutational burden was defined simply as the total number of CNAs within a sample. Pan-cancer CNAs from cBioPortal (August 2018) were also analyzed, however consistent purity and ploidy estimates could not be obtained by using either ABSOLUTE (<xref ref-type="bibr" rid="bib11">Carter et al., 2012</xref>) or TITAN (<xref ref-type="bibr" rid="bib29">Ha et al., 2015</xref>), so this data was not used for published analyses of CNAs.</p></sec><sec id="s4-10"><title>Measurements of selection on CNAs</title><p><italic>dE</italic>/<italic>dI</italic> was calculated using a ‘breakpoint frequency’ metric and a ‘fractional overlap’ metric. For both metrics, the <italic>dE</italic>/<italic>dI</italic> of a particular gene set <italic>i</italic> (e.g. driver or passenger genes) is defined by a genomic track <italic>T<sub>i,g</sub>,</italic> which is one for every annotated region <italic>g</italic> of the track and zero elsewhere. Only non-centromeric and non-telomeric regions are considered in the mappable human genome <italic>G</italic>. Each CNA <italic>C</italic><sub><italic>g</italic>,<italic>m</italic></sub> is defined by its position on the genome <italic>g</italic> and the mutational burden <italic>m</italic> of the tumor harboring the mutation. For ‘breakpoint frequency’ <italic>C</italic><sub><italic>m</italic>,<italic>i</italic></sub> is one at the position of both termini of the CNA and zero elsewhere. For ‘fractional overlap’ <italic>C</italic><sub><italic>m</italic>,<italic>i</italic></sub> is 1/<italic>L</italic>, where <italic>L</italic> is the length of the CNA, for every region of the genome spanned by the CNA and zero elsewhere. For a particular range of mutational burdens <italic>M</italic>, <italic>dE</italic>/<italic>dI</italic> was defined as:<disp-formula id="equ6"><mml:math id="m6"><mml:msub><mml:mrow><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>We note that calculation is accelerated by &gt;×100 by commuting <italic>T</italic><sub><italic>i</italic>,<italic>g</italic></sub> with the outer summation (Σ<italic><sub>m</sub><sup>M</sup></italic>). Lastly, we randomly permuted the start and stop positions of each CNA, while preserving its length, to derive a set of neutral CNAs not experiencing selection. This permutation analysis finds that <italic>dE</italic>/<italic>dI</italic> for both breakpoint frequency and fractional overlap is ~1 in the absence of selection (<xref ref-type="fig" rid="fig2s8">Figure 2—figure supplement 8</xref>).</p></sec><sec id="s4-11"><title>Tumor purity analysis in TCGA samples</title><p>Tumor purity estimates from the ABSOLUTE algorithm (<xref ref-type="bibr" rid="bib11">Carter et al., 2012</xref>) were downloaded from the GDC on May 2020. To evaluate the effects of tumor purity on patterns of selection, tumors below increasing thresholds of tumor purity were removed from the analysis, and <italic>dN</italic>/<italic>dS</italic> was calculated on tumors stratified by mutational burden bins (as described above.)</p></sec><sec id="s4-12"><title>Expression analysis</title><p>Gene expression data was downloaded from the COSMIC database on September 2019. Genes used to identify different protein folding pathways were downloaded from <xref ref-type="bibr" rid="bib36">Kampinga et al., 2009</xref>, genes involved in protein degradation pathways were identified from <xref ref-type="bibr" rid="bib55">Tanaka, 2009</xref>. The median gene expression of all genes in each protein folding pathway was used. Patients were binned by the total number of substitutions (using MC3 SNP calls from TCGA) and CNAs, and the average gene expression of each bin was calculated.</p></sec><sec id="s4-13"><title>Cancer subtype analysis</title><p>All tumor subtypes in were grouped into nine sub-categories, based on broad, predominantly anatomical features. Anatomical features (i.e. organ and systems of organs), rather than histological features or inferred cell-of-origin, were used as groupings because we believe that the fitness effects of mutations should be predominantly defined by the environment of the tumor. Nevertheless, we observed attenuated selection in both drivers and passengers in many broad histologically defined classifications (e.g. adenocarcinomas and sarcomas). For all cancer grouping analysis (broad and subtype), tumors were stratified into bins by the total number of substitutions (<italic>d</italic><sub><italic>N</italic></sub>+<italic>d</italic><sub><italic>S</italic></sub>) on a log-scale. Since tumor subtypes vary in their range of mutational burdens, (e.g. KIRC cancer subtypes only have tumors with &lt;100 substitutions), <italic>dN</italic>/<italic>dS</italic> values in the lowest and highest mutational burden bin for each cancer subtype are shown.</p><p>Specific cancer subtype categories were taken directly from the NCI GDC (<xref ref-type="bibr" rid="bib27">Grossman et al., 2016</xref>). Because CNAs were downloaded from COSMIC, CNA datasets were not classified with this same ontology. <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref> details how CNA classifications were mapped on GDC categories (and sometimes more broadly defined groups). All subtypes with &gt;200 samples were used in our CNA subtype analyses (<xref ref-type="fig" rid="fig2s13">Figure 2—figure supplement 13</xref>).</p></sec><sec id="s4-14"><title>An evolutionary model with Hill-Robertson interference</title><p>Somatic cells in our populations are modeled as individual cells that can stochastically divide and die in a first-order (memoryless) Gillespie algorithm. This model was developed and described previously (<xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>). During division, cells can acquire advantageous drivers with rate <italic>µT</italic><sub><italic>drivers</italic></sub> and deleterious passengers with rate <italic>µT</italic><sub><italic>passengers</italic></sub> – these values specify the mean of Poisson-distributed pseudo-random number (PRN) generators that prescribe the number of drivers and passengers conferred during division (e.g. the number of drivers per division <italic>n</italic><sub><italic>d</italic></sub> = Poisson[<italic>n</italic><sub><italic>d</italic></sub> = <italic>k</italic>; <italic>λ</italic> = <italic>µT</italic><sub><italic>drivers</italic></sub>] = <italic>λ<sup>k</sup> e<sup>−k</sup></italic>/<italic>k</italic>!). The DFEs conferred by each driver and each passenger are exponentially distributed PRNs with probability densities <italic>P</italic>(<italic>s</italic><sub><italic>i</italic></sub> = <italic>x</italic>; <italic>s</italic><sub><italic>drivers</italic></sub>)=Exp[<italic>−x/s</italic><sub><italic>drivers</italic></sub>]/<italic>s</italic><sub><italic>drivers</italic></sub> and <italic>P</italic>(<italic>s</italic><sub><italic>i</italic></sub> = <italic>x</italic>; <italic>s</italic><sub><italic>passengers</italic></sub>) = −Exp[<italic>−x</italic>/<italic>s</italic><sub><italic>passengers</italic></sub>]/<italic>s</italic><sub><italic>passengers</italic></sub>, respectively. Simulations with other exponential-family DFEs do not qualitatively differ from these exponential distributions (<xref ref-type="bibr" rid="bib42">McFarland et al., 2013</xref>). The aggregate absolute cellular fitness is <inline-formula><mml:math id="inf11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>f</mml:mi><mml:mo>=</mml:mo><mml:munderover><mml:mo>∏</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> in our multiplicative epistasis model and <inline-formula><mml:math id="inf12"><mml:mi>Δ</mml:mi><mml:mi>f</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>ν</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:mfenced></mml:mrow></mml:mrow></mml:math></inline-formula> with <italic>ν</italic>=1 in our diminishing-returns epistasis model, where Δ<italic>f</italic> is the change in cellular fitness with each mutation (<xref ref-type="bibr" rid="bib4">Arjan et al., 1999</xref>). The rate of cell birth is inversely proportional to cellular fitness, while the rate of cell death <inline-formula><mml:math id="inf13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>D</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext>Log</mml:mtext><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>e</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> increases with the population size of the tumor <italic>N</italic>. With these birth and death processes, mean population size abides by a Gompertzian growth law in the absence of additional mutations, which is scaled by the mean cellular fitness E[<italic>N</italic>(&lt;<italic>f</italic> &gt; )]=log[1 +&lt;<italic>f</italic> &gt;/ <italic>N</italic> <sup>0</sup>] (derived from master equation <xref ref-type="bibr" rid="bib42">McFarland et al., 2013</xref>). While, programmatically, mutations exclusively affect the birth rate and the constraints on growth exclusively affect the death rate, we previously demonstrated that birth and death rates are generally nearly balanced such that dynamics are not affected by this design choice.</p><p>Because somatic cells do not recombine during cell division, dominance coefficients were not explicitly modeled. Thus in diploid cancers, our selection coefficients estimate the mean heterozygous effect of drivers and passenger (i.e. <italic>hs</italic>). Similarly, loss of heterozygosity (LOH) events (gene losses, gene conversions, mitotic recombination, etc.) are not explicitly modeled either; however, these events can be viewed as additional mutations that may be either adaptive drivers or deleterious passengers in the model. As sequencing data improves, we believe that it will be informative to explicitly model dominance coefficients, tumor ploidy, and LOH events.</p><p>Simulations progressed until tumor extinction (<italic>N</italic>=0 cells), malignant transformation (<italic>N</italic>=10<sup>6</sup> cells), or until ~100 years had passed (18,500 generations). Only fixed mutations (present in the most recent common ancestor) within clinically detectable growths were analyzed in our ABC pipeline. The behavior of this model has been described previously (<xref ref-type="bibr" rid="bib42">McFarland et al., 2013</xref>; <xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>) and the most relevant assumptions of this model and their effects on the conclusions of this study are described in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>.</p><p>Cells in our populations are fully described by their accrued mutations, and birth and death times. Birth and death events were modeled using an implementation of the next reaction (<xref ref-type="bibr" rid="bib23">Gibson and Bruck, 2000</xref>), a Gillespie algorithm that orders events using a heap queue. Generation time in our model was defined as the inverse of the mean birth rate of the population: 1/&lt;<italic>B</italic>(<italic>d</italic>, <italic>p</italic>)&gt;. While all mutation events occurred during cell division, if mutations were to occur per unit of time (rather than per generation), rapidly growing tumors would acquire drivers at a slightly slower rate as generation times decline over time. This effect, however, is negligible compared to the variation in waiting times conferred by the variation in mutation rates (division times merely double, while mutation rates vary by 100,000-fold).</p><p>This simple evolutionary model is defined by five parameters <italic>µT</italic><sub><italic>drivers</italic></sub>, <italic>µT</italic><sub><italic>passengers</italic></sub>, <italic>s</italic><sub><italic>drivers</italic></sub>, <italic>s</italic><sub><italic>passengers</italic></sub>, and <italic>N</italic><sup>0</sup>. The target size of drivers is defined as the approximate number of nonsynonymous mutations in the Bailey driver screen <italic>T</italic><sub><italic>drivers</italic></sub> = (# of driver genes)·(mean driver length)·(fraction of SNVs that are nonsynonymous)=300 genes · 1298 loci/gene · 0.737 nonsynonymous loci/ loci = 286,886 nonsynonymous loci. The target size of passengers was simply the remaining loci in the protein coding genome, <italic>T</italic><sub><italic>passengers</italic></sub> = 20,451,136 nonsynonymous loci. The mutation rate was constant throughout each tumor simulation and randomly sampled from a uniform distribution in log-space that ranged from 10<sup>–12</sup> to 10<sup>–7</sup> mutations·loci<sup>–1</sup>·generation<sup>–1</sup>. While tumors were initiated from this broad range, malignancies (<italic>N</italic>&gt;10<sup>6</sup> cells) were almost always restricted to mutation rates between 10<sup>–10</sup> and 10<sup>–8</sup> (<xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>), as tumors with mutation rates drawn below this range almost never progressed to cancer within 100 years and tumors with mutation rates drawn above this range went extinct through natural selection.</p><p>The likelihood that tumors progress to cancer in the presence of deleterious passengers depends heavily on the initial population size <italic>N</italic><sup>0</sup> of the tumor. This dependence was studied previously (<xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>), where it was demonstrated that reasonable evolutionary simulations (those that progress to cancer &gt;10% of the time, but &lt;90% of the time) are restricted to a four-dimensional manifold <italic>N*</italic> within the five-dimensional phase space of parameters. For this reason, <italic>N<sup>0</sup></italic>=<italic>N*</italic>(<italic>s</italic><sub><italic>drivers</italic></sub>, <italic>s</italic><sub><italic>passengers</italic></sub>, <italic>µT</italic><sub><italic>drivers</italic></sub>, <italic>µT</italic><sub><italic>passengers</italic></sub>) was determined by the other four parameters. To first order, this manifold is <italic>T</italic><sub><italic>passengers</italic></sub><italic>s</italic><sub><italic>paassengers</italic></sub>/ (<italic>T</italic><sub><italic>drivers</italic></sub> <italic>s</italic><sub><italic>drivers</italic></sub>) (<xref ref-type="bibr" rid="bib62">Weghorn and Sunyaev, 2017</xref>), however a more precise estimate (Equation S8 of <xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>) incorporating more precise estimates of Muller’s ratchet and the effects of hitchhiking on both driver and passenger accumulation rates, which does not exist in closed form was used. Additionally, at very low values of <italic>s</italic><sub><italic>drivers</italic></sub>, progression to cancer is limited by time, not by the accumulation of deleterious passengers. Hence, we assigned <italic>N</italic><sup>0</sup> such that:<disp-formula id="equ7"><mml:math id="m7"><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn><mml:mo>,</mml:mo><mml:mover><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo accent="false">¯</mml:mo></mml:mover><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>18</mml:mn><mml:mo>,</mml:mo><mml:mn>500</mml:mn><mml:mspace width="thinmathspace"/><mml:mtext>g</mml:mtext><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></disp-formula></p><p>Here, <italic>P</italic><sub><italic>cancer</italic></sub> and <italic>t</italic><sub><italic>cancer</italic></sub> – the likelihood and waiting time to cancer – are defined byEquation S8 and S12 respectively in <xref ref-type="bibr" rid="bib43">McFarland et al., 2014</xref>. <italic>N</italic><sup>0</sup> was determined from these equations using Brent’s method. <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref> depicts the values of <italic>N</italic><sup>0</sup>, which ranged from 1 to 100 for all simulations.</p><p>In tumors that progress to malignancy (<italic>N</italic>=10<sup>6</sup>), only fixed nonsynonymous mutations (present in all simulated cells) were recorded. We also recorded (i) the fitness effect of these mutations, (ii) the mean population fitness, (iii) the number of generations until malignancy, and (iv) the mutation rate. These two values were used to generate the number of synonymous drivers and passengers, where <italic>P</italic>(<italic>d</italic><sub><italic>s</italic></sub> <italic>= k</italic>)=Poisson[<italic>k</italic>; <italic>λ</italic> = <italic>µT</italic><sub><italic>drivers</italic>/<italic>passengers</italic></sub>/<italic>r t</italic><sub><italic>MRCA</italic></sub>] defines the number of synonymous drivers/passengers conferred, <italic>t</italic><sub><italic>MRCA</italic></sub> represents the number of division until the most recent common ancestor arose in the simulation, <italic>r</italic>=2.795 represents the ratio of nonsynonymous to synonymous loci within the genome, weighted by the genome-wide tri-nucleotide somatic mutation rate, and the Poisson PRN generator was defined above. In simulations where synonymous drivers could arise, a fraction of the recorded nonsynonymous mutations (ranging from 0% to 20%) were simply re-labeled as synonymous drivers (as opposed to nonsynonymous drivers). This was done, again, by Poisson sampling in proportion to the desired fraction for each cancer simulation.</p><p>20×20 combinations of <italic>s</italic><sub><italic>drivers</italic></sub> and <italic>s</italic><sub><italic>passengers</italic></sub> parameters were simulated (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplements 1</xref>–<xref ref-type="fig" rid="fig3s2">2</xref>). Simulations were repeated until 10,000 cancers at each parameter combination were obtained or until 10 million tumor populations were simulated. While we attempted to initiate tumors at a population size where the probability of progression to cancer was 50%, some parameter combinations still did not yield 10,000 cancers after 10 million attempts (i.e. <italic>P</italic><sub><italic>cancer</italic></sub> &lt; 0.1%). These combinations were predominately at low values of <italic>s</italic><sub><italic>drivers</italic></sub>, which were far from the MLE estimate of <italic>s</italic><sub><italic>drivers</italic></sub> and represent unrealistic evolutionary scenarios: drivers cannot be weakly beneficial, relegated to only 300 genes, and still overcome deleterious passengers within 100 years. These simulations are annotated as ‘progression impossible’. Simulation parameter sweeps were performed for both the multiplicative and diminishing returns epistasis models. Twenty fractions of synonymous drivers were also generated (ranging from 0% to 20%). These fractions were generated by simply re-labeling the driver mutations which conferred fitness (generated during the simulation) as synonymous, instead of nonsynonymous.</p></sec><sec id="s4-15"><title>Summary statistics of simulated and observed tumors</title><p>For both simulated and observed data, we summarized <italic>dN</italic>/<italic>dS</italic> rates vs. mutational burden for drivers and for passengers by decade-sized bins: (0, 10], (10, 100], (100, 1,000]. Mutational burden for simulations was defined as the total number of substitutions (<italic>d</italic><sub><italic>N</italic></sub>+<italic>d</italic><sub><italic>S</italic></sub>) – exactly as it was defined for observed data. For simulated data, <italic>dN</italic>/<italic>dS</italic> = <italic>d</italic><sub><italic>N</italic></sub>/(<italic>d</italic><sub><italic>S</italic></sub> · <italic>r</italic>). Like observed data, <italic>dN</italic>/<italic>dS</italic> rates attenuated toward 1 for both drivers and passengers for all values of <italic>s</italic><sub><italic>drivers</italic></sub> and <italic>s</italic><sub><italic>passengers</italic></sub>.</p><p>Mutational burdens (MB) for simulated and observed data were summarized with the parameters of a negative binomial distribution, where <inline-formula><mml:math id="inf14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>MB</mml:mtext><mml:mo>=</mml:mo><mml:mi>k</mml:mi><mml:mo>;</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:msup><mml:mi>p</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula> . This distribution has been used previously to summarize the MB of human tumors (<xref ref-type="bibr" rid="bib59">Turajlic et al., 2012</xref>) and exactly defines the expected number of mutations at transformation in a multi-stage model of tumorigenesis (<xref ref-type="bibr" rid="bib20">Frank, 2007</xref>) when <italic>n</italic> drivers are needed for transformation and the probability that any mutation be a driver is 1 – <italic>p</italic> (<xref ref-type="bibr" rid="bib47">Michor et al., 2005</xref>). Both <italic>n</italic> and <italic>p</italic> were used to summarize MB. These quantities were determined by maximum likelihood optimization of the probability mass function above over the support of mutational burdens of [1, 1,000] substitutions. The Han-Powell quasi-Newton least-squares method was used for optimization.</p><p>Age-dependent cancer incidence rates (CI) were summarized with the parameters of a gamma distribution, where <inline-formula><mml:math id="inf15"><mml:mi>P</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>C</mml:mi><mml:mi>I</mml:mi><mml:mo>≤</mml:mo><mml:mi>t</mml:mi><mml:mo>;</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>θ</mml:mi></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>Γ</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac><mml:mi>γ</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mfrac><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>θ</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced></mml:math></inline-formula> . Here, <inline-formula><mml:math id="inf16"><mml:mi>γ</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo>∫</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msup><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> is the lower incomplete gamma function and Γ(<italic>k</italic>) = <italic>γ</italic>(<italic>k</italic>, ∞) is the regular gamma function. Similar to our summarization of mutational burdens, this distribution is a generalization of the exact waiting time to transformation expected from a multi-stage model of tumorigenesis when tumors arise at a uniform rate over time, require <italic>k</italic> drivers for transformation, and wait an average time of <italic>θ</italic> between drivers (<xref ref-type="bibr" rid="bib47">Michor et al., 2005</xref>). This cumulative distribution function was fit to observed incidence rates for all patients above 20 years of age using the least squares numerical optimization defined above (all cancer sites combined, both sexes, all races, 2012–2016; <xref ref-type="bibr" rid="bib34">Howlader, 2013</xref>). Patients under 20 years of age were excluded because cancers in these patients generally arise from germline predispositions to cancer, which are (i) not directly modeled by our simulations, (ii) not detected as somatic mutations, and (iii) result in age incidence curves that do not agree with a gamma distribution (<xref ref-type="bibr" rid="bib20">Frank, 2007</xref>). Because all cancer simulations are initiated at <italic>t=0</italic> (instead of uniformly in time, as is presumed in the multi-stage model), the simulated data was fit using the probability density function of this distribution (instantaneous derivative) using maximum likelihood and the optimization algorithm described above. The cumulative distribution, then, represents the expected age incidence cancer incidence rate when simulations begin at uniformly distributed moments in time and, thus, was used to generate <xref ref-type="fig" rid="fig3">Figure 3D</xref>. Only the shape parameter <italic>k</italic> was used in ABC (and <italic>θ</italic> was ignored), as this parameter only specifies the dimensionality of time (simulation time was measured in cellular generations, not years) and all values of <italic>θ</italic> in our simulations are equivalent under a gauge transformation. Additionally, we do not expect the exact times of incidence to be particularly informative as the time of transformation is generally somewhat earlier than the time of detection.</p></sec><sec id="s4-16"><title>Use of ABC for model selection and parameter inference</title><p>Like many Bayesian analyses, the main steps of an ABC analysis scheme are: (i) formulating a model, (ii) fitting the model to data (parameter estimation), and (iii) improving the model by checking its fit (posterior-predictive checks), and (iv) comparing this model to other models (<xref ref-type="bibr" rid="bib13">Csilléry et al., 2012</xref>; <xref ref-type="bibr" rid="bib22">Gelman, 2004</xref>).</p><p>The nine summary statistics described above were used to compare simulations to observed data. Agreement was summarized with a log-Euclidean distance, as all summary statistics resided on the domain [0, ∞) and log-transformation of the summary statistics minimized heteroscedasticity of the simulated data relative to a square-root or no transformation. Variance of the summary statistics was not normalized. ABC was performed using the ‘abc’ R package (<xref ref-type="bibr" rid="bib13">Csilléry et al., 2012</xref>).</p><p>The rejection method (feedforward neural net) and tolerance (0.5) were chosen based on their capacity to minimize prediction error of the simulated data using leave-one-out CV (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>). Ten-thousand instances of the neural network, which was restricted to a single layer, were initiated and the median prediction of these networks were used. These parameters were used for both model comparison and parameter inference. For parameter inferencing, the <italic>s</italic><sub><italic>drivers</italic></sub> and <italic>s</italic><sub><italic>passengers</italic></sub> prior values were log-transformed.</p><p>For the synonymous driver model, the base model (without synonymous drivers) was simply the lowest quantity of synonymous drivers (0%) in the parameter sweep of synonymous driver quantities (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3B</xref>). The posterior probability mass of this value 0.017 was used as the one-sided p-value for the null hypothesis that these two models are equally predictive. Although the synonymous driver model agreed with the observed data slightly better, <italic>s</italic><sub><italic>drivers</italic></sub> and <italic>s</italic><sub><italic>passengers</italic></sub> parameters could not be inferred from the data because the potential for synonymous drivers destroys the utility of a <italic>dN</italic>/<italic>dS</italic> statistics, which is predicated on the notion that synonymous mutations are neutral. Virtually any value of <italic>dN</italic>/<italic>dS</italic> is attainable when the right combinations of selective pressures on nonsynonymous and synonymous are paired (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3C</xref>).</p></sec><sec id="s4-17"><title>Code availability</title><p>All code for empirical analysis and generation of summary statistics are publicly available under the open-source MIT License at <ext-link ext-link-type="uri" xlink:href="https://github.com/petrov-lab/cancer-HRI">https://github.com/petrov-lab/cancer-HRI</ext-link>. (<xref ref-type="bibr" rid="bib57">Tilk, 2022a</xref> copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:e4a6f9d73932cdce98d9de0cc1394b45d41252ec;origin=https://github.com/petrov-lab/cancer-HRI;visit=swh:1:snp:2996ca9f1c1709095fb538f591d3ede58d5eac3a;anchor=swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75">swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75</ext-link>). Code for simulations of tumor growth with advantageous drivers and deleterious passengers is also available at <ext-link ext-link-type="uri" xlink:href="https://github.com/mirnylab/pdSim">https://github.com/mirnylab/pdSim</ext-link>, (<xref ref-type="bibr" rid="bib58">Tilk, 2022b</xref> copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:2837cecedb2ca8992d5afa9ee54102a1e8fa61b6;origin=https://github.com/mirnylab/pdSim;visit=swh:1:snp:9023cfe7eac951feac42e3bd4133ce866a0fe0f0;anchor=swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4">swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4</ext-link>).</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Validation</p></fn><fn fn-type="con" id="con3"><p>Supervision, Funding acquisition, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Supervision, Funding acquisition, Validation, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con5"><p>Conceptualization, Software, Formal analysis, Supervision, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-67790-mdarchecklist1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Model assumptions of tumor evolution and its anticipated effects.</title></caption><media xlink:href="elife-67790-supp1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Broad (meta-categories) of cancer groupings used in Figure 2 and Figure 2—figure supplement 12-13.</title></caption><media xlink:href="elife-67790-supp2-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Exonic, open-access SNV calls (WES) of 10,486 cancer patients in (The Cancer Genome Atlas) TCGA were downloaded from the Multi-Center Mutation Calling in Multiple Cancers (MC3) project. This repository uses a consensus of seven mutation-calling algorithms. Expression data of SNVs were downloaded from the Genotype-Tissue Expression (GTEx) project (v7 release). All CNAs were downloaded from the COSMIC database on June 2015.Gene expression data compared to CNAs was downloaded from the COSMIC database on 14 September 2019.</p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset1"><person-group person-group-type="author"><name><surname>Ellrott</surname><given-names>K</given-names></name><name><surname>Bailey</surname><given-names>MH</given-names></name><name><surname>Saksena</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>TCGA - MC3 mutation calls</data-title><source>Genomic Data Commons</source><pub-id pub-id-type="accession" xlink:href="https://gdc.cancer.gov/about-data/publications/mc3-2017">mc3-2017</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><name><surname>Tate</surname><given-names>JG</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>COSMIC - gene expression data</data-title><source>COSMIC</source><pub-id pub-id-type="accession" xlink:href="https://cancer.sanger.ac.uk/cosmic/download">nar/gky1015</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset3"><person-group person-group-type="author"><collab>GTEX Consortium</collab></person-group><year iso-8601-date="2020">2020</year><data-title>GTEX - expression data</data-title><source>GTEX</source><pub-id pub-id-type="accession" xlink:href="https://www.gtexportal.org/home/datasets">phs000424.v8</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Judith Frydman for her contribution on the heat shock response analysis, Monte Winslow for his contribution on cancer subtype analysis, Donate Weghorn for her contribution on the interdependence of <italic>dN</italic>/<italic>dS</italic> and mutational burden, Leonid Mirny, Grant Kinsler, Gabor Boross, Chuan Li, Alison Feder, Eliot Cowan, and other members of the Petrov and Curtis labs for helpful comments and discussions. This work is supported by NIH grants T32-HG000044-21, E25-CA180993; the Director’s Pioneer Award DP1-CA238296 to CC; R01-CA207133, R35-GM118165, and R01-CA231253 to DAP; and K99-CA226506 to CDM.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>1000 Genomes Project Consortium</collab><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Garrison</surname><given-names>EP</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Korbel</surname><given-names>JO</given-names></name><name><surname>Marchini</surname><given-names>JL</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><name><surname>McVean</surname><given-names>GA</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A global reference for human genetic variation</article-title><source>Nature</source><volume>526</volume><fpage>68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature15393</pub-id><pub-id pub-id-type="pmid">26432245</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adzhubei</surname><given-names>IA</given-names></name><name><surname>Schmidt</surname><given-names>S</given-names></name><name><surname>Peshkin</surname><given-names>L</given-names></name><name><surname>Ramensky</surname><given-names>VE</given-names></name><name><surname>Gerasimova</surname><given-names>A</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Kondrashov</surname><given-names>AS</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A method and server for predicting damaging missense mutations</article-title><source>Nature Methods</source><volume>7</volume><fpage>248</fpage><lpage>249</lpage><pub-id pub-id-type="doi">10.1038/nmeth0410-248</pub-id><pub-id pub-id-type="pmid">20354512</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alexandrov</surname><given-names>LB</given-names></name><name><surname>Stratton</surname><given-names>MR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Mutational signatures: the patterns of somatic mutations hidden in cancer genomes</article-title><source>Current Opinion in Genetics &amp; Development</source><volume>24</volume><fpage>52</fpage><lpage>60</lpage><pub-id pub-id-type="doi">10.1016/j.gde.2013.11.014</pub-id><pub-id pub-id-type="pmid">24657537</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arjan</surname><given-names>JA</given-names></name><name><surname>Visser</surname><given-names>M</given-names></name><name><surname>Zeyl</surname><given-names>CW</given-names></name><name><surname>Gerrish</surname><given-names>PJ</given-names></name><name><surname>Blanchard</surname><given-names>JL</given-names></name><name><surname>Lenski</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Diminishing returns from mutation supply rate in asexual populations</article-title><source>Science</source><volume>283</volume><fpage>404</fpage><lpage>406</lpage><pub-id pub-id-type="doi">10.1126/science.283.5400.404</pub-id><pub-id pub-id-type="pmid">9888858</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bachtrog</surname><given-names>D</given-names></name><name><surname>Gordo</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Adaptive evolution of asexual populations under muller’s ratchet</article-title><source>Evolution; International Journal of Organic Evolution</source><volume>58</volume><fpage>1403</fpage><lpage>1413</lpage><pub-id pub-id-type="doi">10.1111/j.0014-3820.2004.tb01722.x</pub-id><pub-id pub-id-type="pmid">15341144</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bailey</surname><given-names>MH</given-names></name><name><surname>Tokheim</surname><given-names>C</given-names></name><name><surname>Porta-Pardo</surname><given-names>E</given-names></name><name><surname>Sengupta</surname><given-names>S</given-names></name><name><surname>Bertrand</surname><given-names>D</given-names></name><name><surname>Weerasinghe</surname><given-names>A</given-names></name><name><surname>Colaprico</surname><given-names>A</given-names></name><name><surname>Wendl</surname><given-names>MC</given-names></name><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Reardon</surname><given-names>B</given-names></name><name><surname>Ng</surname><given-names>PK-S</given-names></name><name><surname>Jeong</surname><given-names>KJ</given-names></name><name><surname>Cao</surname><given-names>S</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Gao</surname><given-names>J</given-names></name><name><surname>Gao</surname><given-names>Q</given-names></name><name><surname>Wang</surname><given-names>F</given-names></name><name><surname>Liu</surname><given-names>EM</given-names></name><name><surname>Mularoni</surname><given-names>L</given-names></name><name><surname>Rubio-Perez</surname><given-names>C</given-names></name><name><surname>Nagarajan</surname><given-names>N</given-names></name><name><surname>Cortés-Ciriano</surname><given-names>I</given-names></name><name><surname>Zhou</surname><given-names>DC</given-names></name><name><surname>Liang</surname><given-names>W-W</given-names></name><name><surname>Hess</surname><given-names>JM</given-names></name><name><surname>Yellapantula</surname><given-names>VD</given-names></name><name><surname>Tamborero</surname><given-names>D</given-names></name><name><surname>Gonzalez-Perez</surname><given-names>A</given-names></name><name><surname>Suphavilai</surname><given-names>C</given-names></name><name><surname>Ko</surname><given-names>JY</given-names></name><name><surname>Khurana</surname><given-names>E</given-names></name><name><surname>Park</surname><given-names>PJ</given-names></name><name><surname>Van Allen</surname><given-names>EM</given-names></name><name><surname>Liang</surname><given-names>H</given-names></name><name><surname>Lawrence</surname><given-names>MS</given-names></name><name><surname>Godzik</surname><given-names>A</given-names></name><name><surname>Lopez-Bigas</surname><given-names>N</given-names></name><name><surname>Stuart</surname><given-names>J</given-names></name><name><surname>Wheeler</surname><given-names>D</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name><name><surname>Chen</surname><given-names>K</given-names></name><name><surname>Lazar</surname><given-names>AJ</given-names></name><name><surname>Mills</surname><given-names>GB</given-names></name><name><surname>Karchin</surname><given-names>R</given-names></name><name><surname>Ding</surname><given-names>L</given-names></name><collab>MC3 Working Group</collab><collab>Cancer Genome Atlas Research Network</collab></person-group><year iso-8601-date="2018">2018</year><article-title>Comprehensive characterization of cancer driver genes and mutations</article-title><source>Cell</source><volume>173</volume><fpage>371</fpage><lpage>385</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.02.060</pub-id><pub-id pub-id-type="pmid">29625053</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Calderone</surname><given-names>A</given-names></name><name><surname>Cesareni</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Mentha: the interactome browser</article-title><source>EMBnet.Journal</source><volume>18</volume><elocation-id>128</elocation-id><pub-id pub-id-type="doi">10.14806/ej.18.A.455</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Campbell</surname><given-names>P</given-names></name><name><surname>Martincorena</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2017">2017</year><source>DNdScv</source><publisher-name>Welcome sanger institute</publisher-name></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Camps</surname><given-names>M</given-names></name><name><surname>Herman</surname><given-names>A</given-names></name><name><surname>Loh</surname><given-names>E</given-names></name><name><surname>Loeb</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Genetic constraints on protein evolution</article-title><source>Critical Reviews in Biochemistry and Molecular Biology</source><volume>42</volume><fpage>313</fpage><lpage>326</lpage><pub-id pub-id-type="doi">10.1080/10409230701597642</pub-id><pub-id pub-id-type="pmid">17917869</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Carithers</surname><given-names>LJ</given-names></name><name><surname>Moore</surname><given-names>HM</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The genotype-tissue expression (gtex) project</article-title><source>Biopreservation and Biobanking</source><volume>13</volume><fpage>307</fpage><lpage>308</lpage><pub-id pub-id-type="doi">10.1089/bio.2015.29031.hmm</pub-id><pub-id pub-id-type="pmid">26484569</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Carter</surname><given-names>SL</given-names></name><name><surname>Cibulskis</surname><given-names>K</given-names></name><name><surname>Helman</surname><given-names>E</given-names></name><name><surname>McKenna</surname><given-names>A</given-names></name><name><surname>Shen</surname><given-names>H</given-names></name><name><surname>Zack</surname><given-names>T</given-names></name><name><surname>Laird</surname><given-names>PW</given-names></name><name><surname>Onofrio</surname><given-names>RC</given-names></name><name><surname>Winckler</surname><given-names>W</given-names></name><name><surname>Weir</surname><given-names>BA</given-names></name><name><surname>Beroukhim</surname><given-names>R</given-names></name><name><surname>Pellman</surname><given-names>D</given-names></name><name><surname>Levine</surname><given-names>DA</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Meyerson</surname><given-names>M</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Absolute quantification of somatic DNA alterations in human cancer</article-title><source>Nature Biotechnology</source><volume>30</volume><fpage>413</fpage><lpage>421</lpage><pub-id pub-id-type="doi">10.1038/nbt.2203</pub-id><pub-id pub-id-type="pmid">22544022</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cassa</surname><given-names>CA</given-names></name><name><surname>Weghorn</surname><given-names>D</given-names></name><name><surname>Balick</surname><given-names>DJ</given-names></name><name><surname>Jordan</surname><given-names>DM</given-names></name><name><surname>Nusinow</surname><given-names>D</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>O’Donnell-Luria</surname><given-names>A</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Beier</surname><given-names>DR</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Estimating the selective effects of heterozygous protein-truncating variants from human exome data</article-title><source>Nature Genetics</source><volume>49</volume><fpage>806</fpage><lpage>810</lpage><pub-id pub-id-type="doi">10.1038/ng.3831</pub-id><pub-id pub-id-type="pmid">28369035</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Csilléry</surname><given-names>K</given-names></name><name><surname>François</surname><given-names>O</given-names></name><name><surname>Blum</surname><given-names>MGB</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>ABC: an R package for approximate bayesian computation (ABC)</article-title><source>Methods in Ecology and Evolution</source><volume>3</volume><fpage>475</fpage><lpage>479</lpage><pub-id pub-id-type="doi">10.1111/j.2041-210X.2011.00179.x</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname><given-names>C</given-names></name><name><surname>Whitesell</surname><given-names>L</given-names></name><name><surname>Rogers</surname><given-names>AB</given-names></name><name><surname>Lindquist</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Heat shock factor 1 is a powerful multifaceted modifier of carcinogenesis</article-title><source>Cell</source><volume>130</volume><fpage>1005</fpage><lpage>1018</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2007.07.020</pub-id><pub-id pub-id-type="pmid">17889646</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Drummond</surname><given-names>DA</given-names></name><name><surname>Wilke</surname><given-names>CO</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Mistranslation-induced protein misfolding as a dominant constraint on coding-sequence evolution</article-title><source>Cell</source><volume>134</volume><fpage>341</fpage><lpage>352</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2008.05.042</pub-id><pub-id pub-id-type="pmid">18662548</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Drummond</surname><given-names>DA</given-names></name><name><surname>Wilke</surname><given-names>CO</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The evolutionary consequences of erroneous protein synthesis</article-title><source>Nature Reviews. Genetics</source><volume>10</volume><fpage>715</fpage><lpage>724</lpage><pub-id pub-id-type="doi">10.1038/nrg2662</pub-id><pub-id pub-id-type="pmid">19763154</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eisenberg</surname><given-names>E</given-names></name><name><surname>Levanon</surname><given-names>EY</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Human housekeeping genes, revisited</article-title><source>Trends in Genetics</source><volume>29</volume><fpage>569</fpage><lpage>574</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2013.05.010</pub-id><pub-id pub-id-type="pmid">23810203</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ellrott</surname><given-names>K</given-names></name><name><surname>Bailey</surname><given-names>MH</given-names></name><name><surname>Saksena</surname><given-names>G</given-names></name><name><surname>Covington</surname><given-names>KR</given-names></name><name><surname>Kandoth</surname><given-names>C</given-names></name><name><surname>Stewart</surname><given-names>C</given-names></name><name><surname>Hess</surname><given-names>J</given-names></name><name><surname>Ma</surname><given-names>S</given-names></name><name><surname>Chiotti</surname><given-names>KE</given-names></name><name><surname>McLellan</surname><given-names>M</given-names></name><name><surname>Sofia</surname><given-names>HJ</given-names></name><name><surname>Hutter</surname><given-names>C</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name><name><surname>Wheeler</surname><given-names>D</given-names></name><name><surname>Ding</surname><given-names>L</given-names></name><collab>MC3 Working Group</collab><collab>Cancer Genome Atlas Research Network</collab></person-group><year iso-8601-date="2018">2018</year><article-title>Scalable open science approach for mutation calling of tumor exomes using multiple genomic pipelines</article-title><source>Cell Systems</source><volume>6</volume><fpage>271</fpage><lpage>281</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2018.03.002</pub-id><pub-id pub-id-type="pmid">29596782</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Forbes</surname><given-names>SA</given-names></name><name><surname>Bhamra</surname><given-names>G</given-names></name><name><surname>Bamford</surname><given-names>S</given-names></name><name><surname>Dawson</surname><given-names>E</given-names></name><name><surname>Kok</surname><given-names>C</given-names></name><name><surname>Clements</surname><given-names>J</given-names></name><name><surname>Menzies</surname><given-names>A</given-names></name><name><surname>Teague</surname><given-names>JW</given-names></name><name><surname>Futreal</surname><given-names>PA</given-names></name><name><surname>Stratton</surname><given-names>MR</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>The catalogue of somatic mutations in cancer (COSMIC)</article-title><source>Current Protocols in Human Genetics</source><volume>Chapter 10</volume><elocation-id>Unit</elocation-id><pub-id pub-id-type="doi">10.1002/0471142905.hg1011s57</pub-id><pub-id pub-id-type="pmid">18428421</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Frank</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="2007">2007</year><source>Dynamics of Cancer: Incidence, Inheritance, and Evolution</source><publisher-name>Princeton University</publisher-name></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Futreal</surname><given-names>PA</given-names></name><name><surname>Coin</surname><given-names>L</given-names></name><name><surname>Marshall</surname><given-names>M</given-names></name><name><surname>Down</surname><given-names>T</given-names></name><name><surname>Hubbard</surname><given-names>T</given-names></name><name><surname>Wooster</surname><given-names>R</given-names></name><name><surname>Rahman</surname><given-names>N</given-names></name><name><surname>Stratton</surname><given-names>MR</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>A census of human cancer genes</article-title><source>Nature Reviews. Cancer</source><volume>4</volume><fpage>177</fpage><lpage>183</lpage><pub-id pub-id-type="doi">10.1038/nrc1299</pub-id><pub-id pub-id-type="pmid">14993899</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gelman</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Parameterization and bayesian modeling</article-title><source>Journal of the American Statistical Association</source><volume>99</volume><fpage>537</fpage><lpage>545</lpage><pub-id pub-id-type="doi">10.1198/016214504000000458</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gibson</surname><given-names>MA</given-names></name><name><surname>Bruck</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Efficient exact stochastic simulation of chemical systems with many species and many channels</article-title><source>The Journal of Physical Chemistry A</source><volume>104</volume><fpage>1876</fpage><lpage>1889</lpage><pub-id pub-id-type="doi">10.1021/jp993732q</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Glaire</surname><given-names>MA</given-names></name><name><surname>Church</surname><given-names>DN</given-names></name></person-group><year iso-8601-date="2017">2017</year><chapter-title>Hypermutated colorectal cancer and neoantigen load</chapter-title><person-group person-group-type="editor"><name><surname>David</surname><given-names>K</given-names></name><name><surname>Rebecca</surname><given-names>J</given-names></name></person-group><source>Immunotherapy for Gastrointestinal Cancer</source><publisher-name>Springer</publisher-name><fpage>187</fpage><lpage>215</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-43063-8_8</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gonzalez-Perez</surname><given-names>A</given-names></name><name><surname>Perez-Llamas</surname><given-names>C</given-names></name><name><surname>Deu-Pons</surname><given-names>J</given-names></name><name><surname>Tamborero</surname><given-names>D</given-names></name><name><surname>Schroeder</surname><given-names>MP</given-names></name><name><surname>Jene-Sanz</surname><given-names>A</given-names></name><name><surname>Santos</surname><given-names>A</given-names></name><name><surname>Lopez-Bigas</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>IntOGen-mutations identifies cancer drivers across tumor types</article-title><source>Nature Methods</source><volume>10</volume><fpage>1081</fpage><lpage>1082</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2642</pub-id><pub-id pub-id-type="pmid">24037244</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gorgoulis</surname><given-names>VG</given-names></name><name><surname>Pefani</surname><given-names>DE</given-names></name><name><surname>Pateras</surname><given-names>IS</given-names></name><name><surname>Trougakos</surname><given-names>IP</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Integrating the DNA damage and protein stress responses during cancer development and treatment</article-title><source>The Journal of Pathology</source><volume>246</volume><fpage>12</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.1002/path.5097</pub-id><pub-id pub-id-type="pmid">29756349</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grossman</surname><given-names>RL</given-names></name><name><surname>Heath</surname><given-names>AP</given-names></name><name><surname>Ferretti</surname><given-names>V</given-names></name><name><surname>Varmus</surname><given-names>HE</given-names></name><name><surname>Lowy</surname><given-names>DR</given-names></name><name><surname>Kibbe</surname><given-names>WA</given-names></name><name><surname>Staudt</surname><given-names>LM</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Toward a shared vision for cancer genomic data</article-title><source>The New England Journal of Medicine</source><volume>375</volume><fpage>1109</fpage><lpage>1112</lpage><pub-id pub-id-type="doi">10.1056/NEJMp1607591</pub-id><pub-id pub-id-type="pmid">27653561</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2020">2020</year><article-title>The gtex Consortium atlas of genetic regulatory effects across human tissues</article-title><source>Science</source><volume>369</volume><fpage>1318</fpage><lpage>1330</lpage><pub-id pub-id-type="doi">10.1126/science.aaz1776</pub-id><pub-id pub-id-type="pmid">32913098</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ha</surname><given-names>G</given-names></name><name><surname>Roth</surname><given-names>A</given-names></name><name><surname>Khattra</surname><given-names>J</given-names></name><name><surname>Ho</surname><given-names>J</given-names></name><name><surname>Yap</surname><given-names>D</given-names></name><name><surname>Prentice</surname><given-names>LM</given-names></name><name><surname>Melnyk</surname><given-names>N</given-names></name><name><surname>McPherson</surname><given-names>A</given-names></name><name><surname>Bashashati</surname><given-names>A</given-names></name><name><surname>Laks</surname><given-names>E</given-names></name><name><surname>Biele</surname><given-names>J</given-names></name><name><surname>Ding</surname><given-names>J</given-names></name><name><surname>Le</surname><given-names>A</given-names></name><name><surname>Rosner</surname><given-names>J</given-names></name><name><surname>Shumansky</surname><given-names>K</given-names></name><name><surname>Marra</surname><given-names>MA</given-names></name><name><surname>Gilks</surname><given-names>CB</given-names></name><name><surname>Huntsman</surname><given-names>DG</given-names></name><name><surname>McAlpine</surname><given-names>JN</given-names></name><name><surname>Aparicio</surname><given-names>S</given-names></name><name><surname>Shah</surname><given-names>SP</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>TITAN: inference of copy number architectures in clonal cell populations from tumor whole-genome sequence data</article-title><source>Genome Research</source><volume>24</volume><fpage>1881</fpage><lpage>1893</lpage><pub-id pub-id-type="doi">10.1101/gr.180281.114</pub-id><pub-id pub-id-type="pmid">25060187</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hanahan</surname><given-names>D</given-names></name><name><surname>Weinberg</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The hallmarks of cancer</article-title><source>Cell</source><volume>100</volume><fpage>57</fpage><lpage>70</lpage><pub-id pub-id-type="doi">10.1016/s0092-8674(00)81683-9</pub-id><pub-id pub-id-type="pmid">10647931</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haradhvala</surname><given-names>NJ</given-names></name><name><surname>Polak</surname><given-names>P</given-names></name><name><surname>Stojanov</surname><given-names>P</given-names></name><name><surname>Covington</surname><given-names>KR</given-names></name><name><surname>Shinbrot</surname><given-names>E</given-names></name><name><surname>Hess</surname><given-names>JM</given-names></name><name><surname>Rheinbay</surname><given-names>E</given-names></name><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Maruvka</surname><given-names>YE</given-names></name><name><surname>Braunstein</surname><given-names>LZ</given-names></name><name><surname>Kamburov</surname><given-names>A</given-names></name><name><surname>Hanawalt</surname><given-names>PC</given-names></name><name><surname>Wheeler</surname><given-names>DA</given-names></name><name><surname>Koren</surname><given-names>A</given-names></name><name><surname>Lawrence</surname><given-names>MS</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mutational strand asymmetries in cancer genomes reveal mechanisms of DNA damage and repair</article-title><source>Cell</source><volume>164</volume><fpage>538</fpage><lpage>549</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2015.12.050</pub-id><pub-id pub-id-type="pmid">26806129</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname><given-names>MA</given-names></name><name><surname>Clark</surname><given-names>J</given-names></name><name><surname>Ireland</surname><given-names>A</given-names></name><name><surname>Lomax</surname><given-names>J</given-names></name><name><surname>Ashburner</surname><given-names>M</given-names></name><name><surname>Foulger</surname><given-names>R</given-names></name><name><surname>Eilbeck</surname><given-names>K</given-names></name><name><surname>Lewis</surname><given-names>S</given-names></name><name><surname>Marshall</surname><given-names>B</given-names></name><name><surname>Mungall</surname><given-names>C</given-names></name><name><surname>Richter</surname><given-names>J</given-names></name><name><surname>Rubin</surname><given-names>GM</given-names></name><name><surname>Blake</surname><given-names>JA</given-names></name><name><surname>Bult</surname><given-names>C</given-names></name><name><surname>Dolan</surname><given-names>M</given-names></name><name><surname>Drabkin</surname><given-names>H</given-names></name><name><surname>Eppig</surname><given-names>JT</given-names></name><name><surname>Hill</surname><given-names>DP</given-names></name><name><surname>Ni</surname><given-names>L</given-names></name><name><surname>Ringwald</surname><given-names>M</given-names></name><name><surname>Balakrishnan</surname><given-names>R</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Christie</surname><given-names>KR</given-names></name><name><surname>Costanzo</surname><given-names>MC</given-names></name><name><surname>Dwight</surname><given-names>SS</given-names></name><name><surname>Engel</surname><given-names>S</given-names></name><name><surname>Fisk</surname><given-names>DG</given-names></name><name><surname>Hirschman</surname><given-names>JE</given-names></name><name><surname>Hong</surname><given-names>EL</given-names></name><name><surname>Nash</surname><given-names>RS</given-names></name><name><surname>Sethuraman</surname><given-names>A</given-names></name><name><surname>Theesfeld</surname><given-names>CL</given-names></name><name><surname>Botstein</surname><given-names>D</given-names></name><name><surname>Dolinski</surname><given-names>K</given-names></name><name><surname>Feierbach</surname><given-names>B</given-names></name><name><surname>Berardini</surname><given-names>T</given-names></name><name><surname>Mundodi</surname><given-names>S</given-names></name><name><surname>Rhee</surname><given-names>SY</given-names></name><name><surname>Apweiler</surname><given-names>R</given-names></name><name><surname>Barrell</surname><given-names>D</given-names></name><name><surname>Camon</surname><given-names>E</given-names></name><name><surname>Dimmer</surname><given-names>E</given-names></name><name><surname>Lee</surname><given-names>V</given-names></name><name><surname>Chisholm</surname><given-names>R</given-names></name><name><surname>Gaudet</surname><given-names>P</given-names></name><name><surname>Kibbe</surname><given-names>W</given-names></name><name><surname>Kishore</surname><given-names>R</given-names></name><name><surname>Schwarz</surname><given-names>EM</given-names></name><name><surname>Sternberg</surname><given-names>P</given-names></name><name><surname>Gwinn</surname><given-names>M</given-names></name><name><surname>Hannick</surname><given-names>L</given-names></name><name><surname>Wortman</surname><given-names>J</given-names></name><name><surname>Berriman</surname><given-names>M</given-names></name><name><surname>Wood</surname><given-names>V</given-names></name><name><surname>de la Cruz</surname><given-names>N</given-names></name><name><surname>Tonellato</surname><given-names>P</given-names></name><name><surname>Jaiswal</surname><given-names>P</given-names></name><name><surname>Seigfried</surname><given-names>T</given-names></name><name><surname>White</surname><given-names>R</given-names></name><collab>Gene Ontology Consortium</collab></person-group><year iso-8601-date="2004">2004</year><article-title>The gene ontology (GO) database and informatics resource</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>D258</fpage><lpage>D261</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh036</pub-id><pub-id pub-id-type="pmid">14681407</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname><given-names>WG</given-names></name><name><surname>Robertson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1966">1966</year><article-title>The effect of linkage on limits to artificial selection</article-title><source>Genetical Research</source><volume>8</volume><fpage>269</fpage><lpage>294</lpage><pub-id pub-id-type="doi">10.1017/S001667230800949X</pub-id><pub-id pub-id-type="pmid">5980116</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Howlader</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2013">2013</year><source>SEER Cancer Stastistics Review 1975-2010</source><publisher-name>National Cancer Institute</publisher-name></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johnson</surname><given-names>T</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Beneficial mutations, hitchhiking and the evolution of mutation rates in sexual populations</article-title><source>Genetics</source><volume>151</volume><fpage>1621</fpage><lpage>1631</lpage><pub-id pub-id-type="doi">10.1093/genetics/151.4.1621</pub-id><pub-id pub-id-type="pmid">10101182</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kampinga</surname><given-names>HH</given-names></name><name><surname>Hageman</surname><given-names>J</given-names></name><name><surname>Vos</surname><given-names>MJ</given-names></name><name><surname>Kubota</surname><given-names>H</given-names></name><name><surname>Tanguay</surname><given-names>RM</given-names></name><name><surname>Bruford</surname><given-names>EA</given-names></name><name><surname>Cheetham</surname><given-names>ME</given-names></name><name><surname>Chen</surname><given-names>B</given-names></name><name><surname>Hightower</surname><given-names>LE</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Guidelines for the nomenclature of the human heat shock proteins</article-title><source>Cell Stress &amp; Chaperones</source><volume>14</volume><fpage>105</fpage><lpage>111</lpage><pub-id pub-id-type="doi">10.1007/s12192-008-0068-7</pub-id><pub-id pub-id-type="pmid">18663603</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Korbel</surname><given-names>JO</given-names></name><name><surname>Urban</surname><given-names>AE</given-names></name><name><surname>Grubert</surname><given-names>F</given-names></name><name><surname>Du</surname><given-names>J</given-names></name><name><surname>Royce</surname><given-names>TE</given-names></name><name><surname>Starr</surname><given-names>P</given-names></name><name><surname>Zhong</surname><given-names>G</given-names></name><name><surname>Emanuel</surname><given-names>BS</given-names></name><name><surname>Weissman</surname><given-names>SM</given-names></name><name><surname>Snyder</surname><given-names>M</given-names></name><name><surname>Gerstein</surname><given-names>MB</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Systematic prediction and validation of breakpoints associated with copy-number variants in the human genome</article-title><source>PNAS</source><volume>104</volume><fpage>10110</fpage><lpage>10115</lpage><pub-id pub-id-type="doi">10.1073/pnas.0703834104</pub-id><pub-id pub-id-type="pmid">17551006</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname><given-names>RD</given-names></name><name><surname>Searleman</surname><given-names>AC</given-names></name><name><surname>Swamidass</surname><given-names>SJ</given-names></name><name><surname>Griffith</surname><given-names>OL</given-names></name><name><surname>Bose</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Statistically identifying tumor suppressors and oncogenes from pan-cancer genome-sequencing data</article-title><source>Bioinformatics</source><volume>31</volume><fpage>3561</fpage><lpage>3568</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv430</pub-id><pub-id pub-id-type="pmid">26209800</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lobkovsky</surname><given-names>AE</given-names></name><name><surname>Wolf</surname><given-names>YI</given-names></name><name><surname>Koonin</surname><given-names>EV</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Universal distribution of protein evolution rates as a consequence of protein folding physics</article-title><source>PNAS</source><volume>107</volume><fpage>2983</fpage><lpage>2988</lpage><pub-id pub-id-type="doi">10.1073/pnas.0910445107</pub-id><pub-id pub-id-type="pmid">20133769</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>López</surname><given-names>S</given-names></name><name><surname>Lim</surname><given-names>EL</given-names></name><name><surname>Horswell</surname><given-names>S</given-names></name><name><surname>Haase</surname><given-names>K</given-names></name><name><surname>Huebner</surname><given-names>A</given-names></name><name><surname>Dietzen</surname><given-names>M</given-names></name><name><surname>Mourikis</surname><given-names>TP</given-names></name><name><surname>Watkins</surname><given-names>TBK</given-names></name><name><surname>Rowan</surname><given-names>A</given-names></name><name><surname>Dewhurst</surname><given-names>SM</given-names></name><name><surname>Birkbak</surname><given-names>NJ</given-names></name><name><surname>Wilson</surname><given-names>GA</given-names></name><name><surname>Van Loo</surname><given-names>P</given-names></name><name><surname>Jamal-Hanjani</surname><given-names>M</given-names></name><collab>TRACERx Consortium</collab><name><surname>Swanton</surname><given-names>C</given-names></name><name><surname>McGranahan</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Interplay between whole-genome doubling and the accumulation of deleterious alterations in cancer evolution</article-title><source>Nature Genetics</source><volume>52</volume><fpage>283</fpage><lpage>293</lpage><pub-id pub-id-type="doi">10.1038/s41588-020-0584-7</pub-id><pub-id pub-id-type="pmid">32139907</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martincorena</surname><given-names>I</given-names></name><name><surname>Raine</surname><given-names>KM</given-names></name><name><surname>Gerstung</surname><given-names>M</given-names></name><name><surname>Dawson</surname><given-names>KJ</given-names></name><name><surname>Haase</surname><given-names>K</given-names></name><name><surname>Van Loo</surname><given-names>P</given-names></name><name><surname>Davies</surname><given-names>H</given-names></name><name><surname>Stratton</surname><given-names>MR</given-names></name><name><surname>Campbell</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Universal patterns of selection in cancer and somatic tissues</article-title><source>Cell</source><volume>171</volume><fpage>1029</fpage><lpage>1041</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2017.09.042</pub-id><pub-id pub-id-type="pmid">29056346</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McFarland</surname><given-names>CD</given-names></name><name><surname>Korolev</surname><given-names>KS</given-names></name><name><surname>Kryukov</surname><given-names>GV</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name><name><surname>Mirny</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Impact of deleterious passenger mutations on cancer progression</article-title><source>PNAS</source><volume>110</volume><fpage>2910</fpage><lpage>2915</lpage><pub-id pub-id-type="doi">10.1073/pnas.1213968110</pub-id><pub-id pub-id-type="pmid">23388632</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McFarland</surname><given-names>CD</given-names></name><name><surname>Mirny</surname><given-names>LA</given-names></name><name><surname>Korolev</surname><given-names>KS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Tug-of-war between driver and passenger mutations in cancer and other adaptive processes</article-title><source>PNAS</source><volume>111</volume><fpage>15138</fpage><lpage>15143</lpage><pub-id pub-id-type="doi">10.1073/pnas.1404341111</pub-id><pub-id pub-id-type="pmid">25277973</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McGrail</surname><given-names>DJ</given-names></name><name><surname>Garnett</surname><given-names>J</given-names></name><name><surname>Yin</surname><given-names>J</given-names></name><name><surname>Dai</surname><given-names>H</given-names></name><name><surname>Shih</surname><given-names>DJH</given-names></name><name><surname>Lam</surname><given-names>TNA</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Sun</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Schmandt</surname><given-names>R</given-names></name><name><surname>Wu</surname><given-names>JY</given-names></name><name><surname>Hu</surname><given-names>L</given-names></name><name><surname>Liang</surname><given-names>Y</given-names></name><name><surname>Peng</surname><given-names>G</given-names></name><name><surname>Jonasch</surname><given-names>E</given-names></name><name><surname>Menter</surname><given-names>D</given-names></name><name><surname>Yates</surname><given-names>MS</given-names></name><name><surname>Kopetz</surname><given-names>S</given-names></name><name><surname>Lu</surname><given-names>KH</given-names></name><name><surname>Broaddus</surname><given-names>R</given-names></name><name><surname>Mills</surname><given-names>GB</given-names></name><name><surname>Sahni</surname><given-names>N</given-names></name><name><surname>Lin</surname><given-names>S-Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Proteome instability is a therapeutic vulnerability in mismatch repair-deficient cancer</article-title><source>Cancer Cell</source><volume>37</volume><fpage>371</fpage><lpage>386</lpage><pub-id pub-id-type="doi">10.1016/j.ccell.2020.01.011</pub-id><pub-id pub-id-type="pmid">32109374</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mermel</surname><given-names>CH</given-names></name><name><surname>Schumacher</surname><given-names>SE</given-names></name><name><surname>Hill</surname><given-names>B</given-names></name><name><surname>Meyerson</surname><given-names>ML</given-names></name><name><surname>Beroukhim</surname><given-names>R</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>GISTIC2.0 facilitates sensitive and confident localization of the targets of focal somatic copy-number alteration in human cancers</article-title><source>Genome Biology</source><volume>12</volume><elocation-id>R41</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2011-12-4-r41</pub-id><pub-id pub-id-type="pmid">21527027</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Messer</surname><given-names>PW</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Measuring the rates of spontaneous mutation from deep and large-scale polymorphism data</article-title><source>Genetics</source><volume>182</volume><fpage>1219</fpage><lpage>1232</lpage><pub-id pub-id-type="doi">10.1534/genetics.109.105692</pub-id><pub-id pub-id-type="pmid">19528323</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Michor</surname><given-names>F</given-names></name><name><surname>Iwasa</surname><given-names>Y</given-names></name><name><surname>Lengauer</surname><given-names>C</given-names></name><name><surname>Nowak</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Dynamics of colorectal cancer</article-title><source>Seminars in Cancer Biology</source><volume>15</volume><fpage>484</fpage><lpage>493</lpage><pub-id pub-id-type="doi">10.1016/j.semcancer.2005.06.005</pub-id><pub-id pub-id-type="pmid">16055342</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="web"><person-group person-group-type="author"><collab>National Cancer Institute</collab></person-group><year iso-8601-date="2007">2007</year><article-title>Cancer incidence – surveillance, epidemiology, and end results (SEER) registries research data</article-title><ext-link ext-link-type="uri" xlink:href="http://www.seer.cancer.gov">http://www.seer.cancer.gov</ext-link><date-in-citation iso-8601-date="2019-07-30">July 30, 2019</date-in-citation></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Neher</surname><given-names>RA</given-names></name><name><surname>Shraiman</surname><given-names>BI</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Fluctuations of fitness distributions and the rate of muller’s ratchet</article-title><source>Genetics</source><volume>191</volume><fpage>1283</fpage><lpage>1293</lpage><pub-id pub-id-type="doi">10.1534/genetics.112.141325</pub-id><pub-id pub-id-type="pmid">22649084</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ostrow</surname><given-names>SL</given-names></name><name><surname>Barshir</surname><given-names>R</given-names></name><name><surname>DeGregori</surname><given-names>J</given-names></name><name><surname>Yeger-Lotem</surname><given-names>E</given-names></name><name><surname>Hershberg</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Cancer evolution is associated with pervasive positive selection on globally expressed genes</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>1004239</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004239</pub-id><pub-id pub-id-type="pmid">24603726</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Potapova</surname><given-names>T</given-names></name><name><surname>Gorbsky</surname><given-names>GJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The consequences of chromosome segregation errors in mitosis and meiosis</article-title><source>Biology</source><volume>6</volume><elocation-id>E12</elocation-id><pub-id pub-id-type="doi">10.3390/biology6010012</pub-id><pub-id pub-id-type="pmid">28208750</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Santagata</surname><given-names>S</given-names></name><name><surname>Hu</surname><given-names>R</given-names></name><name><surname>Lin</surname><given-names>NU</given-names></name><name><surname>Mendillo</surname><given-names>ML</given-names></name><name><surname>Collins</surname><given-names>LC</given-names></name><name><surname>Hankinson</surname><given-names>SE</given-names></name><name><surname>Schnitt</surname><given-names>SJ</given-names></name><name><surname>Whitesell</surname><given-names>L</given-names></name><name><surname>Tamimi</surname><given-names>RM</given-names></name><name><surname>Lindquist</surname><given-names>S</given-names></name><name><surname>Ince</surname><given-names>TA</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>High levels of nuclear heat-shock factor 1 (HSF1) are associated with poor prognosis in breast cancer</article-title><source>PNAS</source><volume>108</volume><fpage>18378</fpage><lpage>18383</lpage><pub-id pub-id-type="doi">10.1073/pnas.1115031108</pub-id><pub-id pub-id-type="pmid">22042860</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sottoriva</surname><given-names>A</given-names></name><name><surname>Kang</surname><given-names>H</given-names></name><name><surname>Ma</surname><given-names>Z</given-names></name><name><surname>Graham</surname><given-names>TA</given-names></name><name><surname>Salomon</surname><given-names>MP</given-names></name><name><surname>Zhao</surname><given-names>J</given-names></name><name><surname>Marjoram</surname><given-names>P</given-names></name><name><surname>Siegmund</surname><given-names>K</given-names></name><name><surname>Press</surname><given-names>MF</given-names></name><name><surname>Shibata</surname><given-names>D</given-names></name><name><surname>Curtis</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A big bang model of human colorectal tumor growth</article-title><source>Nature Genetics</source><volume>47</volume><fpage>209</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1038/ng.3214</pub-id><pub-id pub-id-type="pmid">25665006</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Supek</surname><given-names>F</given-names></name><name><surname>Miñana</surname><given-names>B</given-names></name><name><surname>Valcárcel</surname><given-names>J</given-names></name><name><surname>Gabaldón</surname><given-names>T</given-names></name><name><surname>Lehner</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Synonymous mutations frequently act as driver mutations in human cancers</article-title><source>Cell</source><volume>156</volume><fpage>1324</fpage><lpage>1335</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.01.051</pub-id><pub-id pub-id-type="pmid">24630730</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tanaka</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The proteasome: overview of structure and functions</article-title><source>Proceedings of the Japan Academy. Series B, Physical and Biological Sciences</source><volume>85</volume><fpage>12</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.2183/pjab.85.12</pub-id><pub-id pub-id-type="pmid">19145068</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tate</surname><given-names>JG</given-names></name><name><surname>Bamford</surname><given-names>S</given-names></name><name><surname>Jubb</surname><given-names>HC</given-names></name><name><surname>Sondka</surname><given-names>Z</given-names></name><name><surname>Beare</surname><given-names>DM</given-names></name><name><surname>Bindal</surname><given-names>N</given-names></name><name><surname>Boutselakis</surname><given-names>H</given-names></name><name><surname>Cole</surname><given-names>CG</given-names></name><name><surname>Creatore</surname><given-names>C</given-names></name><name><surname>Dawson</surname><given-names>E</given-names></name><name><surname>Fish</surname><given-names>P</given-names></name><name><surname>Harsha</surname><given-names>B</given-names></name><name><surname>Hathaway</surname><given-names>C</given-names></name><name><surname>Jupe</surname><given-names>SC</given-names></name><name><surname>Kok</surname><given-names>CY</given-names></name><name><surname>Noble</surname><given-names>K</given-names></name><name><surname>Ponting</surname><given-names>L</given-names></name><name><surname>Ramshaw</surname><given-names>CC</given-names></name><name><surname>Rye</surname><given-names>CE</given-names></name><name><surname>Speedy</surname><given-names>HE</given-names></name><name><surname>Stefancsik</surname><given-names>R</given-names></name><name><surname>Thompson</surname><given-names>SL</given-names></name><name><surname>Wang</surname><given-names>S</given-names></name><name><surname>Ward</surname><given-names>S</given-names></name><name><surname>Campbell</surname><given-names>PJ</given-names></name><name><surname>Forbes</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Cosmic: the Catalogue of somatic mutations in cancer</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D941</fpage><lpage>D947</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1015</pub-id><pub-id pub-id-type="pmid">30371878</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Tilk</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022a</year><data-title>Hill-Robertson interference (HRI) in cancer paper</data-title><version designator="swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75">swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75</version><source>SoftwarevHeritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:e4a6f9d73932cdce98d9de0cc1394b45d41252ec;origin=https://github.com/petrov-lab/cancer-HRI;visit=swh:1:snp:2996ca9f1c1709095fb538f591d3ede58d5eac3a;anchor=swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75">https://archive.softwareheritage.org/swh:1:dir:e4a6f9d73932cdce98d9de0cc1394b45d41252ec;origin=https://github.com/petrov-lab/cancer-HRI;visit=swh:1:snp:2996ca9f1c1709095fb538f591d3ede58d5eac3a;anchor=swh:1:rev:5d67f4a946e2d80efdc71c2ef689266678d8ff75</ext-link></element-citation></ref><ref id="bib58"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Tilk</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022b</year><data-title>PdSim</data-title><version designator="swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4">swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4</version><source>SoftwarevHeritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:2837cecedb2ca8992d5afa9ee54102a1e8fa61b6;origin=https://github.com/mirnylab/pdSim;visit=swh:1:snp:9023cfe7eac951feac42e3bd4133ce866a0fe0f0;anchor=swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4">https://archive.softwareheritage.org/swh:1:dir:2837cecedb2ca8992d5afa9ee54102a1e8fa61b6;origin=https://github.com/mirnylab/pdSim;visit=swh:1:snp:9023cfe7eac951feac42e3bd4133ce866a0fe0f0;anchor=swh:1:rev:f08bd75aabf7213e253baf26d219c374a745c8d4</ext-link></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Turajlic</surname><given-names>S</given-names></name><name><surname>Furney</surname><given-names>SJ</given-names></name><name><surname>Lambros</surname><given-names>MB</given-names></name><name><surname>Mitsopoulos</surname><given-names>C</given-names></name><name><surname>Kozarewa</surname><given-names>I</given-names></name><name><surname>Geyer</surname><given-names>FC</given-names></name><name><surname>Mackay</surname><given-names>A</given-names></name><name><surname>Hakas</surname><given-names>J</given-names></name><name><surname>Zvelebil</surname><given-names>M</given-names></name><name><surname>Lord</surname><given-names>CJ</given-names></name><name><surname>Ashworth</surname><given-names>A</given-names></name><name><surname>Thomas</surname><given-names>M</given-names></name><name><surname>Stamp</surname><given-names>G</given-names></name><name><surname>Larkin</surname><given-names>J</given-names></name><name><surname>Reis-Filho</surname><given-names>JS</given-names></name><name><surname>Marais</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Whole genome sequencing of matched primary and metastatic acral melanomas</article-title><source>Genome Research</source><volume>22</volume><fpage>196</fpage><lpage>207</lpage><pub-id pub-id-type="doi">10.1101/gr.125591.111</pub-id><pub-id pub-id-type="pmid">22183965</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>M</given-names></name><name><surname>Hakonarson</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>ANNOVAR: functional annotation of genetic variants from high-throughput sequencing data</article-title><source>Nucleic Acids Research</source><volume>38</volume><elocation-id>e164</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkq603</pub-id><pub-id pub-id-type="pmid">20601685</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>T</given-names></name><name><surname>Birsoy</surname><given-names>K</given-names></name><name><surname>Hughes</surname><given-names>NW</given-names></name><name><surname>Krupczak</surname><given-names>KM</given-names></name><name><surname>Post</surname><given-names>Y</given-names></name><name><surname>Wei</surname><given-names>JJ</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Sabatini</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Identification and characterization of essential genes in the human genome</article-title><source>Science</source><volume>350</volume><fpage>1096</fpage><lpage>1101</lpage><pub-id pub-id-type="doi">10.1126/science.aac7041</pub-id><pub-id pub-id-type="pmid">26472758</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weghorn</surname><given-names>D</given-names></name><name><surname>Sunyaev</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Bayesian inference of negative and positive selection in human cancers</article-title><source>Nature Genetics</source><volume>49</volume><fpage>1785</fpage><lpage>1788</lpage><pub-id pub-id-type="doi">10.1038/ng.3987</pub-id><pub-id pub-id-type="pmid">29106416</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname><given-names>BR</given-names></name><name><surname>Prabhu</surname><given-names>VR</given-names></name><name><surname>Hunter</surname><given-names>KE</given-names></name><name><surname>Glazier</surname><given-names>CM</given-names></name><name><surname>Whittaker</surname><given-names>CA</given-names></name><name><surname>Housman</surname><given-names>DE</given-names></name><name><surname>Amon</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Aneuploidy affects proliferation and spontaneous immortalization in mammalian cells</article-title><source>Science</source><volume>322</volume><fpage>703</fpage><lpage>709</lpage><pub-id pub-id-type="doi">10.1126/science.1160058</pub-id><pub-id pub-id-type="pmid">18974345</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zack</surname><given-names>TI</given-names></name><name><surname>Schumacher</surname><given-names>SE</given-names></name><name><surname>Carter</surname><given-names>SL</given-names></name><name><surname>Cherniack</surname><given-names>AD</given-names></name><name><surname>Saksena</surname><given-names>G</given-names></name><name><surname>Tabak</surname><given-names>B</given-names></name><name><surname>Lawrence</surname><given-names>MS</given-names></name><name><surname>Zhsng</surname><given-names>C-Z</given-names></name><name><surname>Wala</surname><given-names>J</given-names></name><name><surname>Mermel</surname><given-names>CH</given-names></name><name><surname>Sougnez</surname><given-names>C</given-names></name><name><surname>Gabriel</surname><given-names>SB</given-names></name><name><surname>Hernandez</surname><given-names>B</given-names></name><name><surname>Shen</surname><given-names>H</given-names></name><name><surname>Laird</surname><given-names>PW</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name><name><surname>Meyerson</surname><given-names>M</given-names></name><name><surname>Beroukhim</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Pan-cancer patterns of somatic copy number alteration</article-title><source>Nature Genetics</source><volume>45</volume><fpage>1134</fpage><lpage>1140</lpage><pub-id pub-id-type="doi">10.1038/ng.2760</pub-id><pub-id pub-id-type="pmid">24071852</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zapata</surname><given-names>L</given-names></name><name><surname>Pich</surname><given-names>O</given-names></name><name><surname>Serrano</surname><given-names>L</given-names></name><name><surname>Kondrashov</surname><given-names>FA</given-names></name><name><surname>Ossowski</surname><given-names>S</given-names></name><name><surname>Schaefer</surname><given-names>MH</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Negative selection in tumor genome evolution acts on essential cellular functions and the immunopeptidome</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>67</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1434-0</pub-id><pub-id pub-id-type="pmid">29855388</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>WH</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Mammalian housekeeping genes evolve more slowly than tissue-specific genes</article-title><source>Molecular Biology and Evolution</source><volume>21</volume><fpage>236</fpage><lpage>239</lpage><pub-id pub-id-type="doi">10.1093/molbev/msh010</pub-id><pub-id pub-id-type="pmid">14595094</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.67790.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Taylor</surname><given-names>Martin</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01nrxwf90</institution-id><institution>University of Edinburgh</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/764340" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/764340"/></front-stub><body><p>This is an important paper that shows most cancers unavoidably accumulate damaging mutations. Whilst the majority of claims are convincingly supported by the data, evidence that damaging changes are buffered by heat shock pathways is currently incomplete. The insights into selection efficiency are important for the understanding of cancer growth and response to therapy. A broader implication is that high mutation load tumors may use common strategies to tolerate accumulated deleterious mutations, providing a therapeutic target.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.67790.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Taylor</surname><given-names>Martin</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01nrxwf90</institution-id><institution>University of Edinburgh</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Kuzmin</surname><given-names>Elena</given-names></name><role>Reviewer</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0420zvk78</institution-id><institution>Concordia University</institution></institution-wrap><country>Canada</country></aff></contrib><contrib contrib-type="reviewer"><name><surname>Taylor</surname><given-names>Martin</given-names></name><role>Reviewer</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01nrxwf90</institution-id><institution>University of Edinburgh</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/764340">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/764340v2">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Most cancers carry a substantial deleterious load due to Hill-Robertson interference&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, including Martin Taylor as the Reviewing Editor and Reviewer #3, and the evaluation has been overseen by Molly Przeworski as the Senior Editor. The following individual involved in review of your submission has agreed to reveal their identity: Elena Kuzmin (Reviewer #1).</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>(1) The authors state that they are excluding tumors with either nonsynonymous(n)=0 or synonymous(s)=0 mutations. Since nonsynonymous and synonymous variants occur in a ratio of about 3:1, this exclusion of tumors would seem to lead to an inflation of the signal of selection in the first (lowest mutation) bins. Additional demonstration is required to show if this distorts the estimate of selection for low mutation burden tumours. For example, by adding pseudo-counts of mutations or by aggregation over tumours in the same mutation load &quot;bin&quot; as was performed for some analyses.</p><p>(2) The decline of dN/dS on driver genes is the subject of the supplementary text and we note the efforts taken to disentangle Hill-Robertson interference effects from other possible explanations for dN/dS decay in drivers with increasing tumor mutation burden (TMB). Driver genes can be quite tissue-specific (and thus mus-identified for a tumor) and number of drivers per tumour estimated to span approximately 1 to 10 (Martincorena et al., 2017). Consequently, the fitting of a single set of model parameters and showing they do not match well the observed data (Supplementary note figure) is insufficient to exclude the misidentification of driver genes, or presence of nonsynonymous-neutral mutations in annotated driver genes, as an explanation for the decline in dN/dS with increased TMB. We think it important that a range of justifiable parameters are applied in this modelling, to test if the observed data is robustly outside reasonable parametrisation of the model.</p><p>(3) In many figures that show dN/dS as a function of n+s (starting with Figure 2A and extending to Figures S2, 3, 9, 10, 12, 22 and 25), there are no error bars indicated, as opposed to the statement in the figure caption. The error bars/shading should be shown. In Figure 2A, is the observed depletion in the second bin still significant?</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>1. Figure panels should be called out sequentially. For example, Figure 2G is called out before Figure 2D. This happens throughout the text, including main and supplementary figures, and should be corrected.</p><p>2. Figure 2G shows that mean gene expression of genes encoding chaperones and the proteasome increases with increasing mutational burden. What about protein abundance? Is this in agreement with gene expression?</p><p>3. Figure 2 mentions error bars in the figure legend, but no panel displays error bars. This is also true for Figure S13 and other figures. Authors should display the error bars to which they are referring to make their analysis more convincing.</p><p>4. Pg. 9 line 295 describes results of the analysis across genes belonging to different GO terms. However, Figure S13 only shows 3 categories: chromosome segregation, transcription and translation. How were these categories chosen? What about other categories? Such cherry picking doesn't convincingly support the conclusions that no specific GO functions are enriched. Also, translational regulation shows higher dN/dS in low mutation tumors suggesting that there is positive selection for passengers in this category. Authors should discuss in their manuscript why this is the case.</p><p>5. Figure S15 shows the attenuation in selection of CNAs across cancer subtypes and broad cancer groups. However, HNSC and kidney cancer appear to be the exceptions. Authors should provide an explanation for these observations in the main text.</p><p>6. Generally, copy number variations are considered to be &gt; 50 bp. Is there a rationale as to why authors chose 100 kb to be their cut-off in Figure 2C? If the size of CNA is an important parameter, then authors should explain why that is.</p><p>7. Non-allelic recombination and non-homologous recombination mechanisms involving replication accidents that lead to chromosome breakage occur with some frequency in somatic cells. How does the frequency of these events impact the selection efficiency in cancer as it relates to drivers and passengers? Can this also be incorporated in their evolutionary model?</p><p>8. Authors mentioned that haploinsufficiency was not used in the model. What about loss of heterozygosity which is extensive in cancer genomes? Can this parameter be included in the evolutionary model and how would it impact the results?</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>1. The authors have taken great care to study single-nucleotide variants and large CNAs. It would be great if they could confirm their findings by also showing the effect on small insertions and deletions.</p><p>2. Figure S5 is showing a bias in the determination of dN/dS from simulation results and the correlation between mutation rate and n+s. I am not sure I understand why dN/dS under a neutral simulation would be biased. Also, the low median correlation between n+s and the mutation rate (&lt;0.4) is quite surprising. I would have expected these to be almost perfectly correlated. Likewise, I do not understand the formula after l. 631. It states that this is the joint density of the two Poisson random variables that denote nonsynonymous and synonymous mutation count, yet there is an additional unexplained factor in the denominator, which corresponds to the probability of s&gt;0. If the simulation that underlies Figure S5 was also used in the ABC-based parameter inference, this would raise a serious cause for concern.</p><p>3. The simulation starts when the first mutation with positive selective effect initiates population growth, which can be very late in a patient's life. How does the assumption that up to 100 years can pass after this affect the parameter estimates?</p><p>4. To which extent does the inferred distribution of selection effects depend on the allowable parameter range? For example, s_passengers extends beyond the initially allowable range after the fit (Figure 3C).</p><p>5. It is not entirely clear to me how the partitioning of the likelihood between Muller's ratchet and hitchhiking vs other effects can be made and how robust these inferences are with respect to variation of the modeling assumptions (e.g. about initial population size or mode of selection). Is the necessity of inclusion of selected synonymous variants on driver genes a robust result or not, taking into account the discussion on p. 26f.?</p><p>6. In Figure S4, the authors report the correlation of n+s with other measures of tumor mutation load. Given the relative sizes of the different regions that are displayed, i.e. whole genome:intergenic:intronic:exonic:protein-coding, of roughly 100:60:40:2:1, the displayed numbers do not make sense, as their ratios are 100:100:100:1:0.001.</p><p>7. I am not sure I understood well how CNAs were analyzed. Based on the description in l. 669ff., it appears that putative cancer driver genes were identified from the CNA data based on recurrence. Were the same data then analyzed for CNAs falling into said putative cancer driver gene regions to infer selection? This would appear a bit circular.</p><p>8. I do not understand the formula shown after li. 738. It appears it is showing the fraction of genes that intersect a CNA boundary, summed over all tumors in a given n+s bin. Each CNA can be counted twice if both of its boundaries fall into a gene. Why is the mean value of this 1?</p><p>9. In all figures that show dN/dS as a function of n+s (starting with Figure 2A and extending to Figures S2, 3, 9, 10, 12, 22 and 25), there are no error bars indicated, as opposed to the statement in the figure caption. In Figure 2A, is the observed depletion in the second bin still significant?</p><p>10. In l. 290, I understand that the authors argue that differential dominance effects between heterozygous early- and late-arising mutations could be affecting the efficacy of selection on subclonal variants compared to clonal variants. I do not see this claim well motivated or corroborated.</p><p>11. In Figure 2D, the caption states that mutations have been separated into two groups by their clonality, yet the figure shows three curves. What do they correspond to? Are the results still significant given the partitioning of the mutation data into smaller subsets?</p><p>12. Figure 3 does not have a panel G.</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>I enjoyed reading the manuscript, it was well written, generally clear figures and very through provoking.</p><p>Only a small number of specific points to address:</p><p>Line 60.- The description of dN and dS here along with the interpretation of dN/dS=1 as neutral implies that you are just counting non-synonymous and synonymous mutations and dividing one by the other. This of course is not the case. Perhaps dN and dS could be described as rates or dN/dS as the dN:dS odds ratio which is how it's calculated for your permutation metric.</p><p>Line 112 – 40% of what, benefit of ~130% of what. This becomes apparent later into the manuscript but not clear how to interpret when reading at this point for the first time.</p><p>Line 188 – Mutational burden &lt;= 3 (what units).</p><p>Line 189 – &quot;We observed little negative selection in passengers&quot; be clear what passengers (previous identified passenger genes).</p><p>Line 252 – Panel G, y-axis, what units? Why are &quot;all&quot; genes uniformly at approximately -0.2? Assuming this is fold-change or log-fold-change I'd expect 1 or zero respectively.</p><p>Line 275 – Figure 2G, shaded error bars are not visible.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.67790.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>(1) The authors state that they are excluding tumors with either nonsynonymous(n)=0 or synonymous(s)=0 mutations. Since nonsynonymous and synonymous variants occur in a ratio of about 3:1, this exclusion of tumors would seem to lead to an inflation of the signal of selection in the first (lowest mutation) bins. Additional demonstration is required to show if this distorts the estimate of selection for low mutation burden tumours. For example, by adding pseudo-counts of mutations or by aggregation over tumours in the same mutation load &quot;bin&quot; as was performed for some analyses.</p></disp-quote><p>We thank the reviewers for bringing up this important point. Our original reasoning for applying this filtering procedure was the concern that false positive random mutations, which would push dN/dS to 1, would dominate the signal in tumors with very few true mutations. Nonetheless, we agree with the reviewers that this filtering procedure can introduce potential biases into dN/dS estimates and thus, we have now re-analyzed all the main and supplemental figures in the re-submission without this filtering step.</p><p>Overall, we find that although negative selection on passengers is still present in low TMB tumors after removing this filtering procedure, signals of negative selection are diminished (from dN/dS ~0.4 to ~0.7 when combining all tumors in TCGA and ICGC and using both null models of dN/dS).</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-67790-sa2-fig1-v2.tif"/></fig><p>Since we are no longer applying any quality control filters, we need to be particularly stringent about the mutational calls we use to ensure we don’t lose all signal to potential false positives. Indeed, we empirically observe that the power to detect selection is strongly dependent on the quality of mutation calls in the sample. We find that even within the same dataset (TCGA), signals of negative selection on passengers in low TMB tumors disappears (dN/dS ~ 1) when just one mutation caller is used (‘Mutect 2 SNP Calls’) but can still be detected (dN/dS ~ 0.5) when we use a consensus set of mutation calls (‘MC3 SNP Calls’; which uses 7 different mutation callers). This is true when using either null model of dN/dS (dNdScv and dNdS-permutation). Similarly, signals of positive selection on drivers in low TMB tumors also diminishes (from dN/dS ~5 to ~4) when low quality mutation calls are used. The figure is now included in the re-submission as Figure 2—figure supplement 15Since coverage in protein-coding regions in whole-genome data is much lower than in the whole-exome data and combining datasets can potentially introduce further bias, we elected to only use TCGA whole-exome data – which contains the most stringent consensus mutation calls (MC3 SNP Calls) – in the re-submission to focus on patterns of selection in the highest quality set of mutations available. We also note that others have similarly raised concerns about difficulties combining whole-genome sequencing and whole-exome tumors even using the same cancer samples, which produce only 75% of concordant mutations when comparing mutation calls in protein-coding regions (Bailey et al., 2020, Nature Communications). As expected, we find stronger signals of selection in drivers (from dN/dS ~5.5 to ~4) and passengers (from dN/dS in ~0.7 to ~0.6) in low TMB tumors (&lt; 3 protein-coding mutations) in TCGA (‘MC3 SNP Calls’) when compared to the combining both TCGA and ICGC using both null models of mutagenesis.</p><disp-quote content-type="editor-comment"><p>(2) The decline of dN/dS on driver genes is the subject of the supplementary text and we note the efforts taken to disentangle Hill-Robertson interference effects from other possible explanations for dN/dS decay in drivers with increasing tumor mutation burden (TMB). Driver genes can be quite tissue-specific (and thus mus-identified for a tumor) and number of drivers per tumour estimated to span approximately 1 to 10 (Martincorena et al., 2017). Consequently, the fitting of a single set of model parameters and showing they do not match well the observed data (Supplementary note figure) is insufficient to exclude the misidentification of driver genes, or presence of nonsynonymous-neutral mutations in annotated driver genes, as an explanation for the decline in dN/dS with increased TMB. We think it important that a range of justifiable parameters are applied in this modelling, to test if the observed data is robustly outside reasonable parametrisation of the model.</p></disp-quote><p>Our goal for the modeling section was to investigate whether a simple evolutionary model of linkage with damaging passengers and advantageous drivers could reproduce similar patterns of selection observed in the empirical data. We agree with the reviewers that attenuation of dN/dS in drivers can be explained by processes other than Hill-Robertson interference. As the reviewers pointed out, we focus on alternative explanations for attenuation in the drivers within the supplemental portion of the text. In the previous submission, we explored a similar scenario that the reviewers suggest (i.e. 50% of neutral mutations accumulating in driver genes). However, we agree with the reviewers that this does not conclusively allow us to reject or make claims about the contribution of alternative models in drivers. To correct this, we have now included these potential alternative explanations suggested by the reviewers into the main text (Page. 11, Lines 16-26) and removed the supplemental note.</p><p>We would like to emphasize, however, that the focus of our paper is on negative selection on passengers. In all of the alternative models of dN/dS in drivers, passenger mutations are required to impose a fitness cost to be able to observe attenuation of negative selection in passengers.</p><disp-quote content-type="editor-comment"><p>(3) In many figures that show dN/dS as a function of n+s (starting with Figure 2A and extending to Figures S2, 3, 9, 10, 12, 22 and 25), there are no error bars indicated, as opposed to the statement in the figure caption. The error bars/shading should be shown. In Figure 2A, is the observed depletion in the second bin still significant?</p></disp-quote><p>To clarify, error bars were already displayed in all of the figures mentioned, but are currently visualized as shading (rather than traditional error bars). We recognize that the light opacity of this shading might make it difficult to visualize for some readers, especially in print. To correct this, we have now increased the opacity of the shading in all of the figures throughout the text so that this is clearer. We thank the reviewers for bringing up this concern.</p><p>After removing this filtering step in our analysis, we find that tumors in the second bin are no longer significant within Figure 2A.</p><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>1. Figure panels should be called out sequentially. For example, Figure 2G is called out before Figure 2D. This happens throughout the text, including main and supplementary figures, and should be corrected.</p></disp-quote><p>We thank the reviewer for bringing up this point and have now corrected the text.</p><disp-quote content-type="editor-comment"><p>2. Figure 2G shows that mean gene expression of genes encoding chaperones and the proteasome increases with increasing mutational burden. What about protein abundance? Is this in agreement with gene expression?</p></disp-quote><p>We agree with the reviewer that this would be an interesting avenue to explore further. Unfortunately, the only proteomics data that currently exists within TCGA is RPPA, which is very limited (only ~100 genes have been profiled across tumors), and these particular gene sets of interest have not yet been assayed across tumors.</p><disp-quote content-type="editor-comment"><p>3. Figure 2 mentions error bars in the figure legend, but no panel displays error bars. This is also true for Figure S13 and other figures. Authors should display the error bars to which they are referring to make their analysis more convincing.</p></disp-quote><p>As mentioned above, error bars are already displayed in all of the figures the reviewer has mentioned but are currently visualized as shading (rather than error bars). We recognize that the light opacity of this shading might make it difficult to visualize for some readers, especially in print. To correct this, we have now increased the opacity of the shading in the figures so that this is clearer throughout the text. We thank the reviewer for bringing up this concern.</p><disp-quote content-type="editor-comment"><p>4. Pg. 9 line 295 describes results of the analysis across genes belonging to different GO terms. However, Figure S13 only shows 3 categories: chromosome segregation, transcription and translation. How were these categories chosen? What about other categories? Such cherry picking doesn't convincingly support the conclusions that no specific GO functions are enriched. Also, translational regulation shows higher dN/dS in low mutation tumors suggesting that there is positive selection for passengers in this category. Authors should discuss in their manuscript why this is the case.</p></disp-quote><p>We appreciate the reviewer’s concern that there was little justification for why these gene sets were chosen for analysis. These were chosen based on previous literature that has supported these groups of genes being relevant to protein misfolding, and thus we hypothesized they might be under constraint. We have now extended the figure captions to include this.</p><disp-quote content-type="editor-comment"><p>5. Figure S15 shows the attenuation in selection of CNAs across cancer subtypes and broad cancer groups. However, HNSC and kidney cancer appear to be the exceptions. Authors should provide an explanation for these observations in the main text.</p></disp-quote><p>There are many potential reasons for why this signal might be missing in certain tumor types or broad tumor groups. For example, tumors that are not copy number driven or tend to be on the low-spectrum of CNAs overall, might simply not have enough CNAs across all tumors to generate a strong enough signal.</p><disp-quote content-type="editor-comment"><p>6. Generally, copy number variations are considered to be &gt; 50 bp. Is there a rationale as to why authors chose 100 kb to be their cut-off in Figure 2C? If the size of CNA is an important parameter, then authors should explain why that is.</p></disp-quote><p>100kb was used to partition copy number variation into ‘small’ and ‘large’ categories. Copy number variation &lt;100kb was still considered in our study. As mentioned in the text, the reason we chose this cut-off is that the average length of a gene is ~100kb. Thus, we would expect the selective effects of CNAs that are sufficiently large to disrupt genes or multiple genes to be stronger.</p><disp-quote content-type="editor-comment"><p>7. Non-allelic recombination and non-homologous recombination mechanisms involving replication accidents that lead to chromosome breakage occur with some frequency in somatic cells. How does the frequency of these events impact the selection efficiency in cancer as it relates to drivers and passengers? Can this also be incorporated in their evolutionary model?</p></disp-quote><p>We agree that this would be an interesting phenomenon to model. However, it’s hard to make inferences on how this would impact the efficacy of selection overall in drivers and passengers. More empirical information about how frequently these events occur on average per generation in a cell is needed before one can accurately model these scenarios. Thus, we believe this is beyond the scope of the paper.</p><disp-quote content-type="editor-comment"><p>8. Authors mentioned that haploinsufficiency was not used in the model. What about loss of heterozygosity which is extensive in cancer genomes? Can this parameter be included in the evolutionary model and how would it impact the results?</p></disp-quote><p>Our goal for the modeling section was to present a simple model of Hill-Robertson interference to examine if it's plausible that this could explain patterns of attenuated selection with mutational burden as observed in the data. There are many attributes of cancer biology that were not considered and how the relaxation of certain assumptions might impact the results has already been documented in Supplementary File 1. In addition, this particular point about haploinsufficiency has already been explored by others (Lopez et.al 2020 Nature Genetics) and thus, was not considered here.</p><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>1. The authors have taken great care to study single-nucleotide variants and large CNAs. It would be great if they could confirm their findings by also showing the effect on small insertions and deletions.</p></disp-quote><p>We agree with the reviewer that this would be an interesting direction to explore. However, we are not currently aware of any null models that take into account the expected behavior of neutral insertions/deletions. Thus, this is not something that we could explore in this manuscript.</p><disp-quote content-type="editor-comment"><p>2. Figure S5 is showing a bias in the determination of dN/dS from simulation results and the correlation between mutation rate and n+s. I am not sure I understand why dN/dS under a neutral simulation would be biased. Also, the low median correlation between n+s and the mutation rate (&lt;0.4) is quite surprising. I would have expected these to be almost perfectly correlated. Likewise, I do not understand the formula after l. 631. It states that this is the joint density of the two Poisson random variables that denote nonsynonymous and synonymous mutation count, yet there is an additional unexplained factor in the denominator, which corresponds to the probability of s&gt;0. If the simulation that underlies Figure S5 was also used in the ABC-based parameter inference, this would raise a serious cause for concern.</p></disp-quote><p>dN/dS in a neutral simulation could be biased because the probability of a nonsynonymous mutation is larger than the probability of a synonymous mutation (because of constraints of the genetic code). For low mutation counts, there are only certain integer totals possible (e.g. 2 nonsynonymous &amp; 1 synonymous) which might subtly bias results. In the previous submission, Figure S5 showed that this bias is very subtle. For clarity, however, we have elected to remove this figure in the resubmission.</p><p>The equation on line 631 was not used in the ABC-based parameter inference, and it also does not explicitly consider any selection coefficient. The denominator of this equation depends on three variables dN dS and <bold>λ</bold>S, which are the observed nonsynonymous and synonymous mutation counts, and the mean number of synonymous mutations. The subscript S in these equations represents ‘synonymous’ mutations and not ‘selection coefficient’; we apologize for this overloaded use of nomenclature – these nomenclature choices predate our study. In the resubmission, this equation now explicitly states its parameters.</p><disp-quote content-type="editor-comment"><p>3. The simulation starts when the first mutation with positive selective effect initiates population growth, which can be very late in a patient's life. How does the assumption that up to 100 years can pass after this affect the parameter estimates?</p></disp-quote><p>To clarify, the model assumes that a tumor has equal probability of arising (via an initating driver or 1<sup>st</sup> mutation with positive effect) at any point between 0 and 100 years of life. Thus, if a late-stage tumor arises (e.g. at age 70), the tumor doesn’t have 100 years from initiation (it has 30 years to become a successful tumor). This assumption has also been previously described in detail in (McFarland et al. 2014) and is common in mathematical models of cancer age-incidence rates. 4. To which extent does the inferred distribution of selection effects depend on the allowable parameter range? For example, s_passengers extends beyond the initially allowable range after the fit (Figure 3C).</p><p>The reviewed version of the manuscript had this figure reflecting ABC simulations extrapolating beyond the simulated range of parameters, which was needed to model the tail of the S_passenger probability distribution. This extrapolation may create unforeseen issues, so in the revised manuscript, we now simulate s_passengers from 10<sup>-4</sup> to 10<sup>-1</sup>, and the new figures are updated accordingly. The posterior distribution of s_passengers did not change substantially; our estimate is now that S_passengers is 1.03% 95% CI [0.39%, 4.0%].</p><disp-quote content-type="editor-comment"><p>5. It is not entirely clear to me how the partitioning of the likelihood between Muller's ratchet and hitchhiking vs other effects can be made and how robust these inferences are with respect to variation of the modeling assumptions (e.g. about initial population size or mode of selection). Is the necessity of inclusion of selected synonymous variants on driver genes a robust result or not, taking into account the discussion on p. 26f.?</p></disp-quote><p>In the revised manuscript, we now indicate that the rates of Muller’s Ratchet and hitchhiking are derived from analytical theory in the Figure Legend and cite this theory. The robustness of these results are well-discussed in the papers that derive these formulae.</p><p>Reviewer #1 also raised concerned about explanations for the dN/dS attenuation of drivers, so we have removed the Supplementary Note discussing this topic.</p><disp-quote content-type="editor-comment"><p>6. In Figure S4, the authors report the correlation of n+s with other measures of tumor mutation load. Given the relative sizes of the different regions that are displayed, i.e. whole genome:intergenic:intronic:exonic:protein-coding, of roughly 100:60:40:2:1, the displayed numbers do not make sense, as their ratios are 100:100:100:1:0.001.</p></disp-quote><p>This figure has now been removed in the re-submission.7. I am not sure I understood well how CNAs were analyzed. Based on the description in l. 669ff., it appears that putative cancer driver genes were identified from the CNA data based on recurrence. Were the same data then analyzed for CNAs falling into said putative cancer driver gene regions to infer selection? This would appear a bit circular.</p><p>The copy number specific drivers were called using GISTIC 2.0, which are indeed based on recurrence across cancer types. We appreciate the reviewer’s point that the some of the data used to identify drivers was also used to infer selection on these genes. However, the metric used to infer selection (dE/dI) is a different method/static from the one GISTIC uses. Furthermore, this is a problem that generally applies to all methods that are currently used to infer drivers, even for point mutation-specific drivers.</p><disp-quote content-type="editor-comment"><p>8. I do not understand the formula shown after li. 738. It appears it is showing the fraction of genes that intersect a CNA boundary, summed over all tumors in a given n+s bin. Each CNA can be counted twice if both of its boundaries fall into a gene. Why is the mean value of this 1?</p></disp-quote><p>Within the methods section, we reference Figure 2—figure supplement 8 where we show that dEdI of random CNAs (where the start and stop locations are randomly permuted across the genome) recapitulates expected neutral values of 1.9. In all figures that show dN/dS as a function of n+s (starting with Figure 2A and extending to Figures S2, 3, 9, 10, 12, 22 and 25), there are no error bars indicated, as opposed to the statement in the figure caption. In Figure 2A, is the observed depletion in the second bin still significant?</p><p>As mentioned above, error bars are already displayed in all of the figures the reviewer has mentioned but are currently visualized as shading (rather than error bars). We recognize that the light opacity of this shading might make it difficult to visualize for some readers and have corrected the figures by increasing the opacity. The observed depletion of the second bin in passengers is no longer significant.</p><disp-quote content-type="editor-comment"><p>10. In l. 290, I understand that the authors argue that differential dominance effects between heterozygous early- and late-arising mutations could be affecting the efficacy of selection on subclonal variants compared to clonal variants. I do not see this claim well motivated or corroborated.</p></disp-quote><p>To clarify, we don’t claim that dominance affects the efficacy of selection. We were stating that early-arising mutations, which tend to be low frequency variants that are likely to be heterozygous, are expected to experience weaker selection overall.</p><disp-quote content-type="editor-comment"><p>11. In Figure 2D, the caption states that mutations have been separated into two groups by their clonality, yet the figure shows three curves. What do they correspond to? Are the results still significant given the partitioning of the mutation data into smaller subsets?</p></disp-quote><p>The three curves were meant for the reader to visualize and compare how selection in drivers and passengers changes when all variants are included (variants of all VAF values), only subclonal variants are included (VAF &lt; 0.2) or only clonal variants are included (VAF &gt; 0.2). In the resubmission, we have now removed variants of all VAF values to make it simpler to compare selection on clonal vs subclonal variants. The revised results in the resubmission are significant for clonal mutations, but not subclonal mutations.<italic>12. Figure 3 does not have a panel G.</italic></p><p>This has now been corrected, thank you.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>I enjoyed reading the manuscript, it was well written, generally clear figures and very through provoking.</p></disp-quote><p>We thank the reviewer for these kind words.Only a small number of specific points to address:</p><disp-quote content-type="editor-comment"><p>Line 60.- The description of dN and dS here along with the interpretation of dN/dS=1 as neutral implies that you are just counting non-synonymous and synonymous mutations and dividing one by the other. This of course is not the case. Perhaps dN and dS could be described as rates or dN/dS as the dN:dS odds ratio which is how it's calculated for your permutation metric.</p></disp-quote><p>We appreciate the reviewer’s concern that the dN/dS calculations presented here are not simply counts of nonsynonymous and synonymous mutations. In the resubmission, we have now added the word rate in the main text to refer to our dN/dS calculations more accurately. However, we worry that adding additional terms to the metric ‘dN/dS’ can potentially add to more confusion – especially since others that calculate dN/dS with null models of mutagenesis simply use ‘dN/dS’ when referring to this statistic (Martincorena et al. 2017).</p><disp-quote content-type="editor-comment"><p>Line 112 – 40% of what, benefit of ~130% of what. This becomes apparent later into the manuscript but not clear how to interpret when reading at this point for the first time.</p></disp-quote><p>Thank you, we have now corrected this.</p><disp-quote content-type="editor-comment"><p>Line 188 – Mutational burden &lt;= 3 (what units).</p></disp-quote><p>Thank you, we have corrected the text.Line 189 – &quot;We observed little negative selection in passengers&quot; be clear what passengers (previous identified passenger genes).</p><p>We have changed the text to clarify this.</p><disp-quote content-type="editor-comment"><p>Line 252 – Panel G, y-axis, what units? Why are &quot;all&quot; genes uniformly at approximately -0.2? Assuming this is fold-change or log-fold-change I'd expect 1 or zero respectively.</p></disp-quote><p>These values are z-scale normalized and this normalization was already performed by COSMIC when the data was downloaded.</p><p>Line 275 – Figure 2G, shaded error bars are not visible.</p><p>Thank you, this has been fixed.</p></body></sub-article></article>