<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article article-type="research-article" dtd-version="1.2" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">71513</article-id><article-id pub-id-type="doi">10.7554/eLife.71513</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Mutation saturation for fitness effects at human CpG sites</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-159604"><name><surname>Agarwal</surname><given-names>Ipsita</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-8537-0008</contrib-id><email>ia2337@columbia.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-1167"><name><surname>Przeworski</surname><given-names>Molly</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5369-9009</contrib-id><email>mp3284@columbia.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf2"/></contrib><aff id="aff1"><label>1</label><institution>Department of Biological Sciences, Columbia University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Systems Biology, Columbia University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Ross-Ibarra</surname><given-names>Jeffrey</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, Davis</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Senior Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>22</day><month>11</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e71513</elocation-id><history><date date-type="received" iso-8601-date="2021-06-22"><day>22</day><month>06</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2021-11-21"><day>21</day><month>11</month><year>2021</year></date></history><permissions><copyright-statement>© 2021, Agarwal and Przeworski</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Agarwal and Przeworski</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-71513-v3.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-71513-figures-v3.pdf"/><abstract><p>Whole exome sequences have now been collected for millions of humans, with the related goals of identifying pathogenic mutations in patients and establishing reference repositories of data from unaffected individuals. As a result, we are approaching an important limit, in which datasets are large enough that, in the absence of natural selection, every highly mutable site will have experienced at least one mutation in the genealogical history of the sample. Here, we focus on CpG sites that are methylated in the germline and experience mutations to T at an elevated rate of ~10<sup>-7</sup> per site per generation; considering synonymous mutations in a sample of 390,000 individuals, ~ 99 % of such CpG sites harbor a C/T polymorphism. Methylated CpG sites provide a natural mutation saturation experiment for fitness effects: as we show, at current sample sizes, not seeing a non-synonymous polymorphism is indicative of strong selection against that mutation. We rely on this idea in order to directly identify a subset of CpG transitions that are likely to be highly deleterious, including ~27 % of possible loss-of-function mutations, and up to 20 % of possible missense mutations, depending on the type of functional site in which they occur. Unlike methylated CpGs, most mutation types, with rates on the order of 10<sup>-8</sup> or 10<sup>-9</sup>, remain very far from saturation. We discuss what these findings imply for interpreting the potential clinical relevance of mutations from their presence or absence in reference databases and for inferences about the fitness effects of new mutations.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>mutation saturation</kwd><kwd>fitness effects</kwd><kwd>cpg mutations</kwd><kwd>interpreting protein coding variants</kwd><kwd>human genetics</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM122975</award-id><principal-award-recipient><name><surname>Przeworski</surname><given-names>Molly</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM121372</award-id><principal-award-recipient><name><surname>Przeworski</surname><given-names>Molly</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Methylated CpG sites are saturated for T mutations in a sample of 390K human exomes, providing a test case for inferences about fitness effects in human genes, and insight into the interpretation of mutations as pathogenic using reference datasets.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>A central goal of human genetics is to identify pathogenic mutations and predict how likely they are to cause disease. To this end, exome sequencing in cases and controls is often used to help identify variants with potentially large effects on disease risk (<xref ref-type="bibr" rid="bib45">Rauch et al., 2012</xref>; <xref ref-type="bibr" rid="bib48">Sanders et al., 2012</xref>; <xref ref-type="bibr" rid="bib40">Need et al., 2012</xref>; <xref ref-type="bibr" rid="bib3">Akbari et al., 2021</xref>). Even where this approach yields an enrichment of variants in cases, however, the specific subset of mutations that contributes to disease often remains unknown; similarly, in individual patients, sequencing habitually yields candidate mutations of which the significance is unclear (<xref ref-type="bibr" rid="bib47">Richards et al., 2015</xref>; <xref ref-type="bibr" rid="bib24">Harrison et al., 2021</xref>).</p><p>Numerous scores have therefore been developed to help prioritize among candidate mutations, based on protein structure, functional annotations, evolutionary patterns, or other features (<xref ref-type="bibr" rid="bib11">Cooper et al., 2005</xref>; <xref ref-type="bibr" rid="bib1">Adzhubei et al., 2010</xref>; <xref ref-type="bibr" rid="bib43">Pollard et al., 2010</xref>; <xref ref-type="bibr" rid="bib37">McLaren et al., 2016</xref>; <xref ref-type="bibr" rid="bib26">Ioannidis et al., 2016</xref>; <xref ref-type="bibr" rid="bib46">Rentzsch et al., 2019</xref>). In particular, a common approach to pinpoint sites at which mutations are likely to be pathogenic is to examine whether they appear to be under purifying selection. For instance, comparisons of sequences across species have been widely used to identify highly conserved genomic regions maintained by selection over millions of years, presumably because of their functional importance (<xref ref-type="bibr" rid="bib11">Cooper et al., 2005</xref>; <xref ref-type="bibr" rid="bib43">Pollard et al., 2010</xref>; <xref ref-type="bibr" rid="bib6">Boffelli et al., 2003</xref>; <xref ref-type="bibr" rid="bib52">Siepel et al., 2005</xref>).</p><p>The same general approach is also useful when applied within humans, where information about purifying selection is contained in whether or not a site is segregating a mutation and at what frequency (<xref ref-type="bibr" rid="bib49">Sawyer and Hartl, 1992</xref>; <xref ref-type="bibr" rid="bib16">Eyre-Walker and Keightley, 2007</xref>; <xref ref-type="bibr" rid="bib7">Boyko et al., 2008</xref>; <xref ref-type="bibr" rid="bib63">Williamson et al., 2005</xref>; <xref ref-type="bibr" rid="bib35">Lek et al., 2016</xref>; <xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib64">Yi et al., 2010</xref>). For this application, however, the low diversity levels in the genome pose a major difficulty, as a site may be monomorphic simply by chance, that is when mutations at that site have no fitness consequences at all or, at the other extreme, because the mutations are embryonically lethal. In particular, because most sites are monomorphic in samples of hundreds or even thousands of humans, there is little information to distinguish sites under strong selection from those at which mutations are only weakly deleterious.</p><p>With a view to capturing natural variation at a larger number of sites in the genome and identifying more mutations with large effects on disease risk, there have been extensive efforts to collate available exome sequences from hundreds of thousands of individuals (<xref ref-type="bibr" rid="bib35">Lek et al., 2016</xref>; <xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib12">Dewey et al., 2016</xref>; <xref ref-type="bibr" rid="bib58">Szustakowski, 2020</xref>; <xref ref-type="bibr" rid="bib60">Van Hout et al., 2020</xref>; <xref ref-type="bibr" rid="bib59">Taliun et al., 2021</xref>). These efforts were also motivated by the idea that public repositories composed of relatively healthy adults not ascertained for a specific severe disease can serve as reference datasets, such that seeing a variant of unknown function in these datasets is indicative of it being benign (<xref ref-type="bibr" rid="bib35">Lek et al., 2016</xref>; <xref ref-type="bibr" rid="bib10">Claussnitzer et al., 2020</xref>; <xref ref-type="bibr" rid="bib19">Ghouse et al., 2018</xref>). The validity of that assumption remains to be evaluated, however, especially as the repositories grow in size.</p><p>Beyond their utility in human genetics, these datasets provide an unprecedented opportunity to learn about the fitness effects of new mutations. Modeling the distribution of fitness effects (DFE) has a long history in population genetics (<xref ref-type="bibr" rid="bib49">Sawyer and Hartl, 1992</xref>; <xref ref-type="bibr" rid="bib16">Eyre-Walker and Keightley, 2007</xref>; <xref ref-type="bibr" rid="bib42">Otto, 2000</xref>), but until recently, inferences were based on genetic variation in samples of at most a couple of thousand chromosomes (<xref ref-type="bibr" rid="bib16">Eyre-Walker and Keightley, 2007</xref>; <xref ref-type="bibr" rid="bib7">Boyko et al., 2008</xref>; <xref ref-type="bibr" rid="bib63">Williamson et al., 2005</xref>; <xref ref-type="bibr" rid="bib15">Eyre-Walker et al., 2006</xref>; <xref ref-type="bibr" rid="bib32">Kim et al., 2017</xref>). As is well appreciated, the fitness effects at the few sites segregating at such sample sizes are a small and biased draw from the DFE and thus the inferred distribution of fitness effects is unlikely to recapitulate the true DFE in the genome. Moreover, for lack of sufficient information with which to distinguish weakly from strongly selected mutations, a number of approaches have relied on a specific and arbitrary parametric form for the distribution of fitness effects across sites. In that regard, not only do inferences based on small samples result in relatively noisy parameter estimates, the results can be misleading, especially about the fraction of sites under strong selection (<xref ref-type="bibr" rid="bib32">Kim et al., 2017</xref>). Current samples in humans may allow for these limitations to start to be overcome.</p><p>Motivated by these considerations, we focus on a class of mutations known to experience mutations an order of magnitude more frequently than other types of sites in the human genome: CpG sites that are methylated in the germline (<xref ref-type="bibr" rid="bib14">Duncan and Miller, 1980</xref>; <xref ref-type="bibr" rid="bib39">Nachman and Crowell, 2000</xref>; <xref ref-type="bibr" rid="bib33">Kong et al., 2012</xref>; <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). We use these sites as a test case for what can be learned about selection when neutral sites are saturated, i.e., have all experienced at least one mutation in the history of the sample, and draw out implications for the interpretation of mutations as pathogenic and for inferences about fitness effects.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Mutation saturation at CpGs</title><p>An attractive feature of methylated CpG (mCpG) sites is that a single mechanism, the spontaneous deamination of methyl-cytosine, is believed to underlie the uniquely high rate of C &gt; T mutations at these sites (<xref ref-type="bibr" rid="bib14">Duncan and Miller, 1980</xref>); thus, germline methylation at CpG sites is strongly predictive of their mutability (<xref ref-type="bibr" rid="bib33">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib27">Jónsson et al., 2017</xref>; <xref ref-type="bibr" rid="bib18">Gao et al., 2019</xref>; <xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>). Here, we define ‘methylated’ CpG sites in exons as those that are methylated ≥65 % of the time in both testes and ovaries. For these ~1.1 million sites (of 1.8 million total CpG sites in sequenced exons), we calculate a mean haploid, autosomal C &gt; T mutation rate of 1.17 × 10<sup>–7</sup> per generation using de novo mutations (DNMs) in a sample of ~2900 sequenced parent-offspring trios (Materials and methods, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplements 1</xref>–<xref ref-type="fig" rid="fig1s2">2</xref>, <xref ref-type="bibr" rid="bib22">Halldorsson et al., 2019</xref>).</p><p>Although methylation levels are the dominant predictor of mutation rates at CpG sites, they are not the only influence. Notably, CpG transitions differ somewhat in their mutation rates based on their trinucleotide context (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3a</xref>; <xref ref-type="bibr" rid="bib2">Aggarwala and Voight, 2016</xref>); even so, they are consistently an order of magnitude higher than the genome average (<xref ref-type="bibr" rid="bib33">Kong et al., 2012</xref>). Broader scale features, such as replication timing, have also been reported to shape mutation rates (<xref ref-type="bibr" rid="bib56">Stamatoyannopoulos et al., 2009</xref>; <xref ref-type="bibr" rid="bib54">Smith et al., 2018</xref>). Nonetheless, considering methylated CpGs inside and outside exons, which differ in a number of these features, there is no appreciable difference in average DNM rates (Fisher Exact Test (FET) p-value = 0.1, <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4a</xref>). Similarly, the rate at which two DNMs occur at the same site, a summary statistic that reflects the variance in mutation rates, is not significantly different for methylated CpGs inside versus outside exons (FET p-value = 0.35; <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4b</xref>). Thus, while there is some variation in mutability per site among methylated CpGs, it appears to be small relative to the mean mutation rate across all methylated CpGs.</p><p>Considering all such CpG sites therefore, we ask what fraction are segregating at existing sample sizes. To this end, we collate polymorphism data made public by gnomAD (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>), the UK Biobank (<xref ref-type="bibr" rid="bib58">Szustakowski, 2020</xref>), and the DiscovEHR collaboration between the Regeneron Genetics Center and Geisinger Health System (<xref ref-type="bibr" rid="bib12">Dewey et al., 2016</xref>) in order to ascertain whether both C and T alleles are present in a sample of ~390K individuals (Materials and methods).</p><p>To focus on the subset of genic changes most likely to be neutrally-evolving, we consider the ~350,000 methylated CpG sites at which C &gt; T mutations do not change the amino acid. At these sites, 94.7 % of all possible synonymous CpG transitions are observed in the gnomAD data alone, and 98.8 % in the combined sample including all three datasets (<xref ref-type="fig" rid="fig1">Figure 1</xref>). In other words, nearly every methylated CpG site where a mutation to T is putatively neutral has experienced at least one such mutation in the history of the sample of 390K individuals. Even in the least mutable CpG trinucleotide context, 98 % of putatively neutral sites are segregating in current samples (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3b</xref>). These observations imply that in the absence of selection, almost every methylated CpG site would be segregating a T at current sample sizes--and further that not seeing a T provides strong evidence it was removed by selection.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Fraction of methylated CpG sites that are polymorphic for a transition, by sample size.</title><p>The combined dataset encompasses three non-overlapping data sources: gnomAD (v2.1), the UK Biobank (UKB), and the DiscovEHR cohort. ‘European’ samples include the populations designated as ‘EUR’ in 1000 Genomes, ‘Non-Finnish European’ subsets of exome and whole genome datasets in gnomAD, as well as the UK Biobank and DiscovEHR, which have &gt;90% samples labeled as of European ancestry.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig1-v3.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Exonic de novo mutation rates per generation per site estimated from a sample of 2976 parent-offspring trios data from <xref ref-type="bibr" rid="bib22">Halldorsson et al., 2019</xref>, by mutation type.</title><p>‘mCpG’ refers to a CpG site with methylation level ≥65 % in both testes and ovaries, and ‘other CpG’ to a CpG site with methylation level &lt;65% in either testes or ovaries. Error bars reflect the 95 % Poisson confidence interval around mutation counts for each type.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig1-figsupp1-v3.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>De novo mutation rate in exons in a sample of 2976 parent-offspring trios, by average methylation levels.</title><p>(a) De novo mutation rate by average methylation levels in testes (b) De novo mutation rate by average methylation levels in ovaries. Error bars reflect the 95 % Poisson confidence interval around mutation counts in each group (the minimum number of DNMs in each bin is 5).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig1-figsupp2-v3.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Effect of trinucleotide context on mutation rate and mutation saturation at methylated CpG sites.</title><p>(<bold>a</bold>) Exonic de novo mutation rates at methylated CpG sites, by trinucleotide context. Error bars reflect the 95% Poisson confidence interval around mutation counts for each context. (<bold>b</bold>) Fraction of possible synonymous C &gt; T mutations at methylated CpG sites that are observed in a sample of given size, by trinucleotide context.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig1-figsupp3-v3.tif"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Comparing the distribution of CpG transition rates at methylated sites within and ouside exons.</title><p>(<bold>a</bold>) DNM rates for CpG transitions at methylated sites in exons (exonic regions obtained from Gencode V19; see Materials and methods) vs. non-exons, with 95 % Poisson confidence intervals. (<bold>b</bold>) The rate of single hits (one DNM at a site) and double hits (two DNMs at a site) in exons vs non-exons, rescaled to the average rate of single and double hits in the genome, with 95 % Poisson confidence intervals.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig1-figsupp4-v3.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Testing a neutral model for individual sites</title><p>The mutation saturation at methylated CpG sites provides a robust approach to identify individual sites that are not neutrally-evolving. One way to view it is in terms of a p-value: under a null model with no selection, from which we assume that synonymous sites are drawn, all but 1.2 % of neutral sites are segregating in a sample of 390K individuals. Therefore, if a given non-synonymous site, say, is invariant in a sample of ≥390K individuals, we can reject the neutral null model for this site at a significance level of 0.012. Similarly, we can ask about the probability that an invariant non-synonymous site is neutral, using a false discovery rate (FDR) approach: given that 1.2 % of neutral sites are invariant, whereas 7.4 % of non-synonymous sites are, the FDR is 1.2/7.4 = 16%. Thus, at current sample sizes, there is a substantial amount of information about whether individual CpG transitions are deleterious. By contrast, in a smaller sample with only 10 % of putatively neutral sites segregating, there is almost no information about selection in observing individual sites to be invariant (p ≤ 0.9).</p><p>This approach implicitly assumes that synonymous and non-synonymous sites do not differ in their distributions of mutation rates and that their distributions of genealogical histories are also the same, i.e.,, that the two types of sites are subject to comparable effects of linked selection. While we cannot examine whether the distributions of mutation rates are identical for lack of data, we verify that the mean de novo mutation rates do not differ for synonymous sites and for various non-synonymous annotations (<xref ref-type="fig" rid="fig2">Figure 2a</xref>); we also check that the distributions of methylation levels (conditional on ≥65%), an important determinant of mutation rates, are similar for synonymous and non-synonymous sites (with a significant but small shift towards higher methylation and thus presumably higher mutation rates for non-synonymous sites; <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). In turn, the standard assumption of similar distributions of genealogical histories seems sensible, given that the sites are interdigitated within genic regions (<xref ref-type="bibr" rid="bib36">McDonald and Kreitman, 1991</xref>). Under these few and at least somewhat testable assumptions, the approach based on mutation saturation at methylated CpG sites then enables us to directly pinpoint individual sites that are not neutrally evolving. We note further that if synonymous sites are not all neutral and instead some fraction are under selection, the same idea would apply, but the null model would have to be modified accordingly.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Comparing de novo mutation rates and the fraction of segregating sites across annotations.</title><p>(<bold>a</bold>) DNM rates for CpG transitions at highly methylated sites by annotation class, rescaled by the total DNM rate in exons. Fisher exact tests (FETs) of the proportion of sites with DNMs in each annotation compared to all other annotations yield p-values &gt; 0.1 in all cases. (<bold>b</bold>) Fraction of highly methylated CpG sites that are segregating as a C/T polymorphism in an annotation class, relative to the fraction of synonymous sites segregating. Error bars are 95 % confidence intervals assuming the number of segregating sites is binomially distributed (FET p-values &lt;&lt; 10<sup>–5</sup> for comparisons of all annotations with synonymous sites). LOF variants are defined as stop-gained and splice donor/acceptor variants that do not fall near the end of the transcript, and meet the other criteria to be classified as ‘high-confidence’ loss-of-function in the gnomAD data (see Materials and Methods). (<bold>c</bold>) The amount of data for synonymous and missense changes involving highly methylated CpG transitions by the type of functional protein site. (<bold>d</bold>) The proportion of synonymous and missense segregating C/T polymorphisms in different classes of functional sites. Error bars are 95 % confidence intervals assuming the number of segregating sites is binomially distributed (FET p-values &lt;&lt; 10<sup>–5</sup> for comparisons of all missense annotations with synonymous sites; Materials and methods). All annotations are obtained using the canonical transcripts of protein coding genes (see Materials and methods).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-v3.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Distribution of methylation levels at synonymous and non-synonymous methylated CpG sites in testes and ovaries.</title><p>(<bold>a</bold>) The distribution of methylation levels in testes (chi-squared test p-value &lt;&lt; 10<sup>–5</sup>) (<bold>b</bold>) The distribution of methylation levels in ovaries (chi-squared test p-value &lt;&lt; 10<sup>–5</sup>). The small but significant shift towards higher methylation for non-synonymous sites compared to synonymous ones suggests a small shift towards higher mutation rates at these sites compared to synonymous sites, which should be conservative with regard to identifying non-synonymous sites under selection.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-figsupp1-v3.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>The effect of background selection on the fraction of sites segregating in each annotation.</title><p>(<bold>a</bold>) Cumulative distribution of the B-statistic from <xref ref-type="bibr" rid="bib38">McVicker et al., 2009</xref> for all possible CpG transitions at methylated sites by annotation class. (<bold>b</bold>) Fraction of methylated CpG sites that are segregating as a C/T polymorphism in an annotation class, relative to the fraction of synonymous sites segregating, after matching the distribution of the B-statistic across annotations. The fraction segregating without matching for B-statistics (shown in <xref ref-type="fig" rid="fig2">Figure 2b</xref>) is denoted by crosses, to enable comparison. Regulatory variants include non-LOF splice region variants and UTRs. (<bold>c</bold>) Cumulative distribution of the B-statistic for all possible CpG transitions at methylated sites by functional class. (<bold>d</bold>) The proportion of synonymous and missense segregating C/T polymorphisms for four functional classes, after matching the distribution of the B-statistic across categories. Error bars are 95 % confidence intervals assuming the number of segregating sites is binomially distributed. The fraction segregating without matching for B-statistics (shown in <xref ref-type="fig" rid="fig2">Figure 2d</xref>) is denoted by crosses, to enable comparison.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-figsupp2-v3.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Comparing LOF CpG transitions at methylated sites in exons that constitute the first vs. second halves of canonical protein coding transcripts.</title><p>(<bold>a</bold>) DNM rates for synonymous and LOF CpG transitions at methylated sites in exons that constitute the first vs. second halves of canonical protein coding transcripts, rescaled to the total DNM rate in exons, with 95 % Poisson confidence intervals. (<bold>b</bold>) Fraction of methylated CpG sites that are segregating as a synonymous or LOF C/T polymorphism in exons that constitute the first vs. second halves of canonical protein coding transcripts, relative to the fraction of all synonymous sites segregating. Error bars are 95 % confidence intervals assuming the number of segregating sites is binomially distributed (see Methods). LOF variants are defined as stop-gained and splice donor/acceptor variants that do not fall near the end of the transcript, and meet the other criteria to be classified as ‘high-confidence’ loss-of-function in gnomAD (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-figsupp3-v3.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Comparing de novo mutation rates and the fraction of segregating sites across annotations obtained using the worst consequence in protein coding transcripts by predicted severity, instead of canonical transcripts as in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</title><p>The order of preference by which functional sites are assigned to a single category is detailed in Materials and methods. (<bold>a</bold>) DNM rates for CpG transitions at methylated sites by annotation class, rescaled by the total DNM rate in exons, with 95 % Poisson confidence intervals (<bold>b</bold>) Fraction of methylated CpG sites that are segregating as a C/T polymorphism in an annotation class, relative to the fraction of synonymous sites segregating. Error bars are 95 % confidence intervals assuming the number of segregating sites is binomially distributed. LOF variants are defined as stop-gained and splice donor/acceptor variants that do not fall near the end of the transcript, and meet the other criteria to be classified as “high-confidence” loss-of-function in gnomAD. (<bold>c</bold>) The number of opportunities for synonymous and missense changes involving methylated CpG transitions by the type of functional protein site. (<bold>d</bold>) The proportion of synonymous and missense segregating C/T polymorphisms in different classes of functional sites.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-figsupp4-v3.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>De novo C&gt;T mutation rates at methylated CpG sites and the fraction of sites segregating in CADD score bins.</title><p>(<bold>a</bold>) De novo C &gt; T mutation rate at methylated CpGs in deciles of CADD scores in exons, rescaled by the total rate of methylated CpG transitions in exons. Error bars reflect the 95 % Poisson confidence interval around mutation counts in each group. (<bold>b</bold>) Fraction of methylated CpG sites that are segregating as a C/T polymorphism in a CADD score decile, relative to the fraction of synonymous sites segregating. (<bold>c</bold>) The same as (<bold>a</bold>) but for C &gt; T mutations at all CpG sites, including unmethylated and less methylated CpGs as well as methylated ones. (<bold>d</bold>) The same as (<bold>b</bold>) but for C &gt; T mutations at all CpG sites. Higher CADD scores reflect stronger predicted constraint.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig2-figsupp5-v3.tif"/></fig></fig-group></sec><sec id="s2-3"><title>Comparing the fraction of segregating sites across annotations</title><p>Under these same weak assumptions, it is also possible to compare the proportion of methylated CpG sites polymorphic for a transition across annotations. Here, we consider the fraction of sites segregating a transition in each annotation class in a sample of 780K chromosomes, rescaled by the fraction segregating at synonymous sites. All categories of missense, loss-of-function, and regulatory variants show a significant depletion in the fraction of segregating sites compared to synonymous variants (<xref ref-type="fig" rid="fig2">Figure 2b</xref>). The deficit for a given annotation is an indicator of the deleteriousness of de novo mutations in that annotation. Specifically, in our sample of 780K, the deficit for each annotation reflects sites for which we can reject neutrality at a significance level of 0.012.</p><p>These data therefore suggest that there are ~27 % fewer loss-of-function variants than would be expected under neutrality; at invariant sites within this annotation, neutrality can be rejected at an FDR of only 4.4 % ( = 1.2/27). A 27 % deficit of loss-of-function variants is again seen if we match the sites to synonymous mutational opportunities with the same predicted level of linked selection, i.e., with similar genealogical histories (<xref ref-type="bibr" rid="bib38">McVicker et al., 2009</xref>; <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2a</xref>). Supporting the widely used assumption that LOF mutations within a gene are equivalent (after filtering for those at the end of transcripts; <xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib9">Cassa et al., 2017</xref>), when we compare the set of CpG sites at which mutations are annotated as leading to protein-truncation in the first versus the second half of transcripts, approximately the same number are missing mutations relative to synonymous sites in both subsets (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>; FET p-value = 0.9). By comparison, the fraction of missense mutations and splice region variants not observed in current samples is only about 5.3%, and the FDR 22.6 % ( = 1.2/5.3) (whether or not we match for the effects of linked selection; see <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2a</xref>).</p><p>While LOF and missense annotation classes are most commonly used in determinations of variant pathogenicity, any two sets of methylated CpGs with similarly-distributed mutation rates can be ranked in this manner. As one example, we stratify missense mutations by the type of functional site in which they occur. For the subset of sites at which missense mutations may disrupt or alter binding, particularly DNA-binding, there is a ~ 12–20% deficit in segregating sites relative to what is seen at synonymous sites, in contrast, say, to the much smaller deficit at missense changes within trans-membrane regions (<xref ref-type="fig" rid="fig2">Figure 2c–d</xref>, <xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>; <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2b</xref>). In other words, observing a DNA-binding missense site that is invariant provides stronger statistical evidence that it is deleterious than observing an invariant missense site with no additional functional information (e.g. the FDR is ~1/20 vs. ~1/5).</p><p>We can also check that the fraction of sites segregating is inversely proportional to the predicted functional importance of the sites using CADD scores (<xref ref-type="bibr" rid="bib46">Rentzsch et al., 2019</xref>), widely used measures of constraint that incorporate functional annotations and measures of conservation. Across deciles, mean de novo transition rates at methylated CpGs are similar (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5a</xref>) and, as expected, the fraction of segregating sites decreases with increasing CADD scores (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5b</xref>). We note, however, that mutation rates may not always be similar across comparison groups: considering all CpG sites in exons (i.e. not only highly methylated ones), for example, de novo mutation rates are much more variable across CADD deciles (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5c</xref>). Consequently, the depletion of segregating sites no longer has a simple interpretation (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5d</xref>), instead reflecting a combination of differences in mutation rates and fitness effects. By implication, while CADD scores are meant to isolate the effects of selection, they will in some cases classify sites that have high mutation rates as less constrained, and vice versa.</p></sec><sec id="s2-4"><title>What can be learned about other mutation types?</title><p>Given that current exome samples are informative about selection on transitions at methylated CpGs, a natural question is to ask to what extent there is also information for less mutable types, with mutation rates on the order of 10<sup>–8</sup> or 10<sup>–9</sup> per site per generation. For sites with mutation rate on the order of 10<sup>–9</sup>, which is the case for the vast majority of non-CpGs, the fraction of possible synonymous sites that segregate in a sample of 780K chromosomes is very low: for instance, it is 5 % for T &gt; A mutations, which occur at an average rate of 1.2 × 10<sup>–9</sup> (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>) and 27 % even for other C &gt; T mutations, which occur at a rate of 0.9 × 10<sup>–8</sup> per site (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), compared to ~99 % for C &gt; T mutations at methylated CpGs (<xref ref-type="fig" rid="fig3">Figure 3a</xref>). For invariant sites of these less mutable types, there is little information with which to evaluate the fit to the neutral null in current samples. Reflecting this lack of information, in the p-value formulation, monomorphic sites would be assigned p ≤ 0.95 for T &gt; A sites and p ≤ 0.73 for C &gt; T sites.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Comparing the fraction of sites observed and expected to be segregating under neutrality, by mutation type and sample size.</title><p>(<bold>a</bold>) Fraction of possible synonymous C &gt; T mutations at CpG sites methylated in the germline and at all other C sites, and the fraction of possible synonymous T &gt; A mutations that are observed in a sample of given size. (<bold>b</bold>) Fraction of sites segregating in simulations, assuming neutrality, a specific demographic model and a given mutation rate (see Materials and methods).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig3-v3.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>The expected length of the genealogy under different demographic models and for varying sample sizes.</title><p>(<bold>a</bold>) The expected number of neutral mutations at a site, for three mutation rates and varying sample sizes, calculated as the expected length of the genealogy (sum of branch lengths, averaged over 20 simulations) multiplied by the mutation rate, for a CEU population with a recent <italic>N<sub>e</sub></italic> of 10 million for the last 50 generations (see Materials and Methods). (<bold>b</bold>) A comparison of mean genealogy lengths for the standard Schiffels-Durbin demographic model for a CEU population and three variations with increased current <italic>N<sub>e</sub></italic>, namely, CEU demographic history for 50,000 generations with a recent <italic>N<sub>e</sub></italic> of 10 million or 100 million for the last 50 generations, and CEU demographic history with 4.5 % exponential growth for the past ~200 generations. (<bold>c</bold>) A comparison of mean genealogy lengths for samples from YRI and CEU populations, and samples from a structured population derived from an ancestral population 2000 generations ago.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig3-figsupp1-v3.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Mutation saturation in bins of sites compared to single mCpG sites.</title><p>(<bold>a</bold>) <italic>k</italic>, the number of T sites per bin, such that the average T &gt; A mutation rate per bin is the same as the average transition rate at a single methylated CpG site in that annotation (<bold>b</bold>) Fraction of bins of synonymous T sites that have at least one T/A polymorphism. A cross is indicated for the corresponding fraction at synonymous methylated CpG sites. As expected if synonymous sites are neutral and the mutation rate for a bin matches that of methylated CpGs, the two fractions are very similar. (<bold>c</bold>) Fraction of bins that have at least one T/A polymorphism, by non-synonymous annotation. A cross is indicated for the corresponding fraction at methylated CpG sites. Error bars are 95 % confidence intervals assuming the number of segregating bins is binomially distributed. For bins including sites under selection the fractions for CpG sites and other mutation types are not expected to match, depending on the extent of variation in mutation rates and fitness effects across sites within a bin (see Materials and methods).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig3-figsupp2-v3.tif"/></fig></fig-group><p>How large samples have to be for other mutation types to reach saturation depends on the length of the genealogy that relates sampled individuals, i.e., the sum of the branch lengths, which corresponds to the number of generations over which mutations could have arisen at the site. For a mutation that occurs at rate 1.17 × 10<sup>–7</sup> per generation, the average length of the genealogy would have to be greater than 8.5 million (1/1.17 × 10<sup>–7</sup>) generations for at least one such mutation to be expected at a site. That synonymous CpG sites are close to saturation when they experience mutations to T at this rate suggests that this is in fact the case. Indeed, given that more than one mutation has occurred at a substantial fraction of sites (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib23">Harpak et al., 2016</xref>), the average length of the genealogy relating the 390K individuals is expected to be substantially longer: about 39 million generations (calculated from the probability of at least one mutation under a Poisson distribution; see Materials and methods). The observation that mutation types with rates on the order of 10<sup>–8</sup> are far from saturation further indicates that the average length of the genealogy for these 390K individuals is substantially shorter than 100 million generations. These rough calculations thus provide a sense of the length of the genealogical history represented by these 390K individuals.</p><p>To explicitly examine the relationship between sample size, mutation rate and the amount of variation at a locus, we simulate neutral evolution at a single site with the three different mutation rates above, under a variant of the widely-used Schiffels-Durbin demographic model for population growth in Europe (<xref ref-type="bibr" rid="bib50">Schiffels and Durbin, 2014</xref>), in which we set the effective population size <italic>N</italic><sub>e</sub> equal to 10 million for the past 50 generations (Methods). While this model is clearly an oversimplification, it recapitulates observed diversity levels for synonymous mutations reasonably well (<xref ref-type="fig" rid="fig3">Figure 3</xref>). Consistent with the rough estimate above, under our choice of demographic model, a sample of 780K chromosomes has a genealogy spanning an average of 34 million generations (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1a,b</xref>).</p><p>From first principles, the length of the genealogy is expected to increase much more slowly than linearly with the number of samples (<xref ref-type="bibr" rid="bib25">Hudson, 1990</xref>; <xref ref-type="bibr" rid="bib41">Nelson et al., 2012</xref>). Indeed, increasing the number of samples by a factor of 12 only increases the average tree length ~3.3 x (<xref ref-type="fig" rid="fig3">Figure 3b</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1a,b</xref>); thus, a site that mutates at rate 10<sup>–9</sup> per generation is expected to have experienced ~0.04 mutations in the genealogical history of a sample of ~1 million, and 0.1 mutations in a sample of 10 million. The implication is that saturation for mutation rates of 10<sup>–8</sup> or 10<sup>–9</sup> per site per generation may not be achievable any time soon.</p><p>Quantitative predictions of our model are subject to the considerable uncertainty about the demographic history and in particular about the recent effective population size in humans (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1b</xref>). Moreover, for simplicity, we model one or at most two populations, when samples that combine individuals from more diverse genetic ancestries have longer genealogical histories (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1c</xref>; see <xref ref-type="fig" rid="fig1">Figure 1</xref>) and thus capture more variation. Perhaps most importantly, for the very large sample sizes considered here, the multiple merger coalescent is a more appropriate model (<xref ref-type="bibr" rid="bib41">Nelson et al., 2012</xref>; <xref ref-type="bibr" rid="bib5">Bhaskar et al., 2014</xref>). Nonetheless, the qualitative statement that less mutable types will remain very far from saturation in the foreseeable future should hold.</p><p>In the absence of information about single sites for most mutational types in the genome, it is still possible to learn to a limited degree about selection using bins of sites. If we construct a bin of <italic>K</italic> synonymous sites with the same average mutation rate per bin as a single methylated CpG, then at least one site per bin is polymorphic in ~99 % of bins (see <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref> for an example with T &gt; A mutations and <italic>K</italic>~100), just as ~99 % of individual methylated CpG sites are segregating. Thus, if a bin of <italic>K</italic> non-synonymous sites with the same average mutation rate is invariant, the p-value associated with the bin is 0.01, indicating that one or more sites in the bin is likely to be under selection.</p></sec><sec id="s2-5"><title>How strong is the selection that leads to invariant methylated CpG sites?</title><p>Leveraging saturation to identify a subset of sites that are not neutrally-evolving makes appealingly few assumptions, but provides no information about how strong selection is at those sites. To learn about the strength of selection consistent with methylated CpG sites being monomorphic, a series of strong assumptions are needed: we require a demographic model, a prior distribution on <italic>hs</italic> and a mutation rate distribution across sites. Here, we assume a relatively uninformative log-uniform prior on the selection coefficient <italic>s</italic> ranging from 10<sup>–7</sup> to 1 and fix the dominance coefficient <italic>h</italic> = 0.5 (as for autosomal mutations with fitness effects in heterozygotes, we only need to specify the compound parameter <italic>hs</italic>; reviewed in <xref ref-type="bibr" rid="bib17">Fuller et al., 2019</xref>), as well as a fixed mutation rate of 1.2 × 10<sup>–7</sup> per site per generation. We rely on the demographic model for population growth in Europe described above (<xref ref-type="bibr" rid="bib50">Schiffels and Durbin, 2014</xref>); as is standard (<xref ref-type="bibr" rid="bib49">Sawyer and Hartl, 1992</xref>; <xref ref-type="bibr" rid="bib7">Boyko et al., 2008</xref>; <xref ref-type="bibr" rid="bib63">Williamson et al., 2005</xref>; <xref ref-type="bibr" rid="bib15">Eyre-Walker et al., 2006</xref>; <xref ref-type="bibr" rid="bib32">Kim et al., 2017</xref>; <xref ref-type="bibr" rid="bib9">Cassa et al., 2017</xref>; <xref ref-type="bibr" rid="bib53">Simons et al., 2014</xref>), we also assume that <italic>hs</italic> is fixed over time, even as the effective population size changes dramatically. Under these assumptions, we estimate the posterior distribution of <italic>hs</italic> at a site, given that the site is monomorphic, segregating with 10 or fewer derived copies of the T allele, or segregating with more than 10 copies (<xref ref-type="fig" rid="fig4">Figure 4a and b</xref>, Methods). These posterior distributions are estimates of the DFE at an individual mCpG site conditional on seeing 0 copies, 1–10 copies or &gt;10 copies of the T allele.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Quantifying the strength of selection associated with invariant and segregating sites.</title><p>(<bold>a</bold>) Prior and Posterior log densities of <italic>hs</italic> for a C &gt; T mutation at a methylated CpG site observed at 0, 1–10, or &gt;10 copies at various sample sizes. (<bold>b</bold>) Bayes odds (i.e. posterior odds divided by prior odds) of <italic>s</italic> &gt; 0.001 for a C &gt; T mutation at a methylated CpG site observed at 0, 1–10, or &gt;10 copies, at various sample sizes. (<bold>c</bold>) Probability of a methylated CpG site segregating a T allele in simulations, if the mutation has no fitness effects (hs = 0) and if it is deleterious (with a heterozygote selection coefficient hs = 0.05%) or highly deleterious (with a heterozygote selection coefficient hs = 5%).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-v3.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Effect of the choice of prior on Bayes odds of h<italic>s</italic> &gt; 0.5x10<sup>–3</sup>.</title><p>The prior on <italic>hs</italic> (left column) and Bayes odds that <italic>s</italic> &gt; 10<sup>–3</sup> given that a mutation at a site is observed at 0, 1–10, or &gt;10 copies, for various sample sizes (right column). <italic>h</italic> is fixed at 0.5 (see Materials and methods). The odds are calculated using 10,000 draws from the prior and posterior distributions. (<bold>a</bold>) <italic>N<sub>e</sub>s</italic> ~ Gamma(shape = 0.23, scale = 425/0.23), with <italic>N</italic><sub>e</sub> = 10,000, the parameters inferred in <xref ref-type="bibr" rid="bib15">Eyre-Walker et al., 2006</xref>. (<bold>b</bold>) log(<italic>s</italic>)~N(–6,2) (<bold>c</bold>) <italic>s</italic>~Beta(alpha = 0.001,beta = 0.1). Values below 10<sup>–10</sup> are binned as 10<sup>–10</sup>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-figsupp1-v3.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Odds of non-synonymous variants having been classified as pathogenic in ClinVar and DDD if they occur at sites that are either invariant (0 copies) or segregating ( &gt; 0 copies) in a sample of 780 K chromosomes.</title><p>The odds for invariant sites are calculated as the ratio of p(pathogenic | 0 copies) / p(benign | 0 copies) and p(pathogenic)/p(benign) (see Methods). In DDD, variants that fall in 380 ‘consensus’ genes (<xref ref-type="bibr" rid="bib28">Kaplanis et al., 2020</xref>), for which there is strong evidence of being causal for developmental disorders are considered ‘pathogenic’, and variants in all other genes ‘benign’. In ClinVar, variants classified as ‘likely pathogenic’ are assumed to be pathogenic; these are compared to two sets of benign variants, one limited to variants classified in ClinVar as ‘likely benign’, and the other inclusive of variants for which the evidence is uncertain or inconclusive. Note that since both ClinVar classifications and the identification of consensus genes in DDD rely in part on whether a site is segregating in datasets like ExAC, the degree of enrichment is hard to interpret.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-figsupp2-v3.tif"/></fig><fig id="fig4s3" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 3.</label><caption><title>For various sample sizes, prior and posterior log densities for <italic>hs</italic>, and the Bayes odds of <italic>s</italic> &gt; 10<sup>–3</sup> (and <italic>h</italic> = 0.5) for a mutation observed at 0, 1–10, or &gt;10 copies.</title><p>The prior distribution of <italic>s</italic> is log-uniform over [10<sup>–7</sup>,1]. The odds are calculated from 15,000 draws from the prior and posterior distributions. (<bold>a</bold>) At a site with mutation rate ~10<sup>–9</sup> (<bold>b</bold>) At a site with mutation rate ~10<sup>–8</sup>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-figsupp3-v3.tif"/></fig><fig id="fig4s4" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 4.</label><caption><title>Comparison of measures of deleteriousness at 1.1 million mutational opportunities for methylated CpG (mCpG) transitions vs. 90 million other mutational opportunities in exons.</title><p>(<bold>a</bold>) Fraction of mutational opportunities for methylated-CpG transitions vs. all other mutational opportunities in exons by their putative functional effect. The difference is statistically significant for missense, regulatory, and synonymous categories (Fisher exact test p-value &lt;&lt; 10<sup>–5</sup>) but not for the LOF class (p-value = 0.06). (<bold>b</bold>) Cumulative distribution of the B-statistic from <xref ref-type="bibr" rid="bib38">McVicker et al., 2009</xref> for methylated CpG sites vs. all other types of sites in exons (Kolmogorov-Smirnov test p-value &lt;&lt; 10<sup>–5</sup>). (<bold>c</bold>) Distribution of CADD scores at mutational opportunities for methylated-CpG transitions vs. all other mutational opportunities in exons (p-value from a Kolmogorov-Smirnov test &lt;&lt;10<sup>–5</sup>); some skew towards lower values may be expected from the behavior of CADD scores in the presence of mutation rate variation. Despite these significant differences, these statistics are overall pretty similar for methylated CpG sites and other mutation types.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-figsupp4-v3.tif"/></fig><fig id="fig4s5" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 5.</label><caption><title>Estimating the DFE of LOF mutations using posterior densities of <italic>hs</italic> for invariant and segregating mCpG sites.</title><p>(<bold>a</bold>) Posterior log densities of <italic>hs</italic> for a C&gt;T mutation at a methylated CpG site that is observed at 0 or &gt;0 copies, in sample sizes of 15K and 780K, given a log-uniform prior on s and h=0.5 (see Methods). (<bold>b</bold>) The DFEs estimated by weighting the posterior densities in (<bold>a</bold>) by the fraction of LOF mCpG sites that are segregating (73% at 780K; 9% at 15K) and invariant (27% and 91% respectively). (<bold>c</bold>) Posterior log densities of <italic>hs</italic> for a C&gt;T mutation at a methylated CpG site that is observed at 0 or &gt;0 copies, in sample sizes of 15k and 780k, given a gamma prior with parameters inferred in <xref ref-type="bibr" rid="bib15">Eyre-Walker et al., 2006</xref> (see Methods). (<bold>d</bold>) The DFEs estimated by weighting the posterior densities in (<bold>c</bold>) by the fraction of LOF mCpG sites that are segregating (73% at 780K; 9% at 15K) and invariant (27% and 91% respectively). For the sample size of 15K, the posterior distribution recapitulates the prior, because there is little information about selection in whether a site is observed to be segregating or invariant, and particularly about strong selection. In the sample of 780K, there is more information about selection in a site being invariant and therefore, there is a shift towards stronger selection coefficients regardless of the prior.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-fig4-figsupp5-v3.tif"/></fig></fig-group><p>Because Bayes odds provide a natural way to summarize the strength of statistical evidence that comes from the observation at a single site, we consider the Bayes odds that a mutation is subject to hs &gt; 0.5 x 10<sup>–3</sup>, i.e., is under strong selection (see Materials and methods). In small samples, in which most sites are monomorphic, being monomorphic is consistent with both neutrality and very strong selection (<xref ref-type="fig" rid="fig4">Figure 4a</xref>) and the Bayes odds are close to 1, reflecting the fact that the observation barely shifts our prior assumptions (<xref ref-type="fig" rid="fig4">Figure 4b</xref>). In contrast, with larger sample sizes, in which putatively neutral CpG sites reach saturation, the posterior distribution for invariant sites is highly peaked–what is not segregating is likely strongly deleterious–and accordingly the Bayes odds become substantially greater than 1. Notably, at current sample sizes of 390K individuals, there is still some dependence on the prior (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>), but the Bayes odds of hs &gt; 0.5 x 10<sup>–3</sup> at an invariant methylated CpG are large. Given our choice of prior, the odds are 15:1 (<xref ref-type="fig" rid="fig4">Figure 4b</xref>), which suggests that most (~15/16) of the ~27 % of LOF mutations and ~6 % of missense mutations not seen in current samples are subject to this degree of selection.</p><p>While the relationship of selection strengths to clinical pathogenicity is not straight-forward, selection coefficients on that order are likely to be of relevance to determinations of pathogenicity in clinical settings (<xref ref-type="bibr" rid="bib9">Cassa et al., 2017</xref>; <xref ref-type="bibr" rid="bib28">Kaplanis et al., 2020</xref>). Indeed, mutations with hs &gt; 0.5 × 10<sup>–3</sup> may be highly deleterious to some individuals that carry them, enough to produce clinically visible effects, but vary substantially in their penetrance. Accordingly, mutations classified as pathogenic in ClinVar (<xref ref-type="bibr" rid="bib34">Landrum et al., 2018</xref>) or identified as underlying severe developmental disabilities in the Deciphering Developmental Disorders (DDD) cohort (<xref ref-type="bibr" rid="bib28">Kaplanis et al., 2020</xref>) are 6-fold to 51-fold enriched at sites invariant in 390K individuals compared to those classified as benign (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>). This analysis comes with important caveats–notably that the classifications of pathogenicity rely in part on the presence or absence of mutations in reference databases–but it suggests an enrichment on par with estimated Bayes odds of strong selection.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><sec id="s3-1"><title>Interpreting polymorphic sites in current reference databases</title><p>In a sufficiently large sample, even a segregating site can be subject to strong selection (<xref ref-type="fig" rid="fig4">Figure 4a and b</xref>). For instance, in current exome sample sizes, a C &gt; T mutation at a methylated CpG site with hs = 0.5 × 10<sup>–3</sup> is almost always observed segregating (<xref ref-type="fig" rid="fig4">Figure 4c</xref>). This follows from the expectation under mutation-selection-drift balance (<xref ref-type="bibr" rid="bib20">Gillespie, 1998</xref>): in a constant population size, a mutation that arises at rate 1.2 × 10<sup>–7</sup> per generation and is removed by selection at rate <italic>hs</italic> = 0.05% per generation has an expected population frequency of 2.4 × 10<sup>–4</sup>; in a sample of 780K, the mean number of copies is 187. Even with substantial variation due to genetic drift and sampling error, such a site should almost always be segregating at that sample size. In fact, even a mutation with <italic>hs</italic> of 5 % would quite often be observed. Thus, segregating sites in large samples are a mixture of neutral, weakly selected and strongly selected sites. An implication is that, although large reference repositories such as gnomAD were partly motivated by the possibility of excluding deleterious variants, as samples grow in size, it cannot simply be assumed that clinically relevant variants are absent from reference datasets. In principle, the only mutations never seen as samples grow in size would be the ones that are embryonically lethal.</p><p>More generally, any interpretation of variants of unknown function by reference to repositories such as gnomAD or disease cohorts enriched for deleterious variation (<xref ref-type="bibr" rid="bib59">Taliun et al., 2021</xref>), whether the goal is to exclude benign variants or identify likely pathogenic ones, is implicitly reliant on assumptions that change with sample size and dramatically differ by mutation type. At current sample sizes, invariant methylated CpGs are likely highly deleterious; for less mutable types, the information content at invariant sites is very limited at even the largest sample sizes considered (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>). Similarly, learning about the fitness consequences of segregating mutations from their observed frequencies is contingent on assumptions about the mutation rate, and the demographic history of the sample.</p></sec><sec id="s3-2"><title>The distribution of fitness effects in human genes</title><p>As we show, a typical site in the genome, with a mutation rate of 10<sup>–8</sup> per generation, does not provide much information about selection (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>), because the average length of the genealogy is likely substantially less than 10<sup>8</sup> generations. One exception, which is a special case, is gene loss: each gene can be conceived of as a single locus at which many possible LOF mutations have the same fitness impact (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib9">Cassa et al., 2017</xref>; <xref ref-type="bibr" rid="bib17">Fuller et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Weghorn et al., 2019</xref>; Agarwal, Fuller, Przeworski, in prep.). The mutation rate to LOF, calculated by summing rates of individual LOF mutations, is ~10<sup>–6</sup> per gene per generation on average (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>), such that in the absence of selection, many LOF mutations are expected in most genes. At this special subset of sites, the distribution of fitness effects can be inferred by binning loss-of-function variants within genes (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>; <xref ref-type="bibr" rid="bib9">Cassa et al., 2017</xref>; <xref ref-type="bibr" rid="bib62">Weghorn et al., 2019</xref>; Agarwal, Fuller, Przeworski, in prep.).</p><p>An analogous strategy to overcome sample size limitations at other types of sites is to infer selection in bins of sites (<xref ref-type="bibr" rid="bib13">Dukler et al., 2021</xref>); however, if sites within a bin vary in their fitness effects, inferences based on these bins are not straight-forward. Indeed, the mutation frequency in a bin reflects the harmonic mean of <italic>hs</italic> across sites in the bin weighted by (unknown) mutation rates across sites (see Materials and methods).</p><p>Given these limitations, individual methylated CpG sites can provide a useful point of entry to understanding the DFE in humans. Although methylated CpG sites appear under somewhat less constraint than other sites, the differences are subtle (<xref ref-type="fig" rid="fig4s4">Figure 4—figure supplement 4</xref>), and what we learn at these sites can tell us what to expect more generally. As a first step, we can obtain a DFE across non-synonymous mCpG sites by weighting the densities for segregating and invariant sites (<xref ref-type="fig" rid="fig4">Figure 4a</xref>) by the proportion of sites in each category (an example for possible LOF mutations is shown in <xref ref-type="fig" rid="fig4s5">Figure 4—figure supplement 5</xref>, for sample sizes of 15K and 780K) . In current samples, the posterior odds for invariant methylated CpGs having hs ≥ 0.5 × 10<sup>–3</sup> are 92 % under our model, whereas they are 37 % for segregating methylated CpGs. Considering possible LOF mutations at methylated CpGs, of which 27 % are not observed in current samples, these odds imply that the fraction of de novo LOF mutations with hs ≥ 0.5 × 10<sup>–3</sup> is roughly 52 % ( = 0.27 × 0.92 + 0.73 × 0.37).</p><p>We can use a similar approach to estimate the minimum fraction of de novo mutations that lead to a deleterious non-synonymous change. For missense sites, given the same uninformative prior on <italic>hs</italic> as for LOF mutational opportunities, the fraction estimated to be highly deleterious is 40 % ( = 0.05 × 0.92 + 0.95 × 0.37). Since ~0.97 % of all de novo point mutations are missense and ~0.07 % lead to a LOF (see Methods), these estimates translate into roughly a 1 in 236 chance ( = 40%x0.97% + 52% x0.07%) that a de novo mutation has an effect <italic>hs</italic> ≥0.5 × 10<sup>–3</sup>. Assuming, finally, that each individual inherits 70 new mutations (<xref ref-type="bibr" rid="bib33">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib27">Jónsson et al., 2017</xref>), these estimates imply that one out of every 3.4 individuals is born with a new and potentially highly deleterious, non-synonymous mutation. This calculation is based on only two frequency categories, however, discarding the information contained in allele frequencies at segregating sites, and only point mutations are taken into account. Thus, the true fraction is likely substantially higher.</p></sec><sec id="s3-3"><title>Outlook</title><p>Moving forward, we should soon have substantial information not only about the DFE but the strength of selection at individual CpG sites (<xref ref-type="fig" rid="fig4">Figure 4</xref>). Inferences based on them, or indeed any sites, will need to rely on an accurate demographic model, particularly for the recent past of most relevance for deleterious mutations; this problem should be surmountable, given the tremendous recent progress in our reconstruction of population structure and changes in humans (<xref ref-type="bibr" rid="bib50">Schiffels and Durbin, 2014</xref>; <xref ref-type="bibr" rid="bib31">Kelleher et al., 2019</xref>; <xref ref-type="bibr" rid="bib55">Speidel et al., 2019</xref>). Inferences will also require a good characterization of mutation rate variation across CpG sites, as is emerging from human pedigree studies and other sources (<xref ref-type="bibr" rid="bib27">Jónsson et al., 2017</xref>; <xref ref-type="bibr" rid="bib44">Poulos et al., 2017</xref>; <xref ref-type="bibr" rid="bib61">Vöhringer et al., 2020</xref>; <xref ref-type="bibr" rid="bib51">Seplyarskiy and Sunyaev, 2021</xref>), and careful consideration of the effects of multiple hits (<xref ref-type="bibr" rid="bib23">Harpak et al., 2016</xref>) and biased gene conversion (<xref ref-type="bibr" rid="bib21">Glémin et al., 2015</xref>). It will also be of interest to revisit the standard assumptions that go into inferring a DFE, including that all mutations are at least partially dominant in their fitness effects; that the DFE remains fixed even as the effective population size changes by orders of magnitude; and that the distribution is bounded above at 0, when some of the mutations segregating in large samples are likely to be weakly beneficial. Putting these elements together, robust inference of the fitness effects of mutations in human genes should finally be within reach, through the lens of CpG sites.</p></sec></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Processing de novo mutation data</title><p>We focused on ~190,000 published de novo mutations in a sample of 2976 parent-offspring trios that were whole genome sequenced (<xref ref-type="bibr" rid="bib22">Halldorsson et al., 2019</xref>). To date, this is the largest publicly available set of trios that, to our knowledge, have not been sampled on the basis of a disease phenotype. Unless otherwise specified, we used these DNMs to calculate mutation rates, as described in later sections. We converted hg38 coordinates to hg19 coordinates using UCSC Liftover. We excluded indels, and all DNMs that occur outside the ~2.8 billion sites covered by gnomAD v2.1.1 whole genome sequences. We obtained the immediately adjacent 5’ and 3’ bases at each position from the hg19 reference genome, so that we had each de novo mutation within its trinucleotide context; we used this information to identify CpG sites. Where such data were available (for 89 % of CpG de novo mutations), we also annotated each site with its methylation status measured by bisulfite sequencing in testis sperm cells and ovaries (see <xref ref-type="table" rid="app1table1">Appendix 1—table 1</xref>).</p><p>We annotated DNMs with their variant consequences using Variant effect predictor (v87, Gencode V19) and the hg19 LOFTEE tool (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>) to flag high-confidence (‘HC’) loss-of-function variants. We obtained the fraction of DNMs in the genome that occured at sites annotated as missense or LOF in the ‘canonical’ protein-coding transcript for each gene provided by Gencode.</p></sec><sec id="s4-2"><title>Processing polymorphism data</title><p>We downloaded publicly available polymorphism data from gnomAD (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>), the UK Biobank (<xref ref-type="bibr" rid="bib58">Szustakowski, 2020</xref>), the DiscovEHR collaboration between the Regeneron Genetics Center and Geisinger Health System (<xref ref-type="bibr" rid="bib12">Dewey et al., 2016</xref>), and 1000 Genomes Phase 3 (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>). Where needed, we lifted over coordinates to the hg19 reference assembly using the UCSC LiftOver tool. Salient characteristics of these samples are as follows:</p><table-wrap id="inlinetable1" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Dataset</th><th align="left" valign="bottom">RegionsSequenced</th><th align="left" valign="bottom">Individuals</th><th align="left" valign="bottom">Variants</th><th align="left" valign="bottom">Populations sampled</th><th align="left" valign="bottom">Original alignment</th></tr></thead><tbody><tr><td align="left" valign="bottom">1000 genomes Phase 3(also included in gnomAD)</td><td align="left" valign="bottom">Genomes</td><td align="char" char="." valign="bottom">2504</td><td align="char" char="." valign="bottom">84 million</td><td align="left" valign="bottom">mixture</td><td align="left" valign="bottom">hg19-b37</td></tr><tr><td align="left" valign="bottom">gnomAD v2.1.1</td><td align="left" valign="bottom">Exomes</td><td align="char" char="." valign="bottom">125,748</td><td align="char" char="." valign="bottom">15 million</td><td align="left" valign="bottom">mixture</td><td align="left" valign="bottom">hg19</td></tr><tr><td align="left" valign="bottom">gnomAD v2.1.1</td><td align="left" valign="bottom">Genomes</td><td align="char" char="." valign="bottom">15,708</td><td align="char" char="." valign="bottom">230 million</td><td align="left" valign="bottom">mixture</td><td align="left" valign="bottom">hg19</td></tr><tr><td align="left" valign="bottom">UK Biobank</td><td align="left" valign="bottom">Exomes</td><td align="char" char="." valign="bottom">199,932</td><td align="char" char="." valign="bottom">16 million</td><td align="left" valign="bottom">~93 % European ancestry</td><td align="left" valign="bottom">hg38</td></tr><tr><td align="left" valign="bottom">DiscovEHR</td><td align="left" valign="bottom">Exomes</td><td align="char" char="." valign="bottom">50,726</td><td align="char" char="." valign="bottom">8 million</td><td align="left" valign="bottom">~98 % European ancestry</td><td align="left" valign="bottom">hg19-b37</td></tr></tbody></table></table-wrap><p>For the gnomAD data, we obtained the allele frequency for each variant in the full exome and genome samples, as well as their Non-Finnish European (‘NFE’) subsets from the VCF files (in hg19 coordinates) provided. For each sample, we obtained the set of segregating sites (i.e. the set of variants that pass gnomAD quality filters and have an allele frequency &gt;0 in the sample). For the 1000 Genomes Phase-3 data, we obtained the set of variant positions similarly. Note that the 1000 Genomes samples are also contained within the gnomAD sample. For the DiscovEHR sample, allele frequencies are available where MAF &gt;0.001 (and set equal to 0.001 for lower values &gt; 0); we can thus determine the set of sites segregating in this sample, but we do not have access to any other information about individual variants.</p><p>For the UK Biobank exome sequencing data, additional processing was required. We downloaded the population-level plink files with exome-wide genotype information for ~200,000 individuals. We excluded exome samples that did not pass variant or sample quality control criteria in the previously released genotyping array data. Specifically, we excluded samples that have a discrepancy between reported sex and inferred sex from genotype data, a large number of close relatives in the database, or are outliers based on heterozygosity and missing rate, as detailed in <xref ref-type="bibr" rid="bib8">Bycroft et al., 2018</xref>. Finally, we excluded individuals who withdrew from the UK Biobank by the end of 2020. This left us with 199,932 exome samples that overlap with the high-quality subset of the genotyped samples. We additionally limited our analysis to the list of ~39 million exonic sites with an average of 20X sequence coverage provided by UK Biobank (<xref ref-type="bibr" rid="bib58">Szustakowski, 2020</xref>). We transformed the processed plink files into the standard variant call format, polarized variants to the hg38 reference assembly, and obtained the frequency of the non-reference allele in the sample. We then lifted over the coordinates from hg38 to hg19 using the UCSC LiftOver tool. We excluded the few positions where the reference alleles were mismatched or swapped between the two assemblies.</p><p>All but 12 % of segregating mCpG transitions were shared across at least two non-overlapping datasets. Of segregating variants seen in one of the gnomAD or UK Biobank datasets, all but two variants had at least 5 % of individuals (and typically on the order of ~100 K) sequenced at that position. Thus, we think it highly unlikely that we misclassified invariant sites as segregating, or vice versa. For ~9000 variants that are seen only once in the GHS data, we unfortunately did not have access to variant quality metrics. Excluding these sites only very slightly affects our results and does not change any qualitative conclusions.</p></sec><sec id="s4-3"><title>Identifying and annotating mutational opportunities in the exome</title><p>For all possible mutational opportunities in sequenced exons, we collated a variety of functional annotations. To this end, we first generated a list of all possible SNV mutational opportunities in the exome. We obtained the list of sites that fall in exons or within 50 base pairs (bp) of exons in Gencode v19 genes and that are among the ~2.8 billion sites covered by gnomAD v2.1.1 whole genome sequences. For each position, we extracted the reference allele from the hg19 assembly and generated the three possible single-nucleotide derived alleles. We also obtained the immediately adjacent 5’ and 3’ bases at each position from the hg19 reference genome, so that we had each mutational opportunity within its trinucleotide context; we used this information to identify CpG sites. Where such data were available, we also annotated each site with its methylation status in testis sperm cells and ovaries.</p><p>To identify sites at which variants or de novo mutations could be confidently assayed by whole-exome sequencing methods, we obtained regions targeted in whole exome sequencing from gnomAD and the UK Biobank. We limited our analysis to sites that were covered at 20X or more in the exome sequencing subsets of both gnomAD and UK Biobank (that lifted over correctly to the hg19 assembly), which we refer to as ‘accessible sites’.</p><p>We then annotated the ~90 million mutational opportunities (at 30 million sites) with CADD scores and variant consequences using Variant effect predictor (v87, Gencode V19) and the hg19 LOFTEE tool (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>) to flag high-confidence (‘HC’) loss-of-function variants. For loss-of-function variants, we also noted their location in the gene by exon number (e.g. in exon 10 of 12 exons in the gene). We used a published database of curated protein features derived from Refseq proteins (<xref ref-type="bibr" rid="bib57">Stanek et al., 2020</xref>) to annotate all sites in protein coding genes that were associated with a particular type of functional activity (detailed functional annotations were available for about 60,000 of 1.1 million methylated CpG sites). At each site, we used either the primary ‘site-type’ annotation, or when that was missing or listed as ‘other’, we extracted the annotation from the more detailed ‘notes’ field where this information was provided.</p><p>Because there are multiple transcripts for each variant, we limited our analysis to the ‘canonical’ protein-coding transcript for each gene provided by Gencode to obtain a single annotation for each variant. For 10–20% of variants, this approach still yielded multiple possible consequences per variant, for instance, where there are multiple canonical transcripts due to overlapping genes. For these cases, we assigned one of the ‘canonical’ transcripts to the variant at random, to avoid making assumptions about their relative importance. Further overlaps within the same gene, for example, a missense variant that is also a splice variant in the same transcript, or a DNA-binding site that also undergoes a particular post-translational modification, were resolved in the same manner.</p><p>As an alternative approach, we obtained the worst consequence in all protein-coding transcripts for each variant, using the ranks of variant consequences by severity provided by Ensembl (<xref ref-type="table" rid="app1table1">Appendix 1—table 1</xref>). In the absence of systematic ranking criteria for the protein function annotations, we used the following order: sites that were designated as having catalytic activity (‘active’ sites) were given highest priority in overlaps, followed by DNA-binding sites, followed by other types of binding (to metal, polypeptides, ions), and finally by sites that are known to undergo post-translational or other regulatory modifications, and trans-membrane sites. Thus, a transmembrane site with regulatory activity is classified as a regulatory site, while a regulatory site with DNA-binding activity is classified as DNA-binding. Using these alternate criteria to group sites does not affect our conclusions (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>).</p><p>All sources of annotation data are listed in <xref ref-type="table" rid="app1table1">Appendix 1—table 1</xref>. A list of CpG sites and annotations is provided as additional data.</p></sec><sec id="s4-4"><title>Comparing fitness effects across sets of mutational opportunities</title><p>To assess whether the set of 1.1 million C &gt; T mutational opportunities at methylated CpG sites are systematically different from the other ~90 million exonic mutational opportunities in their potential fitness effects, we compared the distribution of CADD scores in the two groups using a Kolmogorov-Smirnov test. We note that this comparison is likely to be somewhat confounded by differences in mutation rates, given our finding that CADD scores do not perfectly isolate the effects of selection from those of variability in mutation rates (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5c</xref>). Since the mutation rate for methylated CpG sites is higher than for other types, this may lead them to appear somewhat less constrained than they actually are.</p><p>We further compared the fraction of C &gt; T mutational opportunities at methylated CpGs in an annotation class vs. the fraction of other mutational opportunities in that class. We used a Fisher exact test (with a Bonferroni correction for four tests) to determine whether the two sets of mutational opportunities were differently distributed across synonymous, missense, regulatory, and LOF variant classes.</p></sec><sec id="s4-5"><title>Obtaining mean de novo mutation rates by mutation type and annotation</title><p>We counted the total number of de novo mutations in sequenced exons (~91 million mutational opportunities) for eight classes of mutations: two transitions and a transversion each at C and T sites, transitions at CpG sites with relatively low levels of methylation (defined here as &lt;65%), and transitions at CpG sites with high levels of methylation ( ≥ 65%). To obtain the mutation rate per site per generation, we divided the counts by the haploid sample size (2 × 2976 individuals; see section 1) and the number of mutational opportunities of each type. We report 95 % confidence intervals assuming a Poisson distribution for mutation counts. The rates obtained (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>) are similar to previous ones (<xref ref-type="bibr" rid="bib33">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib27">Jónsson et al., 2017</xref>; <xref ref-type="bibr" rid="bib18">Gao et al., 2019</xref>) and roughly consistent with the rates predicted by the gnomAD mutation model (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>).</p><p>To evaluate the impact of methylation status on the mutation rate at CpG sites, we obtained the mean mutation rate for C &gt; T mutations at CpG sites in each methylation bin as described above, separately for methylation levels in ovaries and testes. While there is a limited amount of data, especially for some low-methylation bins, our choice of cutoff for ‘methylated’ seems sensible (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>).</p><p>We then calculated the mean mutation rate for methylated CpG transitions, for different compartments in the genome, namely in (a) exons and non-exons, (b) four variant consequence categories: synonymous, missense, regulatory, and LOF variants, (c) CADD score deciles, and (d) in exons that constitute the first half vs the second half of genes. We also calculated the mean mutation rate for methylated CpG transitions in four trinucleotide contexts (ACG, CCG, GCG, and TCG). In each case, we obtained the total number of de novo mutations and the Poisson 95 % confidence interval around mutation counts in each group, and divided by the number of mutational opportunities in the group. We tested if the proportion of methylated CpG sites with de novo C &gt; T mutations in each non-synonymous compartment was different from the proportion of synonymous methylated CpGs with de novo C &gt; T mutations, accounting for multiple tests.</p></sec><sec id="s4-6"><title>Variance in mutation rate at methylated CpGs</title><p>Although current samples of DNM data are large enough to compare the mean mutation rate at methylated CpGs across the annotation classes examined here, there is not enough data to directly compare variances in mutation rates. To learn how much broad scale features beyond methylation and the immediate trinucleotide context shape variation in mutation rates at methylated CpGs, we therefore relied on a broader set of regions for example those that fall inside and outside exons. Exonic and non-exonic regions differ considerably in epigenetic features and replication timing (<xref ref-type="bibr" rid="bib56">Stamatoyannopoulos et al., 2009</xref>); yet, there is no discernable difference in average de novo mutation rates at methylated CpGs inside and outside sequenced exons (FET p-value = 0.10, <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4a</xref>). We also compared the number of double and single de novo hits in exons and non-exons using a Fisher exact test (p-value = 0.35, <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4b</xref>). Since the number of double hits reflects the variance in mutation rates across sites, these results lend some support to there being limited variation due to broad scale genomic features in transition rates at methylated CpGs.</p></sec><sec id="s4-7"><title>Calculating the fraction of sites segregating by annotation</title><p>For each methylated CpG site in the exome, there are three mutational opportunities (C &gt; A, C &gt; G, C &gt; T); we focused only on the opportunities for C &gt; T mutations. For each methylated CpG site then, we noted whether or not it was segregating, or in other words if there was a C &gt; T variant in samples of individuals from gnomAD (<xref ref-type="bibr" rid="bib29">Karczewski et al., 2020</xref>), the UK Biobank (<xref ref-type="bibr" rid="bib58">Szustakowski, 2020</xref>), the DiscovEHR dataset (<xref ref-type="bibr" rid="bib12">Dewey et al., 2016</xref>), and 1000 Genomes Phase 3 (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>), processed as described above, or a combined sample of 390 K non-overlapping individuals.</p><p>Within the set of methylated CpG sites where C &gt; T mutations are synonymous, we calculated the fraction segregating in each sample of interest. Similarly, for different subsets of methylated CpGs, namely those in (a) four variant consequence categories: synonymous, missense, regulatory, and LOF variants, (c) CADD score deciles, (d) functional site categories (e.g. trans-membrane vs catalytic sites in proteins), and (e) the first half vs the second half of genes, we calculated the fraction segregating in the combined sample of 390 K individuals. We rescaled the fraction of sites segregating in each annotation by the fraction of synonymous sites segregating in the sample.</p><p>We verified that the differences in the fraction of sites segregating across annotations are not due to variable impacts of linked selection across annotations. To do so, we calculated the fraction of sites segregating with sites in different annotations matched for B-statistics <xref ref-type="bibr" rid="bib38">McVicker et al., 2009</xref>; we obtained very similar results with this approach (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>).</p><p>We assumed that conditional on the number of mutational opportunities and a fixed probability of segregating for each site in a compartment, the number of sites segregating is binomially distributed, and obtained 95 % confidence intervals on that basis. We tested if the proportion of sites segregating in each compartment is different from the proportion segregating at putatively neutral (here, synonymous) sites using a Fisher exact test, accounting for multiple tests.</p><p>We also calculated the fraction of other types of synonymous sites segregating in each sample size of interest (specifically, for T &gt; A variants, and C &gt; Ts not at methylated CpG sites).</p></sec><sec id="s4-8"><title>Frequency of mutant alleles in bins of <italic>K</italic> sites</title><p>Within each annotation of interest, with an average mutation rate of <italic>u</italic> per site, we construct bins of <italic>k</italic> sites, such that <italic>k = U/u,</italic> where <italic>U</italic> is the mean mutation rate of a transition at methylated CpG site in that annotation class. The mean mutation rates are calculated for each mutation type within each annotation, as described in Section five above. We then count the fraction of bins in which no such mutations are observed. As an example, for T &gt; A mutations, <italic>k</italic> is on the order of 100 (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2a</xref>).</p><p>Since each bin can be treated as being comparable to a single neutral methylated CpG site, bins that contain only neutral sites are expected to contain at least one mutation in 99 % of bins; this is indeed the case for bins of synonymous sites (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2b</xref>).</p><p>When considering sites that contain a mixture of neutral and selected sites, bins of <italic>k</italic> sites are no longer as readily comparable to methylated CpG sites, however (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2c</xref>). If sites within a bin are under varying degrees of selection, then the mutation count reflects the harmonic mean of the strength of selection acting on individual sites. Specifically, under a deterministic model of mutation-selection balance, if q<sub>i</sub> is the allele frequency at the <italic>i<sup>th</sup></italic> site in a bin of <italic>k</italic> sites:<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>then<disp-formula id="equ2"><mml:math id="m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mi>h</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Assuming <italic>u<sub>i</sub></italic> = <italic>u</italic> = U/k,<disp-formula id="equ3"><mml:math id="m3"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>q</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>U</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mi>U</mml:mi><mml:mi>k</mml:mi></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>h</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>that is, <italic>q<sub>bin</sub></italic> is a function of the harmonic mean of <italic>hs</italic> at the <italic>k</italic> sites.</p></sec><sec id="s4-9"><title>Forward simulations</title><p>We used a forward simulation framework initially described in <xref ref-type="bibr" rid="bib53">Simons et al., 2014</xref>, and modified in <xref ref-type="bibr" rid="bib17">Fuller et al., 2019</xref>. Briefly, we modeled evolution at a single non-recombining bi-allelic site, which undergoes mutations each generation at rate 2<italic>N<sub>e</sub>u</italic> in a panmictic diploid population of effective population size <italic>N<sub>e</sub></italic>. Each generation is formed by Wright-Fisher sampling with selection, where fitness is reduced by <italic>hs</italic> in heterozygotes and <italic>s</italic> in homozygotes for the T allele. We fixed the dominance coefficient <italic>h</italic> as 0.5, and we chose one value of the selection coefficient <italic>s</italic> from a log-uniform prior ranging from 10<sup>–7</sup> to 1 for each simulation (for autosomal mutations with fitness effects in heterozygotes, we only need to specify the compound parameter <italic>hs</italic>; reviewed in <xref ref-type="bibr" rid="bib17">Fuller et al., 2019</xref>). Given a mutation rate and a demographic model that specifies <italic>N<sub>e</sub></italic> in each generation, we simulated the evolution of a site forward in time to determine whether the site is segregating in a sample of size <italic>n</italic> at present.</p><p>We used <italic>u</italic> = 1.2 x 10<sup>–7</sup> per site per generation to model CpG&gt; TpG mutation at a methylated CpG site. The simulation framework allows for recurrent mutations, which are expected to arise often at this mutation rate. We also allowed for TpG&gt; CpG back mutations at the rate of 5 × 10<sup>–9</sup> (calculated from de novo mutation data, as CpG&gt; TpG mutations). To model T &gt; A mutations, we used <italic>u</italic> = 1.2 x 10<sup>–9</sup> per site per generation, with a back mutation rate of 1.2 × 10<sup>–9</sup> per site per generation; for C &gt; T mutations not at methylated CpG sites, we used <italic>u</italic> = 0.9 x 10<sup>–8</sup> per site per generation, with a back mutation rate of 5 × 10<sup>–9</sup> per site per generation (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). We note that, since the mutation rate increases with paternal and maternal ages, an implicit assumption is that the distribution of parental ages in the trio data is representative of the parental ages over the evolutionary history of exome samples.</p><p>For the demographic model, we relied on the Schiffels-Durbin model for population size changes in Europe over the past ~55,000 generations, preceded by a ~ 10 <italic>N<sub>e</sub></italic> generation burn-in period of neutral evolution at an initial population size <italic>N<sub>e</sub></italic> of 14,448 following ref (<xref ref-type="bibr" rid="bib53">Simons et al., 2014</xref>). In the last generation, that is at present, we sampled <italic>n</italic> individuals from the simulated population, to match the size of the sample of interest.</p><p>We calculated the probability that a site with the fixed mutation rate <italic>u</italic> is segregating for a given value of <italic>hs</italic> (with <italic>s</italic> = 0 under neutrality) as the proportion of simulations with those parameters in which the site is segregating for different sample sizes at present.</p><p>In comparing the output of these simulations to data, we considered two scenarios where we may either undercount or overcount segregating CpG sites in the data relative to the simulations. First, because we conditioned on the human reference allele being a CpG in data, we did not count sites where the CpG is the ancestral but not the reference allele. To check how often this is expected to occur, we mimicked this scenario in simulations, sampling a single chromosome at the end of the simulation as the mock haploid reference genome. The proportion of simulations in which CpG is the ancestral but not the reference allele is ~0.1%, that is, approximately the heterozygosity levels in humans. The second case is that for a subset of the CpG&gt; TpG variants observed at present, the CpG mutation is the reference allele but is not ancestral. To mimic this scenario in our simulations, we simulated a site that starts as TpG (with a mutation rate of 5 × 10<sup>–9</sup> to CpG, and a back mutation rate ~1.2 x 10<sup>–7</sup> to TpG) forward in time. Then, as above, we drew a single chromosome from the sample at the end of the simulation and set it as the reference. We obtained the proportion of simulations in which the C allele is the reference, starting from a TpG background. Reassuringly, this occurs in only 0.0014 % of simulations. We note that there is in principle a third scenario to consider, in which ApG or GpG sites is ancestral and a C/T polymorphism is found in the sample at present as a result of two mutations, one to T and one to C. Given the various mutation rates involved (all less than 5 × 10<sup>–9</sup>), this double mutation case will be even less likely than the one in which TpG was ancestral. These rare scenarios should not have any substantive effect on our comparison of data to simulations, particularly when we only used such comparisons to examine qualitative trends.</p></sec><sec id="s4-10"><title>Inferring the strength of selection in simulations</title><p>We proposed <italic>s</italic> from a prior distribution (with <italic>h</italic> fixed at 0.5) and inferred the posterior distribution of <italic>hs</italic> for a site with a T allele at 0 copies using a simple Approximate Bayesian Computation (ABC) approach. Specifically, we proposed <italic>s</italic> such that log<sub>10</sub>(<italic>s</italic>)~Uniform(–7,0); we simulated expected T allele counts under our model for 10 million proposals from the prior. We accepted the subset of the proposed values of <italic>s</italic> where simulations yield 0 copies of the T allele in the sample at present; this set of <italic>s</italic> values is a sample from the posterior distribution of <italic>s</italic> given that the site is monomorphic. We calculated the Bayes odds of s &gt; 10<sup>–3</sup> as the ratio of the posterior odds of s &gt; 10<sup>–3</sup> and the prior odds of s &gt; 10<sup>–3</sup>:<disp-formula id="equ4"><mml:math id="m4"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>0.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup><mml:mspace width="thinmathspace"/><mml:mo>∣</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>≤</mml:mo><mml:mn>0.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup><mml:mspace width="thinmathspace"/><mml:mo>∣</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>0.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>≤</mml:mo><mml:mn>0.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>We similarly obtained posterior distributions of <italic>hs</italic> for sites that are segregating at 0, 1–10 copies, or &gt;10 copies, in samples of different sizes, and for three different choices of priors on <italic>s</italic>, namely: s ~ Beta(<italic>α</italic> = 0.001, <italic>β</italic> = 0.1); log(s)~ N(–6,2); and N<sub>e</sub>s ~ Gamma(k = 0.23, <italic>θ</italic> = 425/0.23), with <italic>N<sub>e</sub></italic> = 10,000, based on the parameters inferred in <xref ref-type="bibr" rid="bib15">Eyre-Walker et al., 2006</xref>. These are shown in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>.</p></sec><sec id="s4-11"><title>Calculating odds of being pathogenic in ClinVar and DDD</title><p>We downloaded de novo mutation data for ~35 K individuals with developmental disorders (<xref ref-type="bibr" rid="bib28">Kaplanis et al., 2020</xref>). We also obtained a list of 380 ‘consensus’ genes from the same study; for these genes, there is evidence from multiple sources that LOF or missense mutations are causal in developmental disorders, such that they are used as part of diagnostic criteria in the clinic.</p><p>We downloaded ClinVar variants and excluded those that were not associated with at least one disease. We obtained the ‘CLNSIG’ annotation, which classifies each variant as benign or likely benign, pathogenic or likely pathogenic, or as having uncertain status or conflicting evidence.</p><p>We limited both DDD and ClinVar variants to non-synonymous C &gt; T mutations at the subset of methylated CpG sites considered. Using variants in ClinVar and DDD at sites that are invariant in our sample of 780K, we calculated the odds that an invariant site is pathogenic (vs. benign) as follows:<disp-formula id="equ5"><mml:math id="m5"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mspace width="thinmathspace"/><mml:mo>∣</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>n</mml:mi><mml:mspace width="thinmathspace"/><mml:mo>∣</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where p(pathogenic) refers to the proportion of sites classified as such, and p(benign) is defined analogously.</p><p>In DDD, we considered mutations that fall in 380 consensus genes ‘pathogenic’, and mutations in all other genes benign; thus our ‘benign’ category likely contains some genes in which mutations are in fact pathogenic. In ClinVar, variants classified as ‘pathogenic’ or ‘likely pathogenic’ are assumed to be pathogenic; these are compared to two sets of benign variants, one limited strictly to variants classified in ClinVar as ‘benign’ or ‘likely benign’, and the other inclusive of variants for which the evidence is uncertain or inconclusive. The results are shown in <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>.</p><p>We note that since both ClinVar classifications and the identification of consensus genes in DDD rely in part on whether a site is segregating in datasets like ExAC, the degree of enrichment in <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref> is hard to interpret.</p></sec><sec id="s4-12"><title>Calculating the average length of the genealogy of a sample in which methylated CpGs are saturated</title><p>Methylated CpG sites experience mutations to T at the rate of 1.17 × 10<sup>–7</sup> per generation; 99 % of such sites are segregating in a sample of 390 K individuals. Then the average length (<italic>L</italic>) of the genealogy relating the 390 K individuals can be calculated from the probability under a Poisson distribution of at least one mutation at 99 % of sites as 1-exp(–1.17 × 10<sup>–7</sup> x L) = 0.99, which gives <italic>L</italic> = 39 million generations.</p></sec><sec id="s4-13"><title>Coalescent simulations to obtain the length of genealogy of large samples</title><p>We simulated the genealogy of a sample of varying sizes using <italic>msprime</italic> (<xref ref-type="bibr" rid="bib30">Kelleher et al., 2016</xref>) under different demographic histories, modifying the standard Schiffels-Durbin model (<xref ref-type="bibr" rid="bib50">Schiffels and Durbin, 2014</xref>) as follows:</p><list list-type="alpha-lower"><list-item><p>Demographic history for a sample of Utah residents with Northern and Western European ancestry (CEU) over 55,000 generations, with a recent <italic>N<sub>e</sub></italic> of 10 million for the past 50 generations, described above.</p></list-item><list-item><p>CEU demographic history for 55,000 generations with a recent <italic>N<sub>e</sub></italic> of 100 million for the past 50 generations.</p></list-item><list-item><p>CEU demographic history for 55,000 generations with 4.5 % exponential growth for the past 196 generations, with a current <italic>N<sub>e</sub></italic> of ~100 million.</p></list-item><list-item><p>Demographic history for a sample of Yoruba sampled in Nigeria (YRI) from <xref ref-type="bibr" rid="bib50">Schiffels and Durbin, 2014</xref>, modified with a recent <italic>N<sub>e</sub></italic> of 10 million for the last 50 generations.</p></list-item><list-item><p>A structured sample from two populations that derived from an ancestral population with YRI demographic history 2000 generations ago, with YRI and CEU demographic histories respectively since, and a recent <italic>N<sub>e</sub></italic> of 10 million for the last 50 generations in each.</p></list-item></list><p>In each case, we recorded the mean genealogy length over 20 iterations.</p><p>The code for implementing these different demographic models in <italic>msprime</italic> is available on the project github repository.</p></sec></sec></body><back><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>Senior editor, <italic>eLife</italic></p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Visualization, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Investigation, Methodology, Project administration, Resources, Supervision, Writing - original draft, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-71513-transrepform1-v3.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>All source data are freely available to researchers, with sources provided in the manuscript and summarized in Appendix 1 - Table 1. Source files and code to generate the figures, and additional files containing the annotated set of CpG sites analysed in this manuscript, are available at <ext-link ext-link-type="uri" xlink:href="https://github.com/agarwal-i/cpg_saturation">https://github.com/agarwal-i/cpg_saturation</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:rev:a36cb87c0af373e81eae1935ba710c4417c46f69">https://archive.softwareheritage.org/swh:1:rev:a36cb87c0af373e81eae1935ba710c4417c46f69</ext-link>).</p><p>The following previously published datasets were used:</p><p><element-citation id="dataset1" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Karczewski</surname><given-names>KJ</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>gnomAD</data-title><source>gnomAD v2.1</source><pub-id pub-id-type="accession" xlink:href="https://gnomad.broadinstitute.org/downloads">v2.1</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Peter Andolfatto, Kelley Harris, Hakhamanesh Mostafavi, Magnus Nordborg, Itsik Pe’er, Jonathan Pritchard, Guy Sella, as well as Arbel Harpak, Zach Fuller, and other members of the Andolfatto, Przeworski and Sella labs for helpful discussions. This work was supported by NIH grants GM121372 and GM122975 to MP.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adzhubei</surname><given-names>IA</given-names></name><name><surname>Schmidt</surname><given-names>S</given-names></name><name><surname>Peshkin</surname><given-names>L</given-names></name><name><surname>Ramensky</surname><given-names>VE</given-names></name><name><surname>Gerasimova</surname><given-names>A</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Kondrashov</surname><given-names>AS</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A method and server for predicting damaging missense mutations</article-title><source>Nature Methods</source><volume>7</volume><fpage>248</fpage><lpage>249</lpage><pub-id pub-id-type="doi">10.1038/nmeth0410-248</pub-id><pub-id pub-id-type="pmid">20354512</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aggarwala</surname><given-names>V</given-names></name><name><surname>Voight</surname><given-names>BF</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>An expanded sequence context model broadly explains variability in polymorphism levels across the human genome</article-title><source>Nature Genetics</source><volume>48</volume><fpage>349</fpage><lpage>355</lpage><pub-id pub-id-type="doi">10.1038/ng.3511</pub-id><pub-id pub-id-type="pmid">26878723</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akbari</surname><given-names>P</given-names></name><name><surname>Gilani</surname><given-names>A</given-names></name><name><surname>Sosina</surname><given-names>O</given-names></name><name><surname>Kosmicki</surname><given-names>JA</given-names></name><name><surname>Khrimian</surname><given-names>L</given-names></name><name><surname>Fang</surname><given-names>YY</given-names></name><name><surname>Persaud</surname><given-names>T</given-names></name><name><surname>Garcia</surname><given-names>V</given-names></name><name><surname>Sun</surname><given-names>D</given-names></name><name><surname>Li</surname><given-names>A</given-names></name><name><surname>Mbatchou</surname><given-names>J</given-names></name><name><surname>Locke</surname><given-names>AE</given-names></name><name><surname>Benner</surname><given-names>C</given-names></name><name><surname>Verweij</surname><given-names>N</given-names></name><name><surname>Lin</surname><given-names>N</given-names></name><name><surname>Hossain</surname><given-names>S</given-names></name><name><surname>Agostinucci</surname><given-names>K</given-names></name><name><surname>Pascale</surname><given-names>JV</given-names></name><name><surname>Dirice</surname><given-names>E</given-names></name><name><surname>Dunn</surname><given-names>M</given-names></name><collab>Regeneron Genetics Center</collab><collab>DiscovEHR Collaboration</collab><name><surname>Kraus</surname><given-names>WE</given-names></name><name><surname>Shah</surname><given-names>SH</given-names></name><name><surname>Chen</surname><given-names>YDI</given-names></name><name><surname>Rotter</surname><given-names>JI</given-names></name><name><surname>Rader</surname><given-names>DJ</given-names></name><name><surname>Melander</surname><given-names>O</given-names></name><name><surname>Still</surname><given-names>CD</given-names></name><name><surname>Mirshahi</surname><given-names>T</given-names></name><name><surname>Carey</surname><given-names>DJ</given-names></name><name><surname>Berumen-Campos</surname><given-names>J</given-names></name><name><surname>Kuri-Morales</surname><given-names>P</given-names></name><name><surname>Alegre-Díaz</surname><given-names>J</given-names></name><name><surname>Torres</surname><given-names>JM</given-names></name><name><surname>Emberson</surname><given-names>JR</given-names></name><name><surname>Collins</surname><given-names>R</given-names></name><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Hawes</surname><given-names>A</given-names></name><name><surname>Jones</surname><given-names>M</given-names></name><name><surname>Zambrowicz</surname><given-names>B</given-names></name><name><surname>Murphy</surname><given-names>AJ</given-names></name><name><surname>Paulding</surname><given-names>C</given-names></name><name><surname>Coppola</surname><given-names>G</given-names></name><name><surname>Overton</surname><given-names>JD</given-names></name><name><surname>Reid</surname><given-names>JG</given-names></name><name><surname>Shuldiner</surname><given-names>AR</given-names></name><name><surname>Cantor</surname><given-names>M</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name><name><surname>Karalis</surname><given-names>K</given-names></name><name><surname>Economides</surname><given-names>AN</given-names></name><name><surname>Marchini</surname><given-names>J</given-names></name><name><surname>Yancopoulos</surname><given-names>GD</given-names></name><name><surname>Sleeman</surname><given-names>MW</given-names></name><name><surname>Altarejos</surname><given-names>J</given-names></name><name><surname>Della Gatta</surname><given-names>G</given-names></name><name><surname>Tapia-Conyer</surname><given-names>R</given-names></name><name><surname>Schwartzman</surname><given-names>ML</given-names></name><name><surname>Baras</surname><given-names>A</given-names></name><name><surname>Ferreira</surname><given-names>MAR</given-names></name><name><surname>Lotta</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Sequencing of 640,000 exomes identifies GPR75 variants associated with protection from obesity</article-title><source>Science</source><volume>373</volume><elocation-id>eabf8683</elocation-id><pub-id pub-id-type="doi">10.1126/science.abf8683</pub-id><pub-id pub-id-type="pmid">34210852</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Garrison</surname><given-names>EP</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Korbel</surname><given-names>JO</given-names></name><name><surname>Marchini</surname><given-names>JL</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><name><surname>McVean</surname><given-names>GA</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name><collab>1000 Genomes Project Consortium</collab></person-group><year iso-8601-date="2015">2015</year><article-title>A global reference for human genetic variation</article-title><source>Nature</source><volume>526</volume><fpage>68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature15393</pub-id><pub-id pub-id-type="pmid">26432245</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bhaskar</surname><given-names>A</given-names></name><name><surname>Clark</surname><given-names>AG</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Distortion of genealogical properties when the sample is very large</article-title><source>PNAS</source><volume>111</volume><fpage>2385</fpage><lpage>2390</lpage><pub-id pub-id-type="doi">10.1073/pnas.1322709111</pub-id><pub-id pub-id-type="pmid">24469801</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boffelli</surname><given-names>D</given-names></name><name><surname>McAuliffe</surname><given-names>J</given-names></name><name><surname>Ovcharenko</surname><given-names>D</given-names></name><name><surname>Lewis</surname><given-names>KD</given-names></name><name><surname>Ovcharenko</surname><given-names>I</given-names></name><name><surname>Pachter</surname><given-names>L</given-names></name><name><surname>Rubin</surname><given-names>EM</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Phylogenetic shadowing of primate sequences to find functional regions of the human genome</article-title><source>Science</source><volume>299</volume><fpage>1391</fpage><lpage>1394</lpage><pub-id pub-id-type="doi">10.1126/science.1081331</pub-id><pub-id pub-id-type="pmid">12610304</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boyko</surname><given-names>AR</given-names></name><name><surname>Williamson</surname><given-names>SH</given-names></name><name><surname>Indap</surname><given-names>AR</given-names></name><name><surname>Degenhardt</surname><given-names>JD</given-names></name><name><surname>Hernandez</surname><given-names>RD</given-names></name><name><surname>Lohmueller</surname><given-names>KE</given-names></name><name><surname>Adams</surname><given-names>MD</given-names></name><name><surname>Schmidt</surname><given-names>S</given-names></name><name><surname>Sninsky</surname><given-names>JJ</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name><name><surname>White</surname><given-names>TJ</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Clark</surname><given-names>AG</given-names></name><name><surname>Bustamante</surname><given-names>CD</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Assessing the evolutionary impact of amino acid mutations in the human genome</article-title><source>PLOS Genetics</source><volume>4</volume><elocation-id>e1000083</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000083</pub-id><pub-id pub-id-type="pmid">18516229</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bycroft</surname><given-names>C</given-names></name><name><surname>Freeman</surname><given-names>C</given-names></name><name><surname>Petkova</surname><given-names>D</given-names></name><name><surname>Band</surname><given-names>G</given-names></name><name><surname>Elliott</surname><given-names>LT</given-names></name><name><surname>Sharp</surname><given-names>K</given-names></name><name><surname>Motyer</surname><given-names>A</given-names></name><name><surname>Vukcevic</surname><given-names>D</given-names></name><name><surname>Delaneau</surname><given-names>O</given-names></name><name><surname>O’Connell</surname><given-names>J</given-names></name><name><surname>Cortes</surname><given-names>A</given-names></name><name><surname>Welsh</surname><given-names>S</given-names></name><name><surname>Young</surname><given-names>A</given-names></name><name><surname>Effingham</surname><given-names>M</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name><name><surname>Leslie</surname><given-names>S</given-names></name><name><surname>Allen</surname><given-names>N</given-names></name><name><surname>Donnelly</surname><given-names>P</given-names></name><name><surname>Marchini</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The UK Biobank resource with deep phenotyping and genomic data</article-title><source>Nature</source><volume>562</volume><fpage>203</fpage><lpage>209</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0579-z</pub-id><pub-id pub-id-type="pmid">30305743</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cassa</surname><given-names>CA</given-names></name><name><surname>Weghorn</surname><given-names>D</given-names></name><name><surname>Balick</surname><given-names>DJ</given-names></name><name><surname>Jordan</surname><given-names>DM</given-names></name><name><surname>Nusinow</surname><given-names>D</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>O’Donnell-Luria</surname><given-names>A</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Beier</surname><given-names>DR</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Estimating the selective effects of heterozygous protein-truncating variants from human exome data</article-title><source>Nature Genetics</source><volume>49</volume><fpage>806</fpage><lpage>810</lpage><pub-id pub-id-type="doi">10.1038/ng.3831</pub-id><pub-id pub-id-type="pmid">28369035</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Claussnitzer</surname><given-names>M</given-names></name><name><surname>Cho</surname><given-names>JH</given-names></name><name><surname>Collins</surname><given-names>R</given-names></name><name><surname>Cox</surname><given-names>NJ</given-names></name><name><surname>Dermitzakis</surname><given-names>ET</given-names></name><name><surname>Hurles</surname><given-names>ME</given-names></name><name><surname>Kathiresan</surname><given-names>S</given-names></name><name><surname>Kenny</surname><given-names>EE</given-names></name><name><surname>Lindgren</surname><given-names>CM</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>North</surname><given-names>KN</given-names></name><name><surname>Plon</surname><given-names>SE</given-names></name><name><surname>Rehm</surname><given-names>HL</given-names></name><name><surname>Risch</surname><given-names>N</given-names></name><name><surname>Rotimi</surname><given-names>CN</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name><name><surname>Soranzo</surname><given-names>N</given-names></name><name><surname>McCarthy</surname><given-names>MI</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A brief history of human disease genetics</article-title><source>Nature</source><volume>577</volume><fpage>179</fpage><lpage>189</lpage><pub-id pub-id-type="doi">10.1038/s41586-019-1879-7</pub-id><pub-id pub-id-type="pmid">31915397</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Stone</surname><given-names>EA</given-names></name><name><surname>Asimenos</surname><given-names>G</given-names></name><collab>NISC Comparative Sequencing Program</collab><name><surname>Green</surname><given-names>ED</given-names></name><name><surname>Batzoglou</surname><given-names>S</given-names></name><name><surname>Sidow</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Distribution and intensity of constraint in mammalian genomic sequence</article-title><source>Genome Research</source><volume>15</volume><fpage>901</fpage><lpage>913</lpage><pub-id pub-id-type="doi">10.1101/gr.3577405</pub-id><pub-id pub-id-type="pmid">15965027</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dewey</surname><given-names>FE</given-names></name><name><surname>Murray</surname><given-names>MF</given-names></name><name><surname>Overton</surname><given-names>JD</given-names></name><name><surname>Habegger</surname><given-names>L</given-names></name><name><surname>Leader</surname><given-names>JB</given-names></name><name><surname>Fetterolf</surname><given-names>SN</given-names></name><name><surname>O’Dushlaine</surname><given-names>C</given-names></name><name><surname>Van Hout</surname><given-names>CV</given-names></name><name><surname>Staples</surname><given-names>J</given-names></name><name><surname>Gonzaga-Jauregui</surname><given-names>C</given-names></name><name><surname>Metpally</surname><given-names>R</given-names></name><name><surname>Pendergrass</surname><given-names>SA</given-names></name><name><surname>Giovanni</surname><given-names>MA</given-names></name><name><surname>Kirchner</surname><given-names>HL</given-names></name><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Abul-Husn</surname><given-names>NS</given-names></name><name><surname>Hartzel</surname><given-names>DN</given-names></name><name><surname>Lavage</surname><given-names>DR</given-names></name><name><surname>Kost</surname><given-names>KA</given-names></name><name><surname>Packer</surname><given-names>JS</given-names></name><name><surname>Lopez</surname><given-names>AE</given-names></name><name><surname>Penn</surname><given-names>J</given-names></name><name><surname>Mukherjee</surname><given-names>S</given-names></name><name><surname>Gosalia</surname><given-names>N</given-names></name><name><surname>Kanagaraj</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>AH</given-names></name><name><surname>Mitnaul</surname><given-names>LJ</given-names></name><name><surname>Adams</surname><given-names>LJ</given-names></name><name><surname>Person</surname><given-names>TN</given-names></name><name><surname>Praveen</surname><given-names>K</given-names></name><name><surname>Marcketta</surname><given-names>A</given-names></name><name><surname>Lebo</surname><given-names>MS</given-names></name><name><surname>Austin-Tse</surname><given-names>CA</given-names></name><name><surname>Mason-Suares</surname><given-names>HM</given-names></name><name><surname>Bruse</surname><given-names>S</given-names></name><name><surname>Mellis</surname><given-names>S</given-names></name><name><surname>Phillips</surname><given-names>R</given-names></name><name><surname>Stahl</surname><given-names>N</given-names></name><name><surname>Murphy</surname><given-names>A</given-names></name><name><surname>Economides</surname><given-names>A</given-names></name><name><surname>Skelding</surname><given-names>KA</given-names></name><name><surname>Still</surname><given-names>CD</given-names></name><name><surname>Elmore</surname><given-names>JR</given-names></name><name><surname>Borecki</surname><given-names>IB</given-names></name><name><surname>Yancopoulos</surname><given-names>GD</given-names></name><name><surname>Davis</surname><given-names>FD</given-names></name><name><surname>Faucett</surname><given-names>WA</given-names></name><name><surname>Gottesman</surname><given-names>O</given-names></name><name><surname>Ritchie</surname><given-names>MD</given-names></name><name><surname>Shuldiner</surname><given-names>AR</given-names></name><name><surname>Reid</surname><given-names>JG</given-names></name><name><surname>Ledbetter</surname><given-names>DH</given-names></name><name><surname>Baras</surname><given-names>A</given-names></name><name><surname>Carey</surname><given-names>DJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Distribution and clinical impact of functional variants in 50,726 whole-exome sequences from the DiscovEHR study</article-title><source>Science</source><volume>354</volume><elocation-id>aaf6814</elocation-id><pub-id pub-id-type="doi">10.1126/science.aaf6814</pub-id><pub-id pub-id-type="pmid">28008009</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Dukler</surname><given-names>N</given-names></name><name><surname>Mughal</surname><given-names>MR</given-names></name><name><surname>Ramani</surname><given-names>R</given-names></name><name><surname>Huang</surname><given-names>YF</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Extreme Purifying Selection against Point Mutations in the Human Genome</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.08.23.457339</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duncan</surname><given-names>BK</given-names></name><name><surname>Miller</surname><given-names>JH</given-names></name></person-group><year iso-8601-date="1980">1980</year><article-title>Mutagenic deamination of cytosine residues in DNA</article-title><source>Nature</source><volume>287</volume><fpage>560</fpage><lpage>561</lpage><pub-id pub-id-type="doi">10.1038/287560a0</pub-id><pub-id pub-id-type="pmid">6999365</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eyre-Walker</surname><given-names>A</given-names></name><name><surname>Woolfit</surname><given-names>M</given-names></name><name><surname>Phelps</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The distribution of fitness effects of new deleterious amino acid mutations in humans</article-title><source>Genetics</source><volume>173</volume><fpage>891</fpage><lpage>900</lpage><pub-id pub-id-type="doi">10.1534/genetics.106.057570</pub-id><pub-id pub-id-type="pmid">16547091</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eyre-Walker</surname><given-names>A</given-names></name><name><surname>Keightley</surname><given-names>PD</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The distribution of fitness effects of new mutations</article-title><source>Nature Reviews. Genetics</source><volume>8</volume><fpage>610</fpage><lpage>618</lpage><pub-id pub-id-type="doi">10.1038/nrg2146</pub-id><pub-id pub-id-type="pmid">17637733</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuller</surname><given-names>ZL</given-names></name><name><surname>Berg</surname><given-names>JJ</given-names></name><name><surname>Mostafavi</surname><given-names>H</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Measuring intolerance to mutation in human genetics</article-title><source>Nature Genetics</source><volume>51</volume><fpage>772</fpage><lpage>776</lpage><pub-id pub-id-type="doi">10.1038/s41588-019-0383-1</pub-id><pub-id pub-id-type="pmid">30962618</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>Z</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Sasani</surname><given-names>TA</given-names></name><name><surname>Pedersen</surname><given-names>BS</given-names></name><name><surname>Quinlan</surname><given-names>AR</given-names></name><name><surname>Jorde</surname><given-names>LB</given-names></name><name><surname>Amster</surname><given-names>G</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Overlooked roles of DNA damage and maternal age in generating human germline mutations</article-title><source>PNAS</source><volume>116</volume><fpage>9491</fpage><lpage>9500</lpage><pub-id pub-id-type="doi">10.1073/pnas.1901259116</pub-id><pub-id pub-id-type="pmid">31019089</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghouse</surname><given-names>J</given-names></name><name><surname>Skov</surname><given-names>MW</given-names></name><name><surname>Bigseth</surname><given-names>RS</given-names></name><name><surname>Ahlberg</surname><given-names>G</given-names></name><name><surname>Kanters</surname><given-names>JK</given-names></name><name><surname>Olesen</surname><given-names>MS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Distinguishing pathogenic mutations from background genetic noise in cardiology: The use of large genome databases for genetic interpretation</article-title><source>Clinical Genetics</source><volume>93</volume><fpage>459</fpage><lpage>466</lpage><pub-id pub-id-type="doi">10.1111/cge.13066</pub-id><pub-id pub-id-type="pmid">28589536</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Gillespie</surname><given-names>JH</given-names></name></person-group><year iso-8601-date="1998">1998</year><source>Population genetics: a concise guide / John H. Gillespie</source><publisher-name>The Johns Hopkins University Press</publisher-name></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Glémin</surname><given-names>S</given-names></name><name><surname>Arndt</surname><given-names>PF</given-names></name><name><surname>Messer</surname><given-names>PW</given-names></name><name><surname>Petrov</surname><given-names>D</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Quantification of GC-biased gene conversion in the human genome</article-title><source>Genome Research</source><volume>25</volume><fpage>1215</fpage><lpage>1228</lpage><pub-id pub-id-type="doi">10.1101/gr.185488.114</pub-id><pub-id pub-id-type="pmid">25995268</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Halldorsson</surname><given-names>BV</given-names></name><name><surname>Palsson</surname><given-names>G</given-names></name><name><surname>Stefansson</surname><given-names>OA</given-names></name><name><surname>Jonsson</surname><given-names>H</given-names></name><name><surname>Hardarson</surname><given-names>MT</given-names></name><name><surname>Eggertsson</surname><given-names>HP</given-names></name><name><surname>Gunnarsson</surname><given-names>B</given-names></name><name><surname>Oddsson</surname><given-names>A</given-names></name><name><surname>Halldorsson</surname><given-names>GH</given-names></name><name><surname>Zink</surname><given-names>F</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Frigge</surname><given-names>ML</given-names></name><name><surname>Thorleifsson</surname><given-names>G</given-names></name><name><surname>Sigurdsson</surname><given-names>A</given-names></name><name><surname>Stacey</surname><given-names>SN</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Characterizing mutagenic effects of recombination through a sequence-level genetic map</article-title><source>Science</source><volume>363</volume><elocation-id>eaau1043</elocation-id><pub-id pub-id-type="doi">10.1126/science.aau1043</pub-id><pub-id pub-id-type="pmid">30679340</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harpak</surname><given-names>A</given-names></name><name><surname>Bhaskar</surname><given-names>A</given-names></name><name><surname>Pritchard</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mutation Rate Variation is a Primary Determinant of the Distribution of Allele Frequencies in Humans</article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1006489</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1006489</pub-id><pub-id pub-id-type="pmid">27977673</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Harrison</surname><given-names>SM</given-names></name><name><surname>Pesaran</surname><given-names>TF</given-names></name><name><surname>Mester</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><source>Clinical DNA Variant Interpretation</source><publisher-name>Academic Press</publisher-name></element-citation></ref><ref id="bib25"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hudson</surname><given-names>RR</given-names></name></person-group><year iso-8601-date="1990">1990</year><source>Gene Genealogies and the Coalescent Process</source><publisher-name>Oxford Surveys in Evolutionary Biology</publisher-name></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ioannidis</surname><given-names>NM</given-names></name><name><surname>Rothstein</surname><given-names>JH</given-names></name><name><surname>Pejaver</surname><given-names>V</given-names></name><name><surname>Middha</surname><given-names>S</given-names></name><name><surname>McDonnell</surname><given-names>SK</given-names></name><name><surname>Baheti</surname><given-names>S</given-names></name><name><surname>Musolf</surname><given-names>A</given-names></name><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Holzinger</surname><given-names>E</given-names></name><name><surname>Karyadi</surname><given-names>D</given-names></name><name><surname>Cannon-Albright</surname><given-names>LA</given-names></name><name><surname>Teerlink</surname><given-names>CC</given-names></name><name><surname>Stanford</surname><given-names>JL</given-names></name><name><surname>Isaacs</surname><given-names>WB</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name><name><surname>Cooney</surname><given-names>KA</given-names></name><name><surname>Lange</surname><given-names>EM</given-names></name><name><surname>Schleutker</surname><given-names>J</given-names></name><name><surname>Carpten</surname><given-names>JD</given-names></name><name><surname>Powell</surname><given-names>IJ</given-names></name><name><surname>Cussenot</surname><given-names>O</given-names></name><name><surname>Cancel-Tassin</surname><given-names>G</given-names></name><name><surname>Giles</surname><given-names>GG</given-names></name><name><surname>MacInnis</surname><given-names>RJ</given-names></name><name><surname>Maier</surname><given-names>C</given-names></name><name><surname>Hsieh</surname><given-names>CL</given-names></name><name><surname>Wiklund</surname><given-names>F</given-names></name><name><surname>Catalona</surname><given-names>WJ</given-names></name><name><surname>Foulkes</surname><given-names>WD</given-names></name><name><surname>Mandal</surname><given-names>D</given-names></name><name><surname>Eeles</surname><given-names>RA</given-names></name><name><surname>Kote-Jarai</surname><given-names>Z</given-names></name><name><surname>Bustamante</surname><given-names>CD</given-names></name><name><surname>Schaid</surname><given-names>DJ</given-names></name><name><surname>Hastie</surname><given-names>T</given-names></name><name><surname>Ostrander</surname><given-names>EA</given-names></name><name><surname>Bailey-Wilson</surname><given-names>JE</given-names></name><name><surname>Radivojac</surname><given-names>P</given-names></name><name><surname>Thibodeau</surname><given-names>SN</given-names></name><name><surname>Whittemore</surname><given-names>AS</given-names></name><name><surname>Sieh</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>REVEL: An Ensemble Method for Predicting the Pathogenicity of Rare Missense Variants</article-title><source>American Journal of Human Genetics</source><volume>99</volume><fpage>877</fpage><lpage>885</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2016.08.016</pub-id><pub-id pub-id-type="pmid">27666373</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jónsson</surname><given-names>H</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Kehr</surname><given-names>B</given-names></name><name><surname>Kristmundsdottir</surname><given-names>S</given-names></name><name><surname>Zink</surname><given-names>F</given-names></name><name><surname>Hjartarson</surname><given-names>E</given-names></name><name><surname>Hardarson</surname><given-names>MT</given-names></name><name><surname>Hjorleifsson</surname><given-names>KE</given-names></name><name><surname>Eggertsson</surname><given-names>HP</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Ward</surname><given-names>LD</given-names></name><name><surname>Arnadottir</surname><given-names>GA</given-names></name><name><surname>Helgason</surname><given-names>EA</given-names></name><name><surname>Helgason</surname><given-names>H</given-names></name><name><surname>Gylfason</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Rafnar</surname><given-names>T</given-names></name><name><surname>Frigge</surname><given-names>M</given-names></name><name><surname>Stacey</surname><given-names>SN</given-names></name><name><surname>Th Magnusson</surname><given-names>O</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Kong</surname><given-names>A</given-names></name><name><surname>Halldorsson</surname><given-names>BV</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Parental influence on human germline de novo mutations in 1,548 trios from Iceland</article-title><source>Nature</source><volume>549</volume><fpage>519</fpage><lpage>522</lpage><pub-id pub-id-type="doi">10.1038/nature24018</pub-id><pub-id pub-id-type="pmid">28959963</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaplanis</surname><given-names>J</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>Wiel</surname><given-names>L</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Arvai</surname><given-names>KJ</given-names></name><name><surname>Eberhardt</surname><given-names>RY</given-names></name><name><surname>Gallone</surname><given-names>G</given-names></name><name><surname>Lelieveld</surname><given-names>SH</given-names></name><name><surname>Martin</surname><given-names>HC</given-names></name><name><surname>McRae</surname><given-names>JF</given-names></name><name><surname>Short</surname><given-names>PJ</given-names></name><name><surname>Torene</surname><given-names>RI</given-names></name><name><surname>de Boer</surname><given-names>E</given-names></name><name><surname>Danecek</surname><given-names>P</given-names></name><name><surname>Gardner</surname><given-names>EJ</given-names></name><name><surname>Huang</surname><given-names>N</given-names></name><name><surname>Lord</surname><given-names>J</given-names></name><name><surname>Martincorena</surname><given-names>I</given-names></name><name><surname>Pfundt</surname><given-names>R</given-names></name><name><surname>Reijnders</surname><given-names>MRF</given-names></name><name><surname>Yeung</surname><given-names>A</given-names></name><name><surname>Yntema</surname><given-names>HG</given-names></name><collab>Deciphering Developmental Disorders Study</collab><name><surname>Vissers</surname><given-names>L</given-names></name><name><surname>Juusola</surname><given-names>J</given-names></name><name><surname>Wright</surname><given-names>CF</given-names></name><name><surname>Brunner</surname><given-names>HG</given-names></name><name><surname>Firth</surname><given-names>HV</given-names></name><name><surname>FitzPatrick</surname><given-names>DR</given-names></name><name><surname>Barrett</surname><given-names>JC</given-names></name><name><surname>Hurles</surname><given-names>ME</given-names></name><name><surname>Gilissen</surname><given-names>C</given-names></name><name><surname>Retterer</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Evidence for 28 genetic disorders discovered by combining healthcare and research data</article-title><source>Nature</source><volume>586</volume><fpage>757</fpage><lpage>762</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2832-5</pub-id><pub-id pub-id-type="pmid">33057194</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karczewski</surname><given-names>KJ</given-names></name><name><surname>Francioli</surname><given-names>LC</given-names></name><name><surname>Tiao</surname><given-names>G</given-names></name><name><surname>Cummings</surname><given-names>BB</given-names></name><name><surname>Alföldi</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>Q</given-names></name><name><surname>Collins</surname><given-names>RL</given-names></name><name><surname>Laricchia</surname><given-names>KM</given-names></name><name><surname>Ganna</surname><given-names>A</given-names></name><name><surname>Birnbaum</surname><given-names>DP</given-names></name><name><surname>Gauthier</surname><given-names>LD</given-names></name><name><surname>Brand</surname><given-names>H</given-names></name><name><surname>Solomonson</surname><given-names>M</given-names></name><name><surname>Watts</surname><given-names>NA</given-names></name><name><surname>Rhodes</surname><given-names>D</given-names></name><name><surname>Singer-Berk</surname><given-names>M</given-names></name><name><surname>England</surname><given-names>EM</given-names></name><name><surname>Seaby</surname><given-names>EG</given-names></name><name><surname>Kosmicki</surname><given-names>JA</given-names></name><name><surname>Walters</surname><given-names>RK</given-names></name><name><surname>Tashman</surname><given-names>K</given-names></name><name><surname>Farjoun</surname><given-names>Y</given-names></name><name><surname>Banks</surname><given-names>E</given-names></name><name><surname>Poterba</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>A</given-names></name><name><surname>Seed</surname><given-names>C</given-names></name><name><surname>Whiffin</surname><given-names>N</given-names></name><name><surname>Chong</surname><given-names>JX</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>Pierce-Hoffman</surname><given-names>E</given-names></name><name><surname>Zappala</surname><given-names>Z</given-names></name><name><surname>O’Donnell-Luria</surname><given-names>AH</given-names></name><name><surname>Minikel</surname><given-names>EV</given-names></name><name><surname>Weisburd</surname><given-names>B</given-names></name><name><surname>Lek</surname><given-names>M</given-names></name><name><surname>Ware</surname><given-names>JS</given-names></name><name><surname>Vittal</surname><given-names>C</given-names></name><name><surname>Armean</surname><given-names>IM</given-names></name><name><surname>Bergelson</surname><given-names>L</given-names></name><name><surname>Cibulskis</surname><given-names>K</given-names></name><name><surname>Connolly</surname><given-names>KM</given-names></name><name><surname>Covarrubias</surname><given-names>M</given-names></name><name><surname>Donnelly</surname><given-names>S</given-names></name><name><surname>Ferriera</surname><given-names>S</given-names></name><name><surname>Gabriel</surname><given-names>S</given-names></name><name><surname>Gentry</surname><given-names>J</given-names></name><name><surname>Gupta</surname><given-names>N</given-names></name><name><surname>Jeandet</surname><given-names>T</given-names></name><name><surname>Kaplan</surname><given-names>D</given-names></name><name><surname>Llanwarne</surname><given-names>C</given-names></name><name><surname>Munshi</surname><given-names>R</given-names></name><name><surname>Novod</surname><given-names>S</given-names></name><name><surname>Petrillo</surname><given-names>N</given-names></name><name><surname>Roazen</surname><given-names>D</given-names></name><name><surname>Ruano-Rubio</surname><given-names>V</given-names></name><name><surname>Saltzman</surname><given-names>A</given-names></name><name><surname>Schleicher</surname><given-names>M</given-names></name><name><surname>Soto</surname><given-names>J</given-names></name><name><surname>Tibbetts</surname><given-names>K</given-names></name><name><surname>Tolonen</surname><given-names>C</given-names></name><name><surname>Wade</surname><given-names>G</given-names></name><name><surname>Talkowski</surname><given-names>ME</given-names></name><collab>Genome Aggregation Database Consortium</collab><name><surname>Neale</surname><given-names>BM</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The mutational constraint spectrum quantified from variation in 141,456 humans</article-title><source>Nature</source><volume>581</volume><fpage>434</fpage><lpage>443</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2308-7</pub-id><pub-id pub-id-type="pmid">32461654</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kelleher</surname><given-names>J</given-names></name><name><surname>Etheridge</surname><given-names>AM</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Efficient Coalescent Simulation and Genealogical Analysis for Large Sample Sizes</article-title><source>PLOS Computational Biology</source><volume>12</volume><elocation-id>e1004842</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1004842</pub-id><pub-id pub-id-type="pmid">27145223</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kelleher</surname><given-names>J</given-names></name><name><surname>Wong</surname><given-names>Y</given-names></name><name><surname>Wohns</surname><given-names>AW</given-names></name><name><surname>Fadil</surname><given-names>C</given-names></name><name><surname>Albers</surname><given-names>PK</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Inferring whole-genome histories in large population datasets</article-title><source>Nature Genetics</source><volume>51</volume><fpage>1330</fpage><lpage>1338</lpage><pub-id pub-id-type="doi">10.1038/s41588-019-0483-y</pub-id><pub-id pub-id-type="pmid">31477934</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>BY</given-names></name><name><surname>Huber</surname><given-names>CD</given-names></name><name><surname>Lohmueller</surname><given-names>KE</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Inference of the Distribution of Selection Coefficients for New Nonsynonymous Mutations Using Large Samples</article-title><source>Genetics</source><volume>206</volume><fpage>345</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1534/genetics.116.197145</pub-id><pub-id pub-id-type="pmid">28249985</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kong</surname><given-names>A</given-names></name><name><surname>Frigge</surname><given-names>ML</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Besenbacher</surname><given-names>S</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Magnusson</surname><given-names>G</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Sigurdsson</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Wong</surname><given-names>WSW</given-names></name><name><surname>Sigurdsson</surname><given-names>G</given-names></name><name><surname>Walters</surname><given-names>GB</given-names></name><name><surname>Steinberg</surname><given-names>S</given-names></name><name><surname>Helgason</surname><given-names>H</given-names></name><name><surname>Thorleifsson</surname><given-names>G</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Magnusson</surname><given-names>OT</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Rate of de novo mutations and the importance of father’s age to disease risk</article-title><source>Nature</source><volume>488</volume><fpage>471</fpage><lpage>475</lpage><pub-id pub-id-type="doi">10.1038/nature11396</pub-id><pub-id pub-id-type="pmid">22914163</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landrum</surname><given-names>MJ</given-names></name><name><surname>Lee</surname><given-names>JM</given-names></name><name><surname>Benson</surname><given-names>M</given-names></name><name><surname>Brown</surname><given-names>GR</given-names></name><name><surname>Chao</surname><given-names>C</given-names></name><name><surname>Chitipiralla</surname><given-names>S</given-names></name><name><surname>Gu</surname><given-names>B</given-names></name><name><surname>Hart</surname><given-names>J</given-names></name><name><surname>Hoffman</surname><given-names>D</given-names></name><name><surname>Jang</surname><given-names>W</given-names></name><name><surname>Karapetyan</surname><given-names>K</given-names></name><name><surname>Katz</surname><given-names>K</given-names></name><name><surname>Liu</surname><given-names>C</given-names></name><name><surname>Maddipatla</surname><given-names>Z</given-names></name><name><surname>Malheiro</surname><given-names>A</given-names></name><name><surname>McDaniel</surname><given-names>K</given-names></name><name><surname>Ovetsky</surname><given-names>M</given-names></name><name><surname>Riley</surname><given-names>G</given-names></name><name><surname>Zhou</surname><given-names>G</given-names></name><name><surname>Holmes</surname><given-names>JB</given-names></name><name><surname>Kattman</surname><given-names>BL</given-names></name><name><surname>Maglott</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>ClinVar: improving access to variant interpretations and supporting evidence</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>D1062</fpage><lpage>D1067</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx1153</pub-id><pub-id pub-id-type="pmid">29165669</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lek</surname><given-names>M</given-names></name><name><surname>Karczewski</surname><given-names>KJ</given-names></name><name><surname>Minikel</surname><given-names>EV</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>Banks</surname><given-names>E</given-names></name><name><surname>Fennell</surname><given-names>T</given-names></name><name><surname>O’Donnell-Luria</surname><given-names>AH</given-names></name><name><surname>Ware</surname><given-names>JS</given-names></name><name><surname>Hill</surname><given-names>AJ</given-names></name><name><surname>Cummings</surname><given-names>BB</given-names></name><name><surname>Tukiainen</surname><given-names>T</given-names></name><name><surname>Birnbaum</surname><given-names>DP</given-names></name><name><surname>Kosmicki</surname><given-names>JA</given-names></name><name><surname>Duncan</surname><given-names>LE</given-names></name><name><surname>Estrada</surname><given-names>K</given-names></name><name><surname>Zhao</surname><given-names>F</given-names></name><name><surname>Zou</surname><given-names>J</given-names></name><name><surname>Pierce-Hoffman</surname><given-names>E</given-names></name><name><surname>Berghout</surname><given-names>J</given-names></name><name><surname>Cooper</surname><given-names>DN</given-names></name><name><surname>Deflaux</surname><given-names>N</given-names></name><name><surname>DePristo</surname><given-names>M</given-names></name><name><surname>Do</surname><given-names>R</given-names></name><name><surname>Flannick</surname><given-names>J</given-names></name><name><surname>Fromer</surname><given-names>M</given-names></name><name><surname>Gauthier</surname><given-names>L</given-names></name><name><surname>Goldstein</surname><given-names>J</given-names></name><name><surname>Gupta</surname><given-names>N</given-names></name><name><surname>Howrigan</surname><given-names>D</given-names></name><name><surname>Kiezun</surname><given-names>A</given-names></name><name><surname>Kurki</surname><given-names>MI</given-names></name><name><surname>Moonshine</surname><given-names>AL</given-names></name><name><surname>Natarajan</surname><given-names>P</given-names></name><name><surname>Orozco</surname><given-names>L</given-names></name><name><surname>Peloso</surname><given-names>GM</given-names></name><name><surname>Poplin</surname><given-names>R</given-names></name><name><surname>Rivas</surname><given-names>MA</given-names></name><name><surname>Ruano-Rubio</surname><given-names>V</given-names></name><name><surname>Rose</surname><given-names>SA</given-names></name><name><surname>Ruderfer</surname><given-names>DM</given-names></name><name><surname>Shakir</surname><given-names>K</given-names></name><name><surname>Stenson</surname><given-names>PD</given-names></name><name><surname>Stevens</surname><given-names>C</given-names></name><name><surname>Thomas</surname><given-names>BP</given-names></name><name><surname>Tiao</surname><given-names>G</given-names></name><name><surname>Tusie-Luna</surname><given-names>MT</given-names></name><name><surname>Weisburd</surname><given-names>B</given-names></name><name><surname>Won</surname><given-names>HH</given-names></name><name><surname>Yu</surname><given-names>D</given-names></name><name><surname>Altshuler</surname><given-names>DM</given-names></name><name><surname>Ardissino</surname><given-names>D</given-names></name><name><surname>Boehnke</surname><given-names>M</given-names></name><name><surname>Danesh</surname><given-names>J</given-names></name><name><surname>Donnelly</surname><given-names>S</given-names></name><name><surname>Elosua</surname><given-names>R</given-names></name><name><surname>Florez</surname><given-names>JC</given-names></name><name><surname>Gabriel</surname><given-names>SB</given-names></name><name><surname>Getz</surname><given-names>G</given-names></name><name><surname>Glatt</surname><given-names>SJ</given-names></name><name><surname>Hultman</surname><given-names>CM</given-names></name><name><surname>Kathiresan</surname><given-names>S</given-names></name><name><surname>Laakso</surname><given-names>M</given-names></name><name><surname>McCarroll</surname><given-names>S</given-names></name><name><surname>McCarthy</surname><given-names>MI</given-names></name><name><surname>McGovern</surname><given-names>D</given-names></name><name><surname>McPherson</surname><given-names>R</given-names></name><name><surname>Neale</surname><given-names>BM</given-names></name><name><surname>Palotie</surname><given-names>A</given-names></name><name><surname>Purcell</surname><given-names>SM</given-names></name><name><surname>Saleheen</surname><given-names>D</given-names></name><name><surname>Scharf</surname><given-names>JM</given-names></name><name><surname>Sklar</surname><given-names>P</given-names></name><name><surname>Sullivan</surname><given-names>PF</given-names></name><name><surname>Tuomilehto</surname><given-names>J</given-names></name><name><surname>Tsuang</surname><given-names>MT</given-names></name><name><surname>Watkins</surname><given-names>HC</given-names></name><name><surname>Wilson</surname><given-names>JG</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><collab>Exome Aggregation Consortium</collab></person-group><year iso-8601-date="2016">2016</year><article-title>Analysis of protein-coding genetic variation in 60,706 humans</article-title><source>Nature</source><volume>536</volume><fpage>285</fpage><lpage>291</lpage><pub-id pub-id-type="doi">10.1038/nature19057</pub-id><pub-id pub-id-type="pmid">27535533</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McDonald</surname><given-names>JH</given-names></name><name><surname>Kreitman</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Adaptive protein evolution at the Adh locus in <italic>Drosophila</italic></article-title><source>Nature</source><volume>351</volume><fpage>652</fpage><lpage>654</lpage><pub-id pub-id-type="doi">10.1038/351652a0</pub-id><pub-id pub-id-type="pmid">1904993</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McLaren</surname><given-names>W</given-names></name><name><surname>Gil</surname><given-names>L</given-names></name><name><surname>Hunt</surname><given-names>SE</given-names></name><name><surname>Riat</surname><given-names>HS</given-names></name><name><surname>Ritchie</surname><given-names>GRS</given-names></name><name><surname>Thormann</surname><given-names>A</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Cunningham</surname><given-names>F</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The Ensembl Variant Effect Predictor</article-title><source>Genome Biology</source><volume>17</volume><elocation-id>122</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-016-0974-4</pub-id><pub-id pub-id-type="pmid">27268795</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVicker</surname><given-names>G</given-names></name><name><surname>Gordon</surname><given-names>D</given-names></name><name><surname>Davis</surname><given-names>C</given-names></name><name><surname>Green</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Widespread genomic signatures of natural selection in hominid evolution</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000471</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000471</pub-id><pub-id pub-id-type="pmid">19424416</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nachman</surname><given-names>MW</given-names></name><name><surname>Crowell</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Estimate of the mutation rate per nucleotide in humans</article-title><source>Genetics</source><volume>156</volume><fpage>297</fpage><lpage>304</lpage><pub-id pub-id-type="doi">10.1093/genetics/156.1.297</pub-id><pub-id pub-id-type="pmid">10978293</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Need</surname><given-names>AC</given-names></name><name><surname>Shashi</surname><given-names>V</given-names></name><name><surname>Hitomi</surname><given-names>Y</given-names></name><name><surname>Schoch</surname><given-names>K</given-names></name><name><surname>Shianna</surname><given-names>KV</given-names></name><name><surname>McDonald</surname><given-names>MT</given-names></name><name><surname>Meisler</surname><given-names>MH</given-names></name><name><surname>Goldstein</surname><given-names>DB</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Clinical application of exome sequencing in undiagnosed genetic conditions</article-title><source>Journal of Medical Genetics</source><volume>49</volume><fpage>353</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1136/jmedgenet-2012-100819</pub-id><pub-id pub-id-type="pmid">22581936</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nelson</surname><given-names>MR</given-names></name><name><surname>Wegmann</surname><given-names>D</given-names></name><name><surname>Ehm</surname><given-names>MG</given-names></name><name><surname>Kessner</surname><given-names>D</given-names></name><name><surname>St Jean</surname><given-names>P</given-names></name><name><surname>Verzilli</surname><given-names>C</given-names></name><name><surname>Shen</surname><given-names>J</given-names></name><name><surname>Tang</surname><given-names>Z</given-names></name><name><surname>Bacanu</surname><given-names>SA</given-names></name><name><surname>Fraser</surname><given-names>D</given-names></name><name><surname>Warren</surname><given-names>L</given-names></name><name><surname>Aponte</surname><given-names>J</given-names></name><name><surname>Zawistowski</surname><given-names>M</given-names></name><name><surname>Liu</surname><given-names>X</given-names></name><name><surname>Zhang</surname><given-names>H</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Woollard</surname><given-names>P</given-names></name><name><surname>Topp</surname><given-names>S</given-names></name><name><surname>Hall</surname><given-names>MD</given-names></name><name><surname>Nangle</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Abecasis</surname><given-names>G</given-names></name><name><surname>Cardon</surname><given-names>LR</given-names></name><name><surname>Zöllner</surname><given-names>S</given-names></name><name><surname>Whittaker</surname><given-names>JC</given-names></name><name><surname>Chissoe</surname><given-names>SL</given-names></name><name><surname>Novembre</surname><given-names>J</given-names></name><name><surname>Mooser</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>An abundance of rare functional variants in 202 drug target genes sequenced in 14,002 people</article-title><source>Science</source><volume>337</volume><fpage>100</fpage><lpage>104</lpage><pub-id pub-id-type="doi">10.1126/science.1217876</pub-id><pub-id pub-id-type="pmid">22604722</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Otto</surname><given-names>SP</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Detecting the form of selection from DNA sequence data</article-title><source>Trends in Genetics</source><volume>16</volume><fpage>526</fpage><lpage>529</lpage><pub-id pub-id-type="doi">10.1016/s0168-9525(00)02141-7</pub-id><pub-id pub-id-type="pmid">11102697</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pollard</surname><given-names>KS</given-names></name><name><surname>Hubisz</surname><given-names>MJ</given-names></name><name><surname>Rosenbloom</surname><given-names>KR</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Detection of nonneutral substitution rates on mammalian phylogenies</article-title><source>Genome Research</source><volume>20</volume><fpage>110</fpage><lpage>121</lpage><pub-id pub-id-type="doi">10.1101/gr.097857.109</pub-id><pub-id pub-id-type="pmid">19858363</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Poulos</surname><given-names>RC</given-names></name><name><surname>Olivier</surname><given-names>J</given-names></name><name><surname>Wong</surname><given-names>JWH</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The interaction between cytosine methylation and processes of DNA replication and repair shape the mutational landscape of cancer genomes</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>7786</fpage><lpage>7795</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx463</pub-id><pub-id pub-id-type="pmid">28531315</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rauch</surname><given-names>A</given-names></name><name><surname>Wieczorek</surname><given-names>D</given-names></name><name><surname>Graf</surname><given-names>E</given-names></name><name><surname>Wieland</surname><given-names>T</given-names></name><name><surname>Endele</surname><given-names>S</given-names></name><name><surname>Schwarzmayr</surname><given-names>T</given-names></name><name><surname>Albrecht</surname><given-names>B</given-names></name><name><surname>Bartholdi</surname><given-names>D</given-names></name><name><surname>Beygo</surname><given-names>J</given-names></name><name><surname>Di Donato</surname><given-names>N</given-names></name><name><surname>Dufke</surname><given-names>A</given-names></name><name><surname>Cremer</surname><given-names>K</given-names></name><name><surname>Hempel</surname><given-names>M</given-names></name><name><surname>Horn</surname><given-names>D</given-names></name><name><surname>Hoyer</surname><given-names>J</given-names></name><name><surname>Joset</surname><given-names>P</given-names></name><name><surname>Röpke</surname><given-names>A</given-names></name><name><surname>Moog</surname><given-names>U</given-names></name><name><surname>Riess</surname><given-names>A</given-names></name><name><surname>Thiel</surname><given-names>CT</given-names></name><name><surname>Tzschach</surname><given-names>A</given-names></name><name><surname>Wiesener</surname><given-names>A</given-names></name><name><surname>Wohlleber</surname><given-names>E</given-names></name><name><surname>Zweier</surname><given-names>C</given-names></name><name><surname>Ekici</surname><given-names>AB</given-names></name><name><surname>Zink</surname><given-names>AM</given-names></name><name><surname>Rump</surname><given-names>A</given-names></name><name><surname>Meisinger</surname><given-names>C</given-names></name><name><surname>Grallert</surname><given-names>H</given-names></name><name><surname>Sticht</surname><given-names>H</given-names></name><name><surname>Schenck</surname><given-names>A</given-names></name><name><surname>Engels</surname><given-names>H</given-names></name><name><surname>Rappold</surname><given-names>G</given-names></name><name><surname>Schröck</surname><given-names>E</given-names></name><name><surname>Wieacker</surname><given-names>P</given-names></name><name><surname>Riess</surname><given-names>O</given-names></name><name><surname>Meitinger</surname><given-names>T</given-names></name><name><surname>Reis</surname><given-names>A</given-names></name><name><surname>Strom</surname><given-names>TM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Range of genetic mutations associated with severe non-syndromic sporadic intellectual disability: an exome sequencing study</article-title><source>Lancet</source><volume>380</volume><fpage>1674</fpage><lpage>1682</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(12)61480-9</pub-id><pub-id pub-id-type="pmid">23020937</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rentzsch</surname><given-names>P</given-names></name><name><surname>Witten</surname><given-names>D</given-names></name><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>CADD: predicting the deleteriousness of variants throughout the human genome</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D886</fpage><lpage>D894</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1016</pub-id><pub-id pub-id-type="pmid">30371827</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Richards</surname><given-names>S</given-names></name><name><surname>Aziz</surname><given-names>N</given-names></name><name><surname>Bale</surname><given-names>S</given-names></name><name><surname>Bick</surname><given-names>D</given-names></name><name><surname>Das</surname><given-names>S</given-names></name><name><surname>Gastier-Foster</surname><given-names>J</given-names></name><name><surname>Grody</surname><given-names>WW</given-names></name><name><surname>Hegde</surname><given-names>M</given-names></name><name><surname>Lyon</surname><given-names>E</given-names></name><name><surname>Spector</surname><given-names>E</given-names></name><name><surname>Voelkerding</surname><given-names>K</given-names></name><name><surname>Rehm</surname><given-names>HL</given-names></name><collab>ACMG Laboratory Quality Assurance Committee</collab></person-group><year iso-8601-date="2015">2015</year><article-title>Standards and guidelines for the interpretation of sequence variants: a joint consensus recommendation of the American College of Medical Genetics and Genomics and the Association for Molecular Pathology</article-title><source>Genetics in Medicine</source><volume>17</volume><fpage>405</fpage><lpage>424</lpage><pub-id pub-id-type="doi">10.1038/gim.2015.30</pub-id><pub-id pub-id-type="pmid">25741868</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sanders</surname><given-names>SJ</given-names></name><name><surname>Murtha</surname><given-names>MT</given-names></name><name><surname>Gupta</surname><given-names>AR</given-names></name><name><surname>Murdoch</surname><given-names>JD</given-names></name><name><surname>Raubeson</surname><given-names>MJ</given-names></name><name><surname>Willsey</surname><given-names>AJ</given-names></name><name><surname>Ercan-Sencicek</surname><given-names>AG</given-names></name><name><surname>DiLullo</surname><given-names>NM</given-names></name><name><surname>Parikshak</surname><given-names>NN</given-names></name><name><surname>Stein</surname><given-names>JL</given-names></name><name><surname>Walker</surname><given-names>MF</given-names></name><name><surname>Ober</surname><given-names>GT</given-names></name><name><surname>Teran</surname><given-names>NA</given-names></name><name><surname>Song</surname><given-names>Y</given-names></name><name><surname>El-Fishawy</surname><given-names>P</given-names></name><name><surname>Murtha</surname><given-names>RC</given-names></name><name><surname>Choi</surname><given-names>M</given-names></name><name><surname>Overton</surname><given-names>JD</given-names></name><name><surname>Bjornson</surname><given-names>RD</given-names></name><name><surname>Carriero</surname><given-names>NJ</given-names></name><name><surname>Meyer</surname><given-names>KA</given-names></name><name><surname>Bilguvar</surname><given-names>K</given-names></name><name><surname>Mane</surname><given-names>SM</given-names></name><name><surname>Sestan</surname><given-names>N</given-names></name><name><surname>Lifton</surname><given-names>RP</given-names></name><name><surname>Günel</surname><given-names>M</given-names></name><name><surname>Roeder</surname><given-names>K</given-names></name><name><surname>Geschwind</surname><given-names>DH</given-names></name><name><surname>Devlin</surname><given-names>B</given-names></name><name><surname>State</surname><given-names>MW</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>De novo mutations revealed by whole-exome sequencing are strongly associated with autism</article-title><source>Nature</source><volume>485</volume><fpage>237</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1038/nature10945</pub-id><pub-id pub-id-type="pmid">22495306</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sawyer</surname><given-names>SA</given-names></name><name><surname>Hartl</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Population genetics of polymorphism and divergence</article-title><source>Genetics</source><volume>132</volume><fpage>1161</fpage><lpage>1176</lpage><pub-id pub-id-type="doi">10.1093/genetics/132.4.1161</pub-id><pub-id pub-id-type="pmid">1459433</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schiffels</surname><given-names>S</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Inferring human population size and separation history from multiple genome sequences</article-title><source>Nature Genetics</source><volume>46</volume><fpage>919</fpage><lpage>925</lpage><pub-id pub-id-type="doi">10.1038/ng.3015</pub-id><pub-id pub-id-type="pmid">24952747</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Seplyarskiy</surname><given-names>VB</given-names></name><name><surname>Sunyaev</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The origin of human mutation in light of genomic data</article-title><source>Nature Reviews. Genetics</source><volume>22</volume><fpage>672</fpage><lpage>686</lpage><pub-id pub-id-type="doi">10.1038/s41576-021-00376-2</pub-id><pub-id pub-id-type="pmid">34163020</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siepel</surname><given-names>A</given-names></name><name><surname>Bejerano</surname><given-names>G</given-names></name><name><surname>Pedersen</surname><given-names>JS</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Hou</surname><given-names>M</given-names></name><name><surname>Rosenbloom</surname><given-names>K</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Spieth</surname><given-names>J</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Richards</surname><given-names>S</given-names></name><name><surname>Weinstock</surname><given-names>GM</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Miller</surname><given-names>W</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Evolutionarily conserved elements in vertebrate, insect, worm, and yeast genomes</article-title><source>Genome Research</source><volume>15</volume><fpage>1034</fpage><lpage>1050</lpage><pub-id pub-id-type="doi">10.1101/gr.3715005</pub-id><pub-id pub-id-type="pmid">16024819</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simons</surname><given-names>YB</given-names></name><name><surname>Turchin</surname><given-names>MC</given-names></name><name><surname>Pritchard</surname><given-names>JK</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The deleterious mutation load is insensitive to recent population history</article-title><source>Nature Genetics</source><volume>46</volume><fpage>220</fpage><lpage>224</lpage><pub-id pub-id-type="doi">10.1038/ng.2896</pub-id><pub-id pub-id-type="pmid">24509481</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>TCA</given-names></name><name><surname>Arndt</surname><given-names>PF</given-names></name><name><surname>Eyre-Walker</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Large scale variation in the rate of germ-line de novo mutation, base composition, divergence and diversity in humans</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007254</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007254</pub-id><pub-id pub-id-type="pmid">29590096</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Speidel</surname><given-names>L</given-names></name><name><surname>Forest</surname><given-names>M</given-names></name><name><surname>Shi</surname><given-names>S</given-names></name><name><surname>Myers</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A method for genome-wide genealogy estimation for thousands of samples</article-title><source>Nature Genetics</source><volume>51</volume><fpage>1321</fpage><lpage>1329</lpage><pub-id pub-id-type="doi">10.1038/s41588-019-0484-x</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name><name><surname>Adzhubei</surname><given-names>I</given-names></name><name><surname>Thurman</surname><given-names>RE</given-names></name><name><surname>Kryukov</surname><given-names>GV</given-names></name><name><surname>Mirkin</surname><given-names>SM</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Human mutation rate associated with DNA replication timing</article-title><source>Nature Genetics</source><volume>41</volume><fpage>393</fpage><lpage>395</lpage><pub-id pub-id-type="doi">10.1038/ng.363</pub-id><pub-id pub-id-type="pmid">19287383</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stanek</surname><given-names>D</given-names></name><name><surname>Bis-Brewer</surname><given-names>DM</given-names></name><name><surname>Saghira</surname><given-names>C</given-names></name><name><surname>Danzi</surname><given-names>MC</given-names></name><name><surname>Seeman</surname><given-names>P</given-names></name><name><surname>Lassuthova</surname><given-names>P</given-names></name><name><surname>Zuchner</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Prot2HG: a database of protein domains mapped to the human genome</article-title><source>Database</source><volume>2020</volume><elocation-id>baz161</elocation-id><pub-id pub-id-type="doi">10.1093/database/baz161</pub-id><pub-id pub-id-type="pmid">32293014</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Szustakowski</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Advancing Human Genetics Research and Drug Discovery through Exome Sequencing of the UK Biobank</article-title><source>medRxiv</source><ext-link ext-link-type="uri" xlink:href="https://www.medrxiv.org/content/10.1101/2020.11.02.20222232v1">https://www.medrxiv.org/content/10.1101/2020.11.02.20222232v1</ext-link></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Taliun</surname><given-names>D</given-names></name><name><surname>Harris</surname><given-names>DN</given-names></name><name><surname>Kessler</surname><given-names>MD</given-names></name><name><surname>Carlson</surname><given-names>J</given-names></name><name><surname>Szpiech</surname><given-names>ZA</given-names></name><name><surname>Torres</surname><given-names>R</given-names></name><name><surname>Taliun</surname><given-names>SAG</given-names></name><name><surname>Corvelo</surname><given-names>A</given-names></name><name><surname>Gogarten</surname><given-names>SM</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Pitsillides</surname><given-names>AN</given-names></name><name><surname>LeFaive</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>S-B</given-names></name><name><surname>Tian</surname><given-names>X</given-names></name><name><surname>Browning</surname><given-names>BL</given-names></name><name><surname>Das</surname><given-names>S</given-names></name><name><surname>Emde</surname><given-names>A-K</given-names></name><name><surname>Clarke</surname><given-names>WE</given-names></name><name><surname>Loesch</surname><given-names>DP</given-names></name><name><surname>Shetty</surname><given-names>AC</given-names></name><name><surname>Blackwell</surname><given-names>TW</given-names></name><name><surname>Smith</surname><given-names>AV</given-names></name><name><surname>Wong</surname><given-names>Q</given-names></name><name><surname>Liu</surname><given-names>X</given-names></name><name><surname>Conomos</surname><given-names>MP</given-names></name><name><surname>Bobo</surname><given-names>DM</given-names></name><name><surname>Aguet</surname><given-names>F</given-names></name><name><surname>Albert</surname><given-names>C</given-names></name><name><surname>Alonso</surname><given-names>A</given-names></name><name><surname>Ardlie</surname><given-names>KG</given-names></name><name><surname>Arking</surname><given-names>DE</given-names></name><name><surname>Aslibekyan</surname><given-names>S</given-names></name><name><surname>Auer</surname><given-names>PL</given-names></name><name><surname>Barnard</surname><given-names>J</given-names></name><name><surname>Barr</surname><given-names>RG</given-names></name><name><surname>Barwick</surname><given-names>L</given-names></name><name><surname>Becker</surname><given-names>LC</given-names></name><name><surname>Beer</surname><given-names>RL</given-names></name><name><surname>Benjamin</surname><given-names>EJ</given-names></name><name><surname>Bielak</surname><given-names>LF</given-names></name><name><surname>Blangero</surname><given-names>J</given-names></name><name><surname>Boehnke</surname><given-names>M</given-names></name><name><surname>Bowden</surname><given-names>DW</given-names></name><name><surname>Brody</surname><given-names>JA</given-names></name><name><surname>Burchard</surname><given-names>EG</given-names></name><name><surname>Cade</surname><given-names>BE</given-names></name><name><surname>Casella</surname><given-names>JF</given-names></name><name><surname>Chalazan</surname><given-names>B</given-names></name><name><surname>Chasman</surname><given-names>DI</given-names></name><name><surname>Chen</surname><given-names>Y-DI</given-names></name><name><surname>Cho</surname><given-names>MH</given-names></name><name><surname>Choi</surname><given-names>SH</given-names></name><name><surname>Chung</surname><given-names>MK</given-names></name><name><surname>Clish</surname><given-names>CB</given-names></name><name><surname>Correa</surname><given-names>A</given-names></name><name><surname>Curran</surname><given-names>JE</given-names></name><name><surname>Custer</surname><given-names>B</given-names></name><name><surname>Darbar</surname><given-names>D</given-names></name><name><surname>Daya</surname><given-names>M</given-names></name><name><surname>de Andrade</surname><given-names>M</given-names></name><name><surname>DeMeo</surname><given-names>DL</given-names></name><name><surname>Dutcher</surname><given-names>SK</given-names></name><name><surname>Ellinor</surname><given-names>PT</given-names></name><name><surname>Emery</surname><given-names>LS</given-names></name><name><surname>Eng</surname><given-names>C</given-names></name><name><surname>Fatkin</surname><given-names>D</given-names></name><name><surname>Fingerlin</surname><given-names>T</given-names></name><name><surname>Forer</surname><given-names>L</given-names></name><name><surname>Fornage</surname><given-names>M</given-names></name><name><surname>Franceschini</surname><given-names>N</given-names></name><name><surname>Fuchsberger</surname><given-names>C</given-names></name><name><surname>Fullerton</surname><given-names>SM</given-names></name><name><surname>Germer</surname><given-names>S</given-names></name><name><surname>Gladwin</surname><given-names>MT</given-names></name><name><surname>Gottlieb</surname><given-names>DJ</given-names></name><name><surname>Guo</surname><given-names>X</given-names></name><name><surname>Hall</surname><given-names>ME</given-names></name><name><surname>He</surname><given-names>J</given-names></name><name><surname>Heard-Costa</surname><given-names>NL</given-names></name><name><surname>Heckbert</surname><given-names>SR</given-names></name><name><surname>Irvin</surname><given-names>MR</given-names></name><name><surname>Johnsen</surname><given-names>JM</given-names></name><name><surname>Johnson</surname><given-names>AD</given-names></name><name><surname>Kaplan</surname><given-names>R</given-names></name><name><surname>Kardia</surname><given-names>SLR</given-names></name><name><surname>Kelly</surname><given-names>T</given-names></name><name><surname>Kelly</surname><given-names>S</given-names></name><name><surname>Kenny</surname><given-names>EE</given-names></name><name><surname>Kiel</surname><given-names>DP</given-names></name><name><surname>Klemmer</surname><given-names>R</given-names></name><name><surname>Konkle</surname><given-names>BA</given-names></name><name><surname>Kooperberg</surname><given-names>C</given-names></name><name><surname>Köttgen</surname><given-names>A</given-names></name><name><surname>Lange</surname><given-names>LA</given-names></name><name><surname>Lasky-Su</surname><given-names>J</given-names></name><name><surname>Levy</surname><given-names>D</given-names></name><name><surname>Lin</surname><given-names>X</given-names></name><name><surname>Lin</surname><given-names>K-H</given-names></name><name><surname>Liu</surname><given-names>C</given-names></name><name><surname>Loos</surname><given-names>RJF</given-names></name><name><surname>Garman</surname><given-names>L</given-names></name><name><surname>Gerszten</surname><given-names>R</given-names></name><name><surname>Lubitz</surname><given-names>SA</given-names></name><name><surname>Lunetta</surname><given-names>KL</given-names></name><name><surname>Mak</surname><given-names>ACY</given-names></name><name><surname>Manichaikul</surname><given-names>A</given-names></name><name><surname>Manning</surname><given-names>AK</given-names></name><name><surname>Mathias</surname><given-names>RA</given-names></name><name><surname>McManus</surname><given-names>DD</given-names></name><name><surname>McGarvey</surname><given-names>ST</given-names></name><name><surname>Meigs</surname><given-names>JB</given-names></name><name><surname>Meyers</surname><given-names>DA</given-names></name><name><surname>Mikulla</surname><given-names>JL</given-names></name><name><surname>Minear</surname><given-names>MA</given-names></name><name><surname>Mitchell</surname><given-names>BD</given-names></name><name><surname>Mohanty</surname><given-names>S</given-names></name><name><surname>Montasser</surname><given-names>ME</given-names></name><name><surname>Montgomery</surname><given-names>C</given-names></name><name><surname>Morrison</surname><given-names>AC</given-names></name><name><surname>Murabito</surname><given-names>JM</given-names></name><name><surname>Natale</surname><given-names>A</given-names></name><name><surname>Natarajan</surname><given-names>P</given-names></name><name><surname>Nelson</surname><given-names>SC</given-names></name><name><surname>North</surname><given-names>KE</given-names></name><name><surname>O’Connell</surname><given-names>JR</given-names></name><name><surname>Palmer</surname><given-names>ND</given-names></name><name><surname>Pankratz</surname><given-names>N</given-names></name><name><surname>Peloso</surname><given-names>GM</given-names></name><name><surname>Peyser</surname><given-names>PA</given-names></name><name><surname>Pleiness</surname><given-names>J</given-names></name><name><surname>Post</surname><given-names>WS</given-names></name><name><surname>Psaty</surname><given-names>BM</given-names></name><name><surname>Rao</surname><given-names>DC</given-names></name><name><surname>Redline</surname><given-names>S</given-names></name><name><surname>Reiner</surname><given-names>AP</given-names></name><name><surname>Roden</surname><given-names>D</given-names></name><name><surname>Rotter</surname><given-names>JI</given-names></name><name><surname>Ruczinski</surname><given-names>I</given-names></name><name><surname>Sarnowski</surname><given-names>C</given-names></name><name><surname>Schoenherr</surname><given-names>S</given-names></name><name><surname>Schwartz</surname><given-names>DA</given-names></name><name><surname>Seo</surname><given-names>J-S</given-names></name><name><surname>Seshadri</surname><given-names>S</given-names></name><name><surname>Sheehan</surname><given-names>VA</given-names></name><name><surname>Sheu</surname><given-names>WH</given-names></name><name><surname>Shoemaker</surname><given-names>MB</given-names></name><name><surname>Smith</surname><given-names>NL</given-names></name><name><surname>Smith</surname><given-names>JA</given-names></name><name><surname>Sotoodehnia</surname><given-names>N</given-names></name><name><surname>Stilp</surname><given-names>AM</given-names></name><name><surname>Tang</surname><given-names>W</given-names></name><name><surname>Taylor</surname><given-names>KD</given-names></name><name><surname>Telen</surname><given-names>M</given-names></name><name><surname>Thornton</surname><given-names>TA</given-names></name><name><surname>Tracy</surname><given-names>RP</given-names></name><name><surname>Van Den Berg</surname><given-names>DJ</given-names></name><name><surname>Vasan</surname><given-names>RS</given-names></name><name><surname>Viaud-Martinez</surname><given-names>KA</given-names></name><name><surname>Vrieze</surname><given-names>S</given-names></name><name><surname>Weeks</surname><given-names>DE</given-names></name><name><surname>Weir</surname><given-names>BS</given-names></name><name><surname>Weiss</surname><given-names>ST</given-names></name><name><surname>Weng</surname><given-names>L-C</given-names></name><name><surname>Willer</surname><given-names>CJ</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhao</surname><given-names>X</given-names></name><name><surname>Arnett</surname><given-names>DK</given-names></name><name><surname>Ashley-Koch</surname><given-names>AE</given-names></name><name><surname>Barnes</surname><given-names>KC</given-names></name><name><surname>Boerwinkle</surname><given-names>E</given-names></name><name><surname>Gabriel</surname><given-names>S</given-names></name><name><surname>Gibbs</surname><given-names>R</given-names></name><name><surname>Rice</surname><given-names>KM</given-names></name><name><surname>Rich</surname><given-names>SS</given-names></name><name><surname>Silverman</surname><given-names>EK</given-names></name><name><surname>Qasba</surname><given-names>P</given-names></name><name><surname>Gan</surname><given-names>W</given-names></name><collab>NHLBI Trans-Omics for Precision Medicine (TOPMed) Consortium</collab><name><surname>Papanicolaou</surname><given-names>GJ</given-names></name><name><surname>Nickerson</surname><given-names>DA</given-names></name><name><surname>Browning</surname><given-names>SR</given-names></name><name><surname>Zody</surname><given-names>MC</given-names></name><name><surname>Zöllner</surname><given-names>S</given-names></name><name><surname>Wilson</surname><given-names>JG</given-names></name><name><surname>Cupples</surname><given-names>LA</given-names></name><name><surname>Laurie</surname><given-names>CC</given-names></name><name><surname>Jaquish</surname><given-names>CE</given-names></name><name><surname>Hernandez</surname><given-names>RD</given-names></name><name><surname>O’Connor</surname><given-names>TD</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Sequencing of 53,831 diverse genomes from the NHLBI TOPMed Program</article-title><source>Nature</source><volume>590</volume><fpage>290</fpage><lpage>299</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03205-y</pub-id><pub-id pub-id-type="pmid">33568819</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Van Hout</surname><given-names>CV</given-names></name><name><surname>Tachmazidou</surname><given-names>I</given-names></name><name><surname>Backman</surname><given-names>JD</given-names></name><name><surname>Hoffman</surname><given-names>JD</given-names></name><name><surname>Liu</surname><given-names>D</given-names></name><name><surname>Pandey</surname><given-names>AK</given-names></name><name><surname>Gonzaga-Jauregui</surname><given-names>C</given-names></name><name><surname>Khalid</surname><given-names>S</given-names></name><name><surname>Ye</surname><given-names>B</given-names></name><name><surname>Banerjee</surname><given-names>N</given-names></name><name><surname>Li</surname><given-names>AH</given-names></name><name><surname>O’Dushlaine</surname><given-names>C</given-names></name><name><surname>Marcketta</surname><given-names>A</given-names></name><name><surname>Staples</surname><given-names>J</given-names></name><name><surname>Schurmann</surname><given-names>C</given-names></name><name><surname>Hawes</surname><given-names>A</given-names></name><name><surname>Maxwell</surname><given-names>E</given-names></name><name><surname>Barnard</surname><given-names>L</given-names></name><name><surname>Lopez</surname><given-names>A</given-names></name><name><surname>Penn</surname><given-names>J</given-names></name><name><surname>Habegger</surname><given-names>L</given-names></name><name><surname>Blumenfeld</surname><given-names>AL</given-names></name><name><surname>Bai</surname><given-names>X</given-names></name><name><surname>O’Keeffe</surname><given-names>S</given-names></name><name><surname>Yadav</surname><given-names>A</given-names></name><name><surname>Praveen</surname><given-names>K</given-names></name><name><surname>Jones</surname><given-names>M</given-names></name><name><surname>Salerno</surname><given-names>WJ</given-names></name><name><surname>Chung</surname><given-names>WK</given-names></name><name><surname>Surakka</surname><given-names>I</given-names></name><name><surname>Willer</surname><given-names>CJ</given-names></name><name><surname>Hveem</surname><given-names>K</given-names></name><name><surname>Leader</surname><given-names>JB</given-names></name><name><surname>Carey</surname><given-names>DJ</given-names></name><name><surname>Ledbetter</surname><given-names>DH</given-names></name><collab>Geisinger-Regeneron DiscovEHR Collaboration</collab><name><surname>Cardon</surname><given-names>L</given-names></name><name><surname>Yancopoulos</surname><given-names>GD</given-names></name><name><surname>Economides</surname><given-names>A</given-names></name><name><surname>Coppola</surname><given-names>G</given-names></name><name><surname>Shuldiner</surname><given-names>AR</given-names></name><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Cantor</surname><given-names>M</given-names></name><collab>Regeneron Genetics Center</collab><name><surname>Nelson</surname><given-names>MR</given-names></name><name><surname>Whittaker</surname><given-names>J</given-names></name><name><surname>Reid</surname><given-names>JG</given-names></name><name><surname>Marchini</surname><given-names>J</given-names></name><name><surname>Overton</surname><given-names>JD</given-names></name><name><surname>Scott</surname><given-names>RA</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name><name><surname>Yerges-Armstrong</surname><given-names>L</given-names></name><name><surname>Baras</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Exome sequencing and characterization of 49,960 individuals in the UK Biobank</article-title><source>Nature</source><volume>586</volume><fpage>749</fpage><lpage>756</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2853-0</pub-id><pub-id pub-id-type="pmid">33087929</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Vöhringer</surname><given-names>H</given-names></name><name><surname>van Hoeck</surname><given-names>A</given-names></name><name><surname>Cuppen</surname><given-names>E</given-names></name><name><surname>Gerstung</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Learning Mutational Signatures and Their Multidimensional Genomic Properties with TensorSignatures</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/850453</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weghorn</surname><given-names>D</given-names></name><name><surname>Balick</surname><given-names>DJ</given-names></name><name><surname>Cassa</surname><given-names>C</given-names></name><name><surname>Kosmicki</surname><given-names>JA</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Beier</surname><given-names>DR</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Applicability of the Mutation-Selection Balance Model to Population Genetics of Heterozygous Protein-Truncating Variants in Humans</article-title><source>Molecular Biology and Evolution</source><volume>36</volume><fpage>1701</fpage><lpage>1710</lpage><pub-id pub-id-type="doi">10.1093/molbev/msz092</pub-id><pub-id pub-id-type="pmid">31004148</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williamson</surname><given-names>SH</given-names></name><name><surname>Hernandez</surname><given-names>R</given-names></name><name><surname>Fledel-Alon</surname><given-names>A</given-names></name><name><surname>Zhu</surname><given-names>L</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Bustamante</surname><given-names>CD</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Simultaneous inference of selection and population growth from patterns of variation in the human genome</article-title><source>PNAS</source><volume>102</volume><fpage>7882</fpage><lpage>7887</lpage><pub-id pub-id-type="doi">10.1073/pnas.0502300102</pub-id><pub-id pub-id-type="pmid">15905331</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yi</surname><given-names>X</given-names></name><name><surname>Liang</surname><given-names>Y</given-names></name><name><surname>Huerta-Sanchez</surname><given-names>E</given-names></name><name><surname>Jin</surname><given-names>X</given-names></name><name><surname>Cuo</surname><given-names>ZXP</given-names></name><name><surname>Pool</surname><given-names>JE</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Jiang</surname><given-names>H</given-names></name><name><surname>Vinckenbosch</surname><given-names>N</given-names></name><name><surname>Korneliussen</surname><given-names>TS</given-names></name><name><surname>Zheng</surname><given-names>H</given-names></name><name><surname>Liu</surname><given-names>T</given-names></name><name><surname>He</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>K</given-names></name><name><surname>Luo</surname><given-names>R</given-names></name><name><surname>Nie</surname><given-names>X</given-names></name><name><surname>Wu</surname><given-names>H</given-names></name><name><surname>Zhao</surname><given-names>M</given-names></name><name><surname>Cao</surname><given-names>H</given-names></name><name><surname>Zou</surname><given-names>J</given-names></name><name><surname>Shan</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>Q</given-names></name><name><surname>Ni</surname><given-names>P</given-names></name><name><surname>Tian</surname><given-names>G</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>X</given-names></name><name><surname>Jiang</surname><given-names>T</given-names></name><name><surname>Wu</surname><given-names>R</given-names></name><name><surname>Zhou</surname><given-names>G</given-names></name><name><surname>Tang</surname><given-names>M</given-names></name><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>T</given-names></name><name><surname>Feng</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>G</given-names></name><name><surname>Luosang</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Chen</surname><given-names>F</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Zheng</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name><name><surname>Bianba</surname><given-names>Z</given-names></name><name><surname>Yang</surname><given-names>G</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Tang</surname><given-names>S</given-names></name><name><surname>Gao</surname><given-names>G</given-names></name><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Luo</surname><given-names>Z</given-names></name><name><surname>Gusang</surname><given-names>L</given-names></name><name><surname>Cao</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>Q</given-names></name><name><surname>Ouyang</surname><given-names>W</given-names></name><name><surname>Ren</surname><given-names>X</given-names></name><name><surname>Liang</surname><given-names>H</given-names></name><name><surname>Zheng</surname><given-names>H</given-names></name><name><surname>Huang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Bolund</surname><given-names>L</given-names></name><name><surname>Kristiansen</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>R</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Sequencing of 50 human exomes reveals adaptation to high altitude</article-title><source>Science</source><volume>329</volume><fpage>75</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.1126/science.1190371</pub-id><pub-id pub-id-type="pmid">20595611</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><table-wrap id="app1table1" position="float"><label>Appendix 1—table 1.</label><caption><title>List of data sources.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Annotation type</th><th align="left" valign="bottom">Source</th></tr></thead><tbody><tr><td align="left" valign="bottom">Exon coordinates</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="http://ftp.ebi.ac.uk/pub/databases/gencode/Gencode_human/release_19/gencode.v19.annotation.gtf.gz">http://ftp.ebi.ac.uk/pub/databases/gencode/Gencode_human/release_19/gencode.v19.annotation.gtf.gz</ext-link></td></tr><tr><td align="left" valign="bottom">Exon annotations</td><td align="left" valign="bottom">Variant Effect Predictor (VEP) v87 using Gencode v19Ranks:<ext-link ext-link-type="uri" xlink:href="https://m.ensembl.org/info/genome/variation/prediction/predicted_data.html">https://m.ensembl.org/info/genome/variation/prediction/predicted_data.html</ext-link></td></tr><tr><td align="left" valign="bottom">WGS covered regions and exome target regions (gnomAD v2.1.1)</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://gnomad.broadinstitute.org/downloads">https://gnomad.broadinstitute.org/downloads</ext-link></td></tr><tr><td align="left" valign="bottom">Exome target regions(UK Biobank)</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://biobank.ndph.ox.ac.uk/ukb/ukb/auxdata/xgen_plus_spikein.GRCh38.bed">https://biobank.ndph.ox.ac.uk/ukb/ukb/auxdata/xgen_plus_spikein.GRCh38.bed</ext-link> (liftovered to hg19)</td></tr><tr><td align="left" valign="bottom">CpG methylation Testis</td><td align="left" valign="bottom">GEO Accession GSM1127119 (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/">https://www.ncbi.nlm.nih.gov/geo/</ext-link>)</td></tr><tr><td align="left" valign="bottom">CpG methylation Ovary</td><td align="left" valign="bottom">GEO Accession GSM1010980 (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/">https://www.ncbi.nlm.nih.gov/geo/</ext-link>)</td></tr><tr><td align="left" valign="bottom">CADD</td><td align="left" valign="bottom">CADD v1.4 (<ext-link ext-link-type="uri" xlink:href="https://cadd.gs.washington.edu/download">https://cadd.gs.washington.edu/download</ext-link>)</td></tr><tr><td align="left" valign="bottom">B-statistic</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pgen.1000471">https://doi.org/10.1371/journal.pgen.1000471</ext-link> (lifted over to hg19)</td></tr><tr><td align="left" valign="bottom">Functional site annotations</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/refseq/H_sapiens/mRNA_Prot/">https://ftp.ncbi.nlm.nih.gov/refseq/H_sapiens/mRNA_Prot/</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://www.prot2hg.com">https://www.prot2hg.com</ext-link></td></tr><tr><td align="left" valign="bottom">De novo mutations</td><td align="left" valign="bottom">Decode: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1126/science.aau1043">https://doi.org/10.1126/science.aau1043</ext-link> (<underline>Data S5</underline>)DDD: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41586-020-2832-5">https://doi.org/10.1038/s41586-020-2832-5</ext-link> (<underline>Supp. Table 1</underline>)</td></tr><tr><td align="left" valign="bottom">Polymorphism data</td><td align="left" valign="bottom">gnomAD: <ext-link ext-link-type="uri" xlink:href="https://gnomad.broadinstitute.org/downloads">https://gnomad.broadinstitute.org/downloads</ext-link>UK Biobank: <ext-link ext-link-type="uri" xlink:href="https://biobank.ctsu.ox.ac.uk/showcase/field.cgi?id=23155">https://biobank.ctsu.ox.ac.uk/showcase/field.cgi?id=23155</ext-link>DiscovEHR: <ext-link ext-link-type="uri" xlink:href="http://www.discovehrshare.com/downloads">http://www.discovehrshare.com/downloads</ext-link>1,000 Genomes: <ext-link ext-link-type="uri" xlink:href="ftp://ftp.1000genomes.ebi.ac.uk/vol1/ftp/release/20130502/">ftp://ftp.1000genomes.ebi.ac.uk/vol1/ftp/release/20130502/</ext-link></td></tr><tr><td align="left" valign="bottom">ClinVar</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/pub/clinvar/vcf_GRCh37/clinvar.vcf.gz">https://ftp.ncbi.nlm.nih.gov/pub/clinvar/vcf_GRCh37/clinvar.vcf.gz</ext-link></td></tr></tbody></table></table-wrap></app></app-group></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.71513.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Ross-Ibarra</surname><given-names>Jeffrey</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, Davis</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="box1"><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Mutation saturation for fitness effects at human CpG sites&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, one of whom is a member of our Board of Reviewing Editors, and the evaluation has been overseen by Patricia Wittkopp as the Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission. All three reviewers were very enthusiastic about the manuscript, and with some revision it will clearly be suitable for publication in <italic>eLife</italic>.</p><p>The main revision reviewers agreed would be important to incorporate is the issue of multiple testing (highlighted by Reviewer #2).</p><p>While all the reviewers were enthusiastic about the manuscript as written, and none felt that additional analyses were necessary, there were a number of additional analyses suggested that reviewers felt could strengthen the paper if they were relatively straightforward to incorporate. These included:</p><p>1) Comparison of the nonsynonymous genome-wide DFE from smaller samples to the 780K, perhaps only focusing on methylated nonsynomous sites.</p><p>2) Comparison of overlap between these invariant sites and Clinvar or databases of patients with known developmental disorders.</p><p>3) Comparison of results under a different parameterization of human demography.</p><p>Reviewers had a number of additional suggestions that we felt would improve the manuscript. Perhaps first among these was a general feeling of &quot;to be continued&quot; in the discussion with multiple citations to another manuscript in prep. While reviewers were not against this strategy by any means, efforts to address the novelty of the current paper in the discussion (rather than simply hinting at exciting results in the forthcoming work) would be worthwhile.</p><p>Please do review and respond to the individual reviewer comments as well.</p><p><italic>Reviewer #1:</italic></p><p>Agarwal and Przeworski have performed a very timely and interesting study of the distribution of fitness effects (DFE) of new mutations. This study is timely because modern human population genetic datasets have finally achieved sample sizes for which a certain class of nucleotide sites, i.e. methylated CpG sites, when neutrally evolving, should approach near complete polymorphism saturation. If every neutral site is expected to carry at least one variant, the classic problem of distinguishing sites that are monomorphic due to chance (no mutation) versus sites that are monomorphic due to selective constraint (removed mutations) is greatly simplified. The point at which population genomic datasets are saturated with polymorphisms should represent a major advance in understanding the DFE at individual sites and is what immediately piqued my interest.</p><p>Overall, this manuscript is a thorough and thoughtful examination of this topic; to my enjoyment there were several times where a question came to mind that was addressed shortly later in the paper. I believe the authors have made a compelling case for why methylated CpG sites provide an entry point for understanding the site-specific DFE. I found the section &quot;Interpreting monomorphic and polymorphic sites in current reference databases&quot; particularly insightful as a guide to thinking about future datasets; similarly, I thought the comparison with CADD scores (Figure S9) provided important food for thought regarding confounders to maps of constraint generated from vast numbers of species using modern genomic datasets.</p><p>While the study is addressing an interesting topic, I also felt this manuscript was limited in novel findings to take away. Certainly the study clearly shows that substitution saturation is achieved at synonymous CpG sites. However, subsequent main analyses do not really show anything new: the depletion of segregating sites in functional versus neutral categories (Figure 2) has been extensively shown in the literature and polymorphism saturation is not a necessary condition for observing this pattern. Similarly, the diminishing returns on sampling new variable sites has been shown in previous studies, for example the first &quot;large&quot; human datasets ca. 2012 (e.g. Figure 2 in Nelson et al., 2012, Science) have similar depictions as Figure 3B although with smaller sample sizes and different approaches (projection vs simulation in this study). There are some simulations presented in Figure 4, but this is more of a hypothetical representation of the site-specific DFE under simulation conditions roughly approximating human demography than formal inference on single sites. Again, these all describe the state of the field quite well, but I was disappointed by the lack of a novel finding derived from exploiting the mutation saturation properties at methylated CpG sites.</p><p>Similarly, I felt the authors posed a very important point about limitations of DFE inference methods in the Introduction but ended up not really providing any new insights into this problem. The authors argue (rightly so) that currently available DFE estimates are limited by both the sparsity of polymorphisms and limited flexibility in parametric forms of the DFE. However, the nonsynonymous human DFE estimates in the literature appear to be surprisingly robust to sample size: older estimates (Eyre-Walker et al., 2006 Genetics, Boyko et al., 2008 PLOS Genetics) seem to at least be somewhat consistent with newer estimates (assuming the same mutation rate) from samples that are orders of magnitude larger (Kim et al., 2017 Genetics). Whether a DFE inferred under polymorphism saturation conditions with different methods is different, and how it is different, is an issue of broad and immediate relevance to all those conducting population genomic simulations involving purifying selection. The analyses presented as Figure 4A and 4B kind of show this, but they are more a demonstration of what information one might have at 1M+ sample sizes rather than an analysis of whether genome-wide nonsynonymous DFE estimates are accurate. In other words, this manuscript makes it clear that a problem exists, that it is a fundamental and important problem in population genetics, and that with modern datasets we are now poised to start addressing this problem with some types of sites, but all of this is already very well-appreciated except for perhaps the last point.</p><p>At least a crude analysis to directly compare the nonsynonymous genome-wide DFE from smaller samples to the 780K sample would be helpful, but it should be noted that these kinds of analyses could be well beyond the scope of the current manuscript. For example, if methylated nonsynonymous CpG sites are under a different level of constraint than other nonsynonymous sites (Figure S14) then comparing results to a genome-wide nonsynonymous DFE might not make sense and any new analysis would have to try and infer a DFE independently from synonymous/nonsynonymous methylated CpG sites.</p><p>Abstract: where it says &quot;Here, we focus on putatively-neutral, synonymous CpG sites…&quot; I thought the phrase &quot;putatively-neutral, synonymous&quot; could be clearer to the reader if moved to &quot;… not seeing a polymorphism [at putatively-neutral, synonymous sites] is indicative of strong…&quot;.</p><p>Page 3 – &quot;DNM&quot; and &quot;FET&quot; were not defined before the first usage of the acronyms.</p><p>Page 7 – &quot;That synonymous sites are close to saturation…&quot;: Here, wouldn't the expected length of the genealogy such that 1 mutation is expected per synonymous CpG site be a pretty drastic underestimate of the length of the genealogy such that saturation is observed (99% of synonymous CpG sites w/mutation)? Wouldn't a more precise estimate be something like 39 million generations, [1-Pois(0|1.17e-7*39e6)] ~ 99% of sites?</p><p><italic>Reviewer #2:</italic></p><p>This manuscript presents a simple and elegant argument that neutrally evolving CpG sites are now mutationally saturated, with each having a 99% probability of containing variation in modern datasets containing hundreds of thousands of exomes. The authors make a compelling argument that for CpG sites where mutations would create genic stop codons or impair DNA binding, about 20% of such mutations are strongly deleterious (likely impairing fitness by 5% or more). Although it is not especially novel to make such statements about the selective constraint acting on large classes of sites, the more novel aspect of this work is the strong site-by-site prediction it makes that most individual sites without variation in UK Biobank are likely to be under strong selection.</p><p>The authors rightly point out that since 99% of neutrally evolving CpG sites contain variation in the data they are looking at, a CpG site without variation is likely evolving under constraint with a p value significance of 0.01. However, a weakness of their argument is that they do not discuss the associated multiple testing problem-in other words, how likely is it that a given non synonymous CpG site is devoid of variation but actually not under strong selection? Since one of the most novel and useful deliverables of this paper is single-base-pair-resolution predictions about which sites are under selection, such a multiple testing correction would provide important &quot;error bars&quot; for evaluating how likely it is that an individual CpG site is actually constrained, not just the proportion of constrained sites within a particular functional category.</p><p>The paper provides a comparison of their functional predictions to CADD scores, an older machine-learning-based attempt at identifying site by site constraint at single base pair resolution. While this section is useful and informative, I would have liked to see a discussion of the degree to which the comparison might be circular due to CADD's reliance on information about which sites are and are not variable. I had trouble assessing this for myself given that CADD appears to have used genetic variation data available a few years ago, but obviously did not use the biobank scale datasets that were not available when that work was published.</p><p>Reading this paper left me excited about the possibility of examining individual invariant CpG sites and deducing how many of them are already associated with known disease phenotypes. I believe the paper does not mention how many of these invariant sites appear in Clinvar or in databases of patients with known developmental disorders, and I wondered how close to saturation disease gene databases might be given that individuals with developmental disorders are much more likely to have their exomes sequenced compared to healthy individuals. One could imagine some such analyses being relatively low hanging fruit that could strengthen the current paper, but the authors also make several reference to a companion paper in preparation that deals more directly with the problem of assessing clinical variant significance. This is a reasonable strategy, but it does give the Discussion section of the paper somewhat of a &quot;to be continued&quot; feel.</p><p>I think the paper could be strengthened by calculating the proportion of non-variable CpG sites in teach category are likely to be truly under constraint, making use of some kind of multiple testing correction. This would build upon the intuition that a non-variable CpG is likely functional with a non-corrected p value of 0.01.</p><p>My point about the possible circularity of comparison to CADD could be addressed with further discussion of the degree to which CADD is informed by patterns of human genetic variation and how incorporation of genetic variation into CADD scores might affect the conclusions of this section. As an additional point in the CADD section, it's not totally clear whether the statement &quot;Mean transition rates at methylated CpGs are similar across CADD deciles&quot; is based on de novo mutation data or some other data source.</p><p>Another addition that would add a lot to the paper, though is not strictly necessary, would be to comment on the overlap between sites identified as under selection by the current paper and sites where mutations are already annotated as clinically relevant or suspected to be so based on their occurrence in a disease cohort.</p><p><italic>Reviewer #3:</italic></p><p>Agarwal et al., combine a few well-known ideas in population genetics – diminishing returns in sampling new alleles with increasing sample size and the enrichment of invariant sites for sites under strong purifying selection – and point out the exciting result that sample sizes of modern human data sets are sufficiently large that, for highly mutable sites, saturation mutation has been reached. This is my favorite kind of result – one that is strikingly obvious in retrospect but that I had never considered (and probably wouldn't have). The manuscript is well written, and a number of my concerns or questions while reading were resolved directly by the authors later on. I have no major concerns, but a few potential suggestions that might strengthen the presentation.</p><p>The authors emphasize several times how important an accurate demographic model is. While we may be close to a solid demographic model for humans, this is certainly not the case for many other organisms. Yet we are not far off from sufficient sample sizes in a number of species to begin to reach saturation. I found myself wondering how different the results/inference would be under a different model of human demographic history. Though likely the results would be supplemental, it would be nice in the main text to be able to say something about whether results are qualitatively different under a somewhat different published model.</p><p>On a similar note, while a fixed hs simplifies much of the analysis, I wondered how results would differ for (1) completely recessive mutations and (2) under a distribution of dominance coefficients, especially one in which the most deleterious alleles were more recessive. Again, though I think it would strengthen the manuscript by no means do I feel this is a necessary addition, though some discussion of variation in dominance would be an easy and helpful add.</p><p>There's some discussion of population structure, but I also found myself wondering about GxE. That is, another reason a variant might be segregating is that it's conditionally neutral in some populations and only deleterious in a subset. I think no analysis to be done here, but perhaps some discussion?</p><p>Maybe I missed it, but I don't think the acronym DNM is explained anywhere. While it was fairly self-explanatory, I did have a moment of wondering whether it was methylation or mutation and can't hurt to be explicit.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.71513.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>All three reviewers were very enthusiastic about the manuscript, and with some revision it will clearly be suitable for publication in eLife.</p><p>The main revision reviewers agreed would be important to incorporate is the issue of multiple testing (highlighted by Reviewer #2).</p><p>While all the reviewers were enthusiastic about the manuscript as written, and none felt that additional analyses were necessary, there were a number of additional analyses suggested that reviewers felt could strengthen the paper if they were relatively straightforward to incorporate. These included:</p><p>1) Comparison of the nonsynonymous genome-wide DFE from smaller samples to the 780K, perhaps only focusing on methylated nonsynomous sites.</p><p>2) Comparison of overlap between these invariant sites and Clinvar or databases of patients with known developmental disorders.</p><p>3) Comparison of results under a different parameterization of human demography.</p><p>Reviewers had a number of additional suggestions that we felt would improve the manuscript. Perhaps first among these was a general feeling of &quot;to be continued&quot; in the discussion with multiple citations to another manuscript in prep. While reviewers were not against this strategy by any means, efforts to address the novelty of the current paper in the discussion (rather than simply hinting at exciting results in the forthcoming work) would be worthwhile.</p><p>Please do review and respond to the individual reviewer comments as well.</p></disp-quote><p>We greatly appreciate the reviewers’ enthusiasm for the manuscript and thank them for their helpful comments.</p><p>We agree with the reviewers that it is important to address the question of how likely it is that a given non-synonymous mCpG site is invariant in current samples but nonetheless neutral. We had intended for this question to be addressed by our Bayesian analysis for an individual site in Figure 4, which examines how one’s beliefs about selection should be updated after observing a site to be invariant (i.e., using Bayes odds), with the p-value analogy serving simply to point out what makes mutation saturation special. We apologize for our failure to make this explicit, and have clarified the following points in the text:</p><p>A) First, we have now added a discussion of false discovery rates (FDR) in our mutation saturation based test for whether a non-syn site is neutral. As we now state, given that 1.2% of neutral sites are invariant, and 7.4% of all non-syn sites are invariant, the FDR across all non-synonymous invariant sites is ~16% (1.2/7.4). Within just LOFs it is ~4% (1.2/27).</p><p>B) With our demographic model and given our choice of prior, the Bayes odds for an invariant mCpG site at the current sample size are 15:1 in favor of hs &gt; 0.5x10<sup>-3</sup>; thus there is a 1/16 chance an invariant site is not under “strong selection”.</p><p>We note that these odds depend on our prior and demographic model while the FDR calculation is based on the assumption that the null model is given by synonymous sites.</p><p>With regard to the three other points raised above, we have addressed (3) by showing results for the widely used model of Tennessen et al., (Tennessen et al., 2012), which was inferred from a relatively small sample size and thus did not detect the rapid exponential growth in population size towards the present; as expected, this model does not predict observed levels of variation as well and leads to different expectations about allele frequencies of deleterious alleles. In other words, and as expected, the results in Figure 4 depend on the demographic model choice.</p><p>We have also taken up suggestion (2) and now include an analysis of ClinVar and Deciphering Developmental Disorders (DDD) datasets. While both data sets are far from saturation, as expected, variants in those data sets, which are likely strongly deleterious, are enriched among invariant sites relative to segregating sites.</p><p>Regarding the analysis suggested in (1), we would argue we already know what will happen: given that small sample sizes carry very little information (see Figure 4a,b), the answer will depend greatly on the choice of parametric form for the DFE. That a priori choice is necessarily arbitrary, given that it is precisely what we are trying to learn about. Indeed, a recent preprint by Dukler et al., (Dukler et al., 2021) reports estimates of <italic>sh</italic> against amino-acid mutations inconsistent with those of Kim et al., (see p. 11 of their Discussion) and those of Kim et al., are in poor agreement with those of Boyko et al., and others (see Figure 4 in Kim et al., 2017).</p><p>Finally, we apologize for the impression that we postponed interesting analyses to a different paper, which is not a continuation of the current manuscript, but instead presents a complementary analysis of loss-of-function variation in human exomes and includes inference of a site-level DFE for such variants. We have now tried to make this clearer in the text, as well as adding a Figure on DDD and ClinVar and other analyses, as suggested.</p><disp-quote content-type="editor-comment"><p>Reviewer #1:</p><p>Agarwal and Przeworski have performed a very timely and interesting study of the distribution of fitness effects (DFE) of new mutations. This study is timely because modern human population genetic datasets have finally achieved sample sizes for which a certain class of nucleotide sites, i.e. methylated CpG sites, when neutrally evolving, should approach near complete polymorphism saturation. If every neutral site is expected to carry at least one variant, the classic problem of distinguishing sites that are monomorphic due to chance (no mutation) versus sites that are monomorphic due to selective constraint (removed mutations) is greatly simplified. The point at which population genomic datasets are saturated with polymorphisms should represent a major advance in understanding the DFE at individual sites and is what immediately piqued my interest.</p><p>Overall, this manuscript is a thorough and thoughtful examination of this topic; to my enjoyment there were several times where a question came to mind that was addressed shortly later in the paper. I believe the authors have made a compelling case for why methylated CpG sites provide an entry point for understanding the site-specific DFE. I found the section &quot;Interpreting monomorphic and polymorphic sites in current reference databases&quot; particularly insightful as a guide to thinking about future datasets; similarly, I thought the comparison with CADD scores (Figure S9) provided important food for thought regarding confounders to maps of constraint generated from vast numbers of species using modern genomic datasets.</p><p>While the study is addressing an interesting topic, I also felt this manuscript was limited in novel findings to take away. Certainly the study clearly shows that substitution saturation is achieved at synonymous CpG sites. However, subsequent main analyses do not really show anything new: the depletion of segregating sites in functional versus neutral categories (Figure 2) has been extensively shown in the literature and polymorphism saturation is not a necessary condition for observing this pattern.</p></disp-quote><p>We agree with the reviewer that many of the points raised were appreciated previously and did not mean to convey another impression. Our aim was instead to highlight some unique opportunities provided by being at or very near saturation for mCpG transitions. In that regard, we note that although depletion of variation in functional categories is to be expected at any sample size, the selection strength that this depletion reflects is very different in samples that are far from saturated, where invariant sites span the entire spectrum from neutral to lethal. Consider the depletion per functional category relative to synonymous sites in <xref ref-type="fig" rid="sa2fig1">Author response image 1</xref> in a sample of 100k: ~40% of mCpG LOF sites do not have T mutations. From Figure 4, it can be seen that these sites are associated with a much broader range of <italic>hs</italic> values than sites invariant at 780K, so that information about selection at an individual site is quite limited (indeed, in our p-value formulation, these sites would be assigned p≤0.35, see Figure 1). Thus, only now can we really start to tease apart weakly deleterious mutations from strongly deleterious or even embryonic lethal mutations. This allows us to identify individual sites that are most likely to underlie pathogenic mutations and functional categories that harbor deleterious variation at the extreme end of the spectrum of possible selection coefficients. More generally, saturation is useful because it allows one to learn about selection with many fewer untested assumptions than previously feasible.</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-sa2-fig1-v3.tif"/></fig><disp-quote content-type="editor-comment"><p>Similarly, the diminishing returns on sampling new variable sites has been shown in previous studies, for example the first &quot;large&quot; human datasets ca. 2012 (e.g. Figure 2 in Nelson et al., 2012, Science) have similar depictions as Figure 3B although with smaller sample sizes and different approaches (projection vs simulation in this study).</p></disp-quote><p>We agree completely: diminishing returns is expected on first principles from coalescent theory, which is why we cited a classic theory paper when making that point in the previous version of the manuscript. Nonetheless, the degree of saturation is an empirical question, since it depends on the unknown underlying demography of the recent past. In that regard, we note that Nelson et al., predict that at sample sizes of 400K chromosomes in Europeans, approximately 20% of all synonymous sites will be segregating at least one of three possible alleles, when the observed number is 29%. Regardless, not citing Nelson et al., 2012 was a clear oversight on our part, for which we apologize; we now cite it in that context and in mentioning the multiple merger coalescent.</p><disp-quote content-type="editor-comment"><p>There are some simulations presented in Figure 4, but this is more of a hypothetical representation of the site-specific DFE under simulation conditions roughly approximating human demography than formal inference on single sites. Again, these all describe the state of the field quite well, but I was disappointed by the lack of a novel finding derived from exploiting the mutation saturation properties at methylated CpG sites.</p></disp-quote><p>As noted above, in our view, the novelty of our results lies in their leveraging saturation in order to identify sites under extremely strong selection and make inferences about selection without the need to rely on strong, untested assumptions.</p><p>However, we note that Figure 4 is not simply a hypothetical representation, in that it shows the inferred DFE for single mCpG sites for a fixed mutation rate and given a plausible demographic model, given data summarized in terms of three ranges of allele frequency (i.e., = 0, between 1 and 10 copies, or above 10 copies). One could estimate a DFE across all sites from those summaries of the data (i.e., from the proportion of mCpG sites in each of the three frequency categories), by weighting the three densities in Figure 4 by those proportions. That is, in fact, what is done in a recent preprint by Dukler et al., (2021, BioRxiv): they infer the DFE from two summaries of the allele frequency spectrum (in bins of sites), the proportion of invariant sites and the proportion of alleles at 1-70 copies, in a sample of 70K chromosomes.</p><p>To illustrate how something similar could be done with Figure 4 based on individual sites, we obtain an estimate of the DFE for LOF mutations (shown in <xref ref-type="fig" rid="sa2fig2">Author response image 2</xref> Panel B and D for two different prior distributions on hs) by weighting the posterior densities in Panel A by the fraction of LOF mutations that are segregating (73% at 780K; 9% at 15K) and invariant (27% and 91% respectively); in panel C, we show the same for a different choice of prior. For the smaller sample size considered, the posterior distribution recapitulates the prior, because there is little information about selection in whether a site is observed to be segregating or invariant, and particularly about strong selection. In the sample of 780K, there is much more information about selection in a site being invariant and therefore, there is a shift towards stronger selection coefficients for LOF mutations regardless of the prior.</p><fig id="sa2fig2" position="float"><label>Author response image 2.</label><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-sa2-fig2-v3.tif"/></fig><p>Our goal was to highlight these points rather than infer a DFE using these two summaries, which throw out much of the information in the data (i.e., the allele frequency differences among segregating sites). In that regard, we note that the DFE inference would be improved by using the allele frequency at each of 1.1 million individual mCpG sites in the exome. We outline this next step in the Discussion but believe it is beyond the scope of our paper, as it is a project in itself--in particular it would require careful attention to robustness with regard to both the demographic model (and its impact on multiple hits), biased gene conversion and variability in mutation rates among mCpG sites. We now make these points explicitly in the Outlook.</p><disp-quote content-type="editor-comment"><p>Similarly, I felt the authors posed a very important point about limitations of DFE inference methods in the Introduction but ended up not really providing any new insights into this problem. The authors argue (rightly so) that currently available DFE estimates are limited by both the sparsity of polymorphisms and limited flexibility in parametric forms of the DFE. However, the nonsynonymous human DFE estimates in the literature appear to be surprisingly robust to sample size: older estimates (Eyre-Walker et al., 2006 Genetics, Boyko et al., 2008 PLOS Genetics) seem to at least be somewhat consistent with newer estimates (assuming the same mutation rate) from samples that are orders of magnitude larger (Kim et al., 2017 Genetics).</p></disp-quote><p>We are not quite sure what the reviewer has in mind by “somewhat consistent,” as Boyko et al., estimate that 35% of non-synonymous mutations have s&gt;10<sup>-2</sup> while Kim et al., find that proportion to be “0.38–0.84 fold lower” than the Boyko et al., estimate (see, e.g., Figure 4 in Kim et al., 2017). Moreover, the preprint by Dukler et al., mentioned above, which infers the DFE based on ~70K chromosomes, finds estimates inconsistent with those of Kim et al., (see SOM Table 2 and SOM Figure S5 in Dukler et al., 2021).</p><p>More generally, given that even 70K chromosomes carry little information about much of the distribution of selection coefficients (see our Figure 4), we expect that studies based on relatively sample sizes will basically recover something close to their prior; therefore, they should agree when they use the same or similar parametric forms for the distribution of selection coefficients and disagree otherwise. The dependence on that choice is nicely illustrated in Kim et al., who consider different choices and then perform inference on the same data set and with the same fixed mutation rate for exomes; depending on their choice anywhere between 5%-28% of non-synonymous changes are inferred to be under strong selection with s&gt;=10<sup>-2</sup> (see their Table S4).</p><disp-quote content-type="editor-comment"><p>Whether a DFE inferred under polymorphism saturation conditions with different methods is different, and how it is different, is an issue of broad and immediate relevance to all those conducting population genomic simulations involving purifying selection. The analyses presented as Figure 4A and 4B kind of show this, but they are more a demonstration of what information one might have at 1M+ sample sizes rather than an analysis of whether genome-wide nonsynonymous DFE estimates are accurate. In other words, this manuscript makes it clear that a problem exists, that it is a fundamental and important problem in population genetics, and that with modern datasets we are now poised to start addressing this problem with some types of sites, but all of this is already very well-appreciated except for perhaps the last point.</p><p>At least a crude analysis to directly compare the nonsynonymous genome-wide DFE from smaller samples to the 780K sample would be helpful, but it should be noted that these kinds of analyses could be well beyond the scope of the current manuscript. For example, if methylated nonsynonymous CpG sites are under a different level of constraint than other nonsynonymous sites (Figure S14) then comparing results to a genome-wide nonsynonymous DFE might not make sense and any new analysis would have to try and infer a DFE independently from synonymous/nonsynonymous methylated CpG sites.</p></disp-quote><p>We are not sure what would be learned from this comparison, given that Figure 4 shows that, at least with an uninformative prior, there is little information about the true DFE in samples, even of tens of thousands of individuals. Thus, if some of the genome-wide nonsynonymous DFE estimates based on small sample sizes turn out to be accurate, it will be because the guess about the parametric shape of the DFE was an inspired one. In our view, that is certainly possible but not likely, given that the shape of the DFE is precisely what the field has been aiming to learn and, we would argue, what we are now finally in a position to do for CpG mutations in humans.</p><disp-quote content-type="editor-comment"><p>Abstract: where it says &quot;Here, we focus on putatively-neutral, synonymous CpG sites…&quot; I thought the phrase &quot;putatively-neutral, synonymous&quot; could be clearer to the reader if moved to &quot;… not seeing a polymorphism [at putatively-neutral, synonymous sites] is indicative of strong…&quot;.</p></disp-quote><p>We apologize for the unclear phrasing; we meant to say that given mutation saturation at putatively neutral sites, not seeing a <italic>non-synonymous polymorphism</italic> is indicative of strong selection against that mutation.We have now amended the text to make this clear.</p><disp-quote content-type="editor-comment"><p>Page 3 – &quot;DNM&quot; and &quot;FET&quot; were not defined before the first usage of the acronyms.</p></disp-quote><p>We have fixed this issue in the text.</p><disp-quote content-type="editor-comment"><p>Page 7 – &quot;That synonymous sites are close to saturation…&quot;: Here, wouldn't the expected length of the genealogy such that 1 mutation is expected per synonymous CpG site be a pretty drastic underestimate of the length of the genealogy such that saturation is observed (99% of synonymous CpG sites w/mutation)? Wouldn't a more precise estimate be something like 39 million generations, [1-Pois(0|1.17e-7*39e6)] ~ 99% of sites?</p></disp-quote><p>We thank the reviewer for pointing out that we were being imprecise and have now updated the text to follow the suggestion.</p><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>This manuscript presents a simple and elegant argument that neutrally evolving CpG sites are now mutationally saturated, with each having a 99% probability of containing variation in modern datasets containing hundreds of thousands of exomes. The authors make a compelling argument that for CpG sites where mutations would create genic stop codons or impair DNA binding, about 20% of such mutations are strongly deleterious (likely impairing fitness by 5% or more). Although it is not especially novel to make such statements about the selective constraint acting on large classes of sites, the more novel aspect of this work is the strong site-by-site prediction it makes that most individual sites without variation in UK Biobank are likely to be under strong selection.</p><p>The authors rightly point out that since 99% of neutrally evolving CpG sites contain variation in the data they are looking at, a CpG site without variation is likely evolving under constraint with a p value significance of 0.01. However, a weakness of their argument is that they do not discuss the associated multiple testing problem-in other words, how likely is it that a given non synonymous CpG site is devoid of variation but actually not under strong selection? Since one of the most novel and useful deliverables of this paper is single-base-pair-resolution predictions about which sites are under selection, such a multiple testing correction would provide important &quot;error bars&quot; for evaluating how likely it is that an individual CpG site is actually constrained, not just the proportion of constrained sites within a particular functional category.</p></disp-quote><p>We thank the reviewer for pointing this out. As we outline in response to the editorial comments, one way to think about this problem might be in terms of false discovery rates, in which case the FDR would be 16% across all non-synonymous mCpG sites that are invariant in current samples, and ~4% for the subset of those sites where mutations lead to loss-of-function of genes.</p><p>Another way to address this issue, which we had included but not emphasized previously, is by examining how one’s beliefs about selection should be updated after observing a site to be invariant (i.e., using Bayes odds). At current sample sizes and assuming our uninformative prior, for a non-synonymous mCpG site that does not have a C&gt;T mutation, the Bayes odds are 15:1 in favor of <italic>hs</italic>&gt;0.5x10<sup>-3</sup>; thus the chance that such a site is not under strong selection is 1/16, given our prior and demographic model.</p><p>As noted in response to the editor, these two approaches (FDR and Bayes odds) are based on somewhat distinct assumptions.</p><p>We have now added and/or emphasized these two points in the main text.</p><disp-quote content-type="editor-comment"><p>The paper provides a comparison of their functional predictions to CADD scores, an older machine-learning-based attempt at identifying site by site constraint at single base pair resolution. While this section is useful and informative, I would have liked to see a discussion of the degree to which the comparison might be circular due to CADD's reliance on information about which sites are and are not variable. I had trouble assessing this for myself given that CADD appears to have used genetic variation data available a few years ago, but obviously did not use the biobank scale datasets that were not available when that work was published.</p></disp-quote><p>We apologize for the lack of clarity in the presentation. We meant to emphasize that de novo <italic>mutation rates</italic> vary across CADD deciles when considering all CpG sites (Figure 2—figure supplement 5c), which confounds CADD precisely because it is based in part on which sites are variable. We now write:</p><p>“We can also check that the fraction of sites segregating is inversely proportional to the predicted functional importance of the sites using CADD scores (12), widely used measures of constraint that incorporate functional annotations and measures of conservation. Across deciles, mean de novo transition rates at methylated CpGs are similar (Figure 2—figure supplement 5a) and, as expected, the fraction of segregating sites decreases with increasing CADD scores (Figure 2—figure supplement 5b). We note, however, that mutation rates may not always be similar across comparison groups: considering all CpG sites in exons (i.e., not only highly methylated ones), for example, de novo mutation rates are much more variable across CADD deciles (Figure 2—figure supplement 5c). Consequently the depletion of segregating sites no longer has a simple interpretation (Figure 2—figure supplement 5d), instead reflecting a combination of differences in mutation rates and fitness effects. By implication, while CADD scores are meant to isolate the effects of selection, they will in some cases classify sites that have high mutation rates as less constrained, and vice versa.”</p><disp-quote content-type="editor-comment"><p>Reading this paper left me excited about the possibility of examining individual invariant CpG sites and deducing how many of them are already associated with known disease phenotypes. I believe the paper does not mention how many of these invariant sites appear in Clinvar or in databases of patients with known developmental disorders, and I wondered how close to saturation disease gene databases might be given that individuals with developmental disorders are much more likely to have their exomes sequenced compared to healthy individuals. One could imagine some such analyses being relatively low hanging fruit that could strengthen the current paper, but the authors also make several reference to a companion paper in preparation that deals more directly with the problem of assessing clinical variant significance. This is a reasonable strategy, but it does give the Discussion section of the paper somewhat of a &quot;to be continued&quot; feel.</p></disp-quote><p>We apologize for the confusion that arose from our references to a second manuscript in prep. The companion paper is not a continuation of the current manuscript: it contains an analysis of fitness and pathogenic effects of loss-of-function variation in human exomes.</p><p>Following the reviewer’s suggestion to address the clinical significance of our results, we have now examined the relationship of mCpG sites invariant in current samples with ClinVar variants. We find that of the approximately 59,000 non-synonymous mCpG sites that are invariant, only ~3.6% overlap with C&gt;T variants associated with at least one disease and classified as likely pathogenic in ClinVar (~5.8% if we include those classified as uncertain or with conflicting evidence as pathogenic). Approximately 2% of invariant mCpGs have C&gt;T mutations in what is, to our knowledge, the largest collection of de novo variants ascertained in ~35,000 individuals with developmental disorders (DDD, Kaplanis et al., 2020). At the level of genes, of the 10k genes that have at least one invariant non-synonymous mCpG, only 8% (11% including uncertain variants) have any non-synonymous hits in ClinVar, and ~8% in DDD. We think it highly unlikely that the large number of remaining invariant sites are not seen with mutations in these databases because such mutations are lethal; rather it seems to us to be the case that these disease databases are far from saturation as they contain variants from a relatively small number of individuals, are subject to various ascertainment biases both at the variant level and at the individual level, and only contain data for a small subset of existing severe diseases.</p><p>With a view to assessing clinical relevance however, we can ask a related question, namely how informative being invariant in a sample of 780K is about pathogenicity in ClinVar. Although the relationship between selection and pathogenicity is far from straightforward, being an invariant non-synonymous mCpG in current samples not only substantially increases (15-fold) the odds of hs &gt; 0.5x10<sup>-3</sup> (see Figure 4b), it also increases the odds of being classified as pathogenic vs. benign in ClinVar 8-51 fold. In the DDD sample, we don’t know which variants are pathogenic; however, if we consider non-synonymous mutations that occur in consensus DDD genes as pathogenic (a standard diagnostic criterion), being invariant increases the odds of being classified as pathogenic 6-fold. We caution that both ClinVar classifications and the identification of consensus genes in DDD relies in part on whether a site is segregating in datasets like ExAC, so this exercise is somewhat circular. Nonetheless it illustrates that there is some information about clinical importance in mCpG sites that are invariant in current samples, and that the degree of enrichment (6 to 51-fold) is very roughly on par with the Bayes odds that we estimate of strong selection conditional on a site being invariant. We have added these findings to the main text and added the plot as Figure 4—figure supplement 2.</p><disp-quote content-type="editor-comment"><p>I think the paper could be strengthened by calculating the proportion of non-variable CpG sites in teach category are likely to be truly under constraint, making use of some kind of multiple testing correction. This would build upon the intuition that a non-variable CpG is likely functional with a non-corrected p value of 0.01.</p></disp-quote><p>As we write above, we now address this concern in the paper in two ways: using false discovery rates and Bayes odds of <italic>hs</italic>&gt;0.5x10<sup>-3</sup>.</p><disp-quote content-type="editor-comment"><p>My point about the possible circularity of comparison to CADD could be addressed with further discussion of the degree to which CADD is informed by patterns of human genetic variation and how incorporation of genetic variation into CADD scores might affect the conclusions of this section. As an additional point in the CADD section, it's not totally clear whether the statement &quot;Mean transition rates at methylated CpGs are similar across CADD deciles&quot; is based on de novo mutation data or some other data source.</p></disp-quote><p>We apologize for the lack of clarity in the presentation but meant to emphasize that de novo mutation rates vary across all CpG sites by CADD score, which confounds CADD precisely because it is based in part on which sites are variable. We have now revised the text to hopefully clarify this point (see response to reviewer 1).</p><disp-quote content-type="editor-comment"><p>Another addition that would add a lot to the paper, though is not strictly necessary, would be to comment on the overlap between sites identified as under selection by the current paper and sites where mutations are already annotated as clinically relevant or suspected to be so based on their occurrence in a disease cohort.</p></disp-quote><p>We thank the reviewer for this suggestion. As detailed in our response above, we now show that CpG sites invariant in population cohorts that overlap with ClinVar and among de novo variants in the DDD cohort are informative about pathogenicity of variants. We have added Figure 4—figure supplement 2 to illustrate this point.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>Agarwal et al., combine a few well-known ideas in population genetics – diminishing returns in sampling new alleles with increasing sample size and the enrichment of invariant sites for sites under strong purifying selection – and point out the exciting result that sample sizes of modern human data sets are sufficiently large that, for highly mutable sites, saturation mutation has been reached. This is my favorite kind of result – one that is strikingly obvious in retrospect but that I had never considered (and probably wouldn't have). The manuscript is well written, and a number of my concerns or questions while reading were resolved directly by the authors later on. I have no major concerns, but a few potential suggestions that might strengthen the presentation.</p><p>The authors emphasize several times how important an accurate demographic model is. While we may be close to a solid demographic model for humans, this is certainly not the case for many other organisms. Yet we are not far off from sufficient sample sizes in a number of species to begin to reach saturation. I found myself wondering how different the results/inference would be under a different model of human demographic history. Though likely the results would be supplemental, it would be nice in the main text to be able to say something about whether results are qualitatively different under a somewhat different published model.</p></disp-quote><p>We had previously examined the effect of a few demographic scenarios with large increases in population size towards the present on the average length of the genealogy of a sample (and hence the expected number of mutations at a site) in Figure 3—figure supplement 1b, but without quantifying the effect on our selection inference. Following this suggestion, we now consider a widely used model of human demography inferred from a relatively small sample, and therefore not powered to detect the huge increase in population size towards the present (Tennessen et al., 2012). Using this model, we find a poor fit to the proportion of segregating CpG sites (the observed fraction is 99% in 780K exomes, when the model predicts 49%). Also, as expected, inferences about selection depend on the accuracy of the demographic model (as can be seen by comparing <xref ref-type="fig" rid="sa2fig3">Author response image 3</xref> panel B to Figure 4B in the main text).</p><fig id="sa2fig3" position="float"><label>Author response image 3.</label><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71513-sa2-fig3-v3.tif"/></fig><disp-quote content-type="editor-comment"><p>On a similar note, while a fixed hs simplifies much of the analysis, I wondered how results would differ for (1) completely recessive mutations and (2) under a distribution of dominance coefficients, especially one in which the most deleterious alleles were more recessive. Again, though I think it would strengthen the manuscript by no means do I feel this is a necessary addition, though some discussion of variation in dominance would be an easy and helpful add.</p><p>There's some discussion of population structure, but I also found myself wondering about GxE. That is, another reason a variant might be segregating is that it's conditionally neutral in some populations and only deleterious in a subset. I think no analysis to be done here, but perhaps some discussion?</p></disp-quote><p>We agree that our analysis ignores the possibilities of complete recessivity in fitness (<italic>h</italic>=0) as well as more complicated selection scenarios, such as spatially-varying selection (of the type that might be induced by GxE). We note however that so long as there are any fitness effects in heterozygotes, the allele dynamics will be primarily governed by <italic>hs;</italic> one might also imagine that under some conditions, the mean selection effect across environments would predict allele dynamics reasonably well even in the presence of GxE. Also worth exploring in our view is the standard assumption that <italic>hs</italic> remains fixed even as <italic>N</italic><sub>e</sub> changes dramatically. We now mention these points in the Outlook.</p><disp-quote content-type="editor-comment"><p>Maybe I missed it, but I don't think the acronym DNM is explained anywhere. While it was fairly self-explanatory, I did have a moment of wondering whether it was methylation or mutation and can't hurt to be explicit.</p></disp-quote><p>We apologize for the oversight and have updated the text accordingly.</p></body></sub-article></article>