<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">76065</article-id><article-id pub-id-type="doi">10.7554/eLife.76065</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Broad-scale variation in human genetic diversity levels is predicted by purifying selection on coding and non-coding elements</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-262786"><name><surname>Murphy</surname><given-names>David A</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0715-3355</contrib-id><email>david-murphy@omrf.org</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-18497"><name><surname>Elyashiv</surname><given-names>Eyal</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-123977"><name><surname>Amster</surname><given-names>Guy</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9108-5200</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf3"/></contrib><contrib contrib-type="author" corresp="yes" id="author-19148"><name><surname>Sella</surname><given-names>Guy</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5239-7930</contrib-id><email>gs2747@columbia.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hj8s172</institution-id><institution>Department of Biological Sciences, Columbia University</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/035z6xf33</institution-id><institution>Genes and Human Disease Research Program, Oklahoma Medical Research Foundation, Oklahoma City</institution></institution-wrap><addr-line><named-content content-type="city">Oklahoma City</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>MyHeritage</institution><addr-line><named-content content-type="city">Or Yehuda</named-content></addr-line><country>Israel</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0508h6p74</institution-id><institution>Flatiron Health Inc</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hj8s172</institution-id><institution>Program for Mathematical Genomics, Columbia University</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Nordborg</surname><given-names>Magnus</given-names></name><role>Reviewing Editor</role><aff><institution>Gregor Mendel Institute</institution><country>Austria</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>23</day><month>06</month><year>2023</year></pub-date><pub-date pub-type="collection"><year>2023</year></pub-date><volume>12</volume><elocation-id>e76065</elocation-id><history><date date-type="received" iso-8601-date="2021-12-03"><day>03</day><month>12</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2022-08-22"><day>22</day><month>08</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2021-07-02"><day>02</day><month>07</month><year>2021</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2021.07.02.450762"/></event></pub-history><permissions><copyright-statement>© 2023, Murphy et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Murphy et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-76065-v2.pdf"/><abstract><p>Analyses of genetic variation in many taxa have established that neutral genetic diversity is shaped by natural selection at linked sites. Whether the mode of selection is primarily the fixation of strongly beneficial alleles (selective sweeps) or purifying selection on deleterious mutations (background selection) remains unknown, however. We address this question in humans by fitting a model of the joint effects of selective sweeps and background selection to autosomal polymorphism data from the 1000 Genomes Project. After controlling for variation in mutation rates along the genome, a model of background selection alone explains ~60% of the variance in diversity levels at the megabase scale. Adding the effects of selective sweeps driven by adaptive substitutions to the model does not improve the fit, and when both modes of selection are considered jointly, selective sweeps are estimated to have had little or no effect on linked neutral diversity. The regions under purifying selection are best predicted by phylogenetic conservation, with ~80% of the deleterious mutations affecting neutral diversity occurring in non-exonic regions. Thus, background selection is the dominant mode of linked selection in humans, with marked effects on diversity levels throughout autosomes.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>neutral diversity</kwd><kwd>positive selection</kwd><kwd>purifying selection</kwd><kwd>Selective sweeps</kwd><kwd>Background selection</kwd><kwd>demographic history</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM115889</award-id><principal-award-recipient><name><surname>Sella</surname><given-names>Guy</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32GM008798</award-id><principal-award-recipient><name><surname>Murphy</surname><given-names>David A</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Background selection is shown to be the dominant mode of linked selection in humans, with marked effects on diversity levels throughout autosomes.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Selection at a given locus in the genome affects diversity levels at sites linked to it (<xref ref-type="bibr" rid="bib48">Hill and Robertson, 1966</xref>; <xref ref-type="bibr" rid="bib102">Smith and Haigh, 1974</xref>; <xref ref-type="bibr" rid="bib57">Kaplan et al., 1989</xref>; <xref ref-type="bibr" rid="bib8">Begun and Aquadro, 1992</xref>; <xref ref-type="bibr" rid="bib17">Charlesworth et al., 1993</xref>; <xref ref-type="bibr" rid="bib53">Hudson and Kaplan, 1995</xref>; <xref ref-type="bibr" rid="bib74">Nordborg et al., 1996</xref>; <xref ref-type="bibr" rid="bib18">Charlesworth, 2013</xref>; <xref ref-type="bibr" rid="bib27">Cutter and Payseur, 2013</xref>). When a new, strongly beneficial mutation increases in frequency to fixation in the population, it carries with it the haplotype on which it arose, thus reducing levels of neutral diversity nearby, in what is sometimes called a ‘hard selective sweep’ (<xref ref-type="bibr" rid="bib102">Smith and Haigh, 1974</xref>; <xref ref-type="bibr" rid="bib57">Kaplan et al., 1989</xref>). ‘Soft sweeps’, particularly those in which an allele segregates at low frequency before becoming beneficial and sweeping to fixation, and ‘partial sweeps’, in which a beneficial mutation rapidly increases to an intermediate frequency, also reduce neutral diversity levels near the selected sites (<xref ref-type="bibr" rid="bib46">Hermisson and Pennings, 2005</xref>; <xref ref-type="bibr" rid="bib86">Przeworski et al., 2005</xref>; <xref ref-type="bibr" rid="bib79">Pennings and Hermisson, 2006a</xref>; <xref ref-type="bibr" rid="bib80">Pennings and Hermisson, 2006b</xref>; <xref ref-type="bibr" rid="bib25">Coop and Ralph, 2012</xref>; <xref ref-type="bibr" rid="bib11">Berg and Coop, 2015</xref>). Similarly, when deleterious mutations are eliminated from the population by selection, so are the haplotypes on which they lie. This process too reduces diversity levels near selected sites, in a phenomenon known as ‘background selection’ (<xref ref-type="bibr" rid="bib17">Charlesworth et al., 1993</xref>; <xref ref-type="bibr" rid="bib53">Hudson and Kaplan, 1995</xref>; <xref ref-type="bibr" rid="bib74">Nordborg et al., 1996</xref>; <xref ref-type="bibr" rid="bib20">Comeron and Kreitman, 2002</xref>; <xref ref-type="bibr" rid="bib39">Good et al., 2014</xref>; <xref ref-type="bibr" rid="bib28">Cvijović et al., 2018</xref>). Because the lengths of the haplotypes associated with selected alleles depend on the recombination rate, selection causes a greater reduction in levels of linked neutral genetic diversity in regions with lower rates of recombination or a greater density of selected sites. These predicted relationships have been observed in numerous taxa, including plants, <italic>Drosophila</italic>, rodents, and primates, establishing that the effects of linked selection are widespread (<xref ref-type="bibr" rid="bib8">Begun and Aquadro, 1992</xref>; <xref ref-type="bibr" rid="bib72">Nachman, 1997</xref>; <xref ref-type="bibr" rid="bib78">Payseur and Nachman, 2002</xref>; <xref ref-type="bibr" rid="bib75">Nordborg et al., 2005</xref>; <xref ref-type="bibr" rid="bib123">Wright et al., 2006</xref>; <xref ref-type="bibr" rid="bib2">Andolfatto, 2007</xref>; <xref ref-type="bibr" rid="bib9">Begun et al., 2007</xref>; <xref ref-type="bibr" rid="bib66">Macpherson et al., 2007</xref>; <xref ref-type="bibr" rid="bib124">Wright and Andolfatto, 2008</xref>; <xref ref-type="bibr" rid="bib16">Cai et al., 2009</xref>; <xref ref-type="bibr" rid="bib96">Sella et al., 2009</xref>; <xref ref-type="bibr" rid="bib27">Cutter and Payseur, 2013</xref>).</p><p>More recently, the advent of large genomic datasets and detailed functional annotations have made it possible to infer the effects of linked selection and build maps that predict levels of diversity along the genome (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>; <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>; also see <xref ref-type="bibr" rid="bib53">Hudson and Kaplan, 1995</xref>; <xref ref-type="bibr" rid="bib74">Nordborg et al., 1996</xref>; <xref ref-type="bibr" rid="bib21">Comeron, 2014</xref>). The first effort predated the availability of genome-wide resequencing data, relying instead on information about incomplete lineage sorting among human, chimpanzee and gorilla, which reflects variation in diversity levels along the genome in the common ancestor of humans and chimpanzees (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>). This pioneering paper showed that a model of background selection fits variation in human-chimpanzee divergence levels along the genome remarkably well, with only a few parameters.</p><p>What remained unclear is whether this remarkable fit should be attributed to the effects of background selection alone. Notably, the estimate of the rate of deleterious mutations underlying the effects of background selection was unrealistically high—substantially greater than the upper limit based on estimates of the total mutation rate per site in humans (<xref ref-type="bibr" rid="bib64">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>; Appendix 1 Section 5). In light of this finding, <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref> suggested that the model might be soaking up effects of other modes of selection, particularly those of selective sweeps (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>). Subsequent work indicated that selective sweeps had little effect on diversity levels in humans (<xref ref-type="bibr" rid="bib24">Coop et al., 2009</xref>; <xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>), however, with no more of a reduction in diversity around plausible targets of positive selection (nonsynonymous substitutions) than around sites assumed to be predominantly neutral (synonymous substitutions) (<xref ref-type="bibr" rid="bib24">Coop et al., 2009</xref>; <xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>). Yet, the interpretation of these findings was contested: it was suggested that on average, background selection causes more of a reduction in diversity around synonymous than nonsynonymous substitutions, and consequently that the comparison between the two types of sites may obscure the reduction due to sweeps around nonsynonymous substitutions (<xref ref-type="bibr" rid="bib33">Enard et al., 2014</xref>). The map of predicted background selection effects offered little help in evaluating this hypothesis, because it provided poor quantitative fits of diversity levels around both synonymous and nonsynonymous substitutions (<xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>). Thus, despite clear evidence for the impact of background selection, we still lack an understanding of its contribution relative to sweeps (<xref ref-type="bibr" rid="bib105">Stephan, 2010</xref>), as well as maps of their respective effects on human diversity levels.</p></sec><sec id="s2"><title>Results and disussion</title><sec id="s2-1"><title>Model and inference</title><p>Here we resolve these issues by considering the effects of background selection and selective sweeps on diversity levels jointly (<xref ref-type="fig" rid="fig1">Figure 1</xref> and Appendix 1 Section 1). We model the effects on the expected neutral heterozygosity (i.e., the probability of observing different alleles in a sample size of two) at a given autosomal position <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, as<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local mutation rate, <inline-formula><mml:math id="inf3"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the effective population size without linked selection, <inline-formula><mml:math id="inf4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local (multiplicative) reduction in effective population size due to background selection, and <inline-formula><mml:math id="inf5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local coalescence rate caused by selective sweeps (<xref ref-type="bibr" rid="bib119">Wiehe and Stephan, 1993</xref>; <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). This model can be understood by thinking about a pair of lineages backward in time and noting that, considering mutation vs. coalescence events, <inline-formula><mml:math id="inf6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the probability that a mutation occurs (at a rate <inline-formula><mml:math id="inf7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> per generation) before the pair coalesces, owing either to genetic drift (at a rate <inline-formula><mml:math id="inf8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>), which includes the effect of background selection, or to selective sweeps (at a rate <inline-formula><mml:math id="inf9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>) (<xref ref-type="bibr" rid="bib51">Hudson, 1990</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Modeling and inferring the effects of linked selection in humans.</title><p>Given the putative targets of selection and corresponding selection parameters (<bold>A and B</bold>), we calculate the expected neutral diversity levels along the genome (<bold>C</bold>). We infer the selection parameters by maximizing their composite likelihood given observed diversity levels (<bold>C</bold>). Based on these parameter estimates, we calculate a map of the expected effects of selection on linked diversity levels.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-fig1-v2.tif"/></fig><p>We model the effects of background selection, <inline-formula><mml:math id="inf10"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, as a function of genetic distance from regions that may be under purifying selection (<xref ref-type="fig" rid="fig1">Figure 1A</xref>) following the theory developed by <xref ref-type="bibr" rid="bib53">Hudson and Kaplan, 1995</xref> and <xref ref-type="bibr" rid="bib74">Nordborg et al., 1996</xref>. In this model, the deleterious mutation rate per site and distribution of selection effects in a given type of region (e.g. exons) are parameters to be estimated (see Appendix 1 Section 1.1 for details). In turn, we model the effects of sweeps, <inline-formula><mml:math id="inf11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, as a function of genetic distance from substitutions on the human lineage that may have been beneficial (<xref ref-type="fig" rid="fig1">Figure 1B</xref>), following <xref ref-type="bibr" rid="bib6">Barton, 1998</xref> and <xref ref-type="bibr" rid="bib38">Gillespie, 2000</xref>. Here, the fraction of substitutions of a given type (e.g. nonsynonymous) that were beneficial and their distribution of selection effects are parameters to be estimated (see Appendix 1 Section 1.1 for details). Importantly, our model should capture the effects of any kind of sweeps, be they hard, partial or soft, so long as they eventually resulted in a substitution and affected diversity levels nearby (see <xref ref-type="bibr" rid="bib25">Coop and Ralph, 2012</xref> and SOM Section D in <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>).</p><p>Given the positions of different types of putatively selected regions and substitutions, their corresponding selection parameters, and a fine-scale genetic map, the model allows us to calculate the marginal probability that any given neutral site in the genome is polymorphic in a sample (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). Provided measurements of polymorphism at neutral positions throughout the genome, we combine information across sites and samples to calculate the composite likelihood of selection parameters, and find the parameter values that maximize this likelihood (<xref ref-type="fig" rid="fig1">Figure 1</xref>). In addition to parameter estimation, this approach yields a map of the expected neutral diversity levels along the genome (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). The mathematical form of the model and of the algorithms used for inference are detailed in Appendix 1 Section 1.</p><p>To infer the effects of background selection and selective sweeps on human diversity levels, we analyze autosomal polymorphism data from 26 human populations, collected in Phase III of the 1000 Genomes Project (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>). Here, we focus on data from 108 genomes sampled from the Yoruba population (YRI), but we get similar results for the other populations (Appendix 1 Sections 7 and 9). To estimate diversity levels at neutral sites, we focus on non-genic autosomal sites that are the least conserved in a multiple sequence alignment of 25 supra-primates (see Appendix 1 Section 3.1). To account for variation in mutation rates among neutral sites, we use estimates of the relative mutation rate for contiguous, non-overlapping blocks of 6000 putatively neutral sites, obtained from substitution rates in an eight-primate phylogeny (see Appendix 1 Section 3.3). To minimize the confounding of recombination rate estimates and diversity levels, we use a high-resolution genetic map inferred from ancestry switches in African-Americans (<xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref>), which is highly correlated with other maps (<xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref>) but is less dependent on diversity levels.</p></sec><sec id="s2-2"><title>Background selection</title><p>We first focus on two of our best-fitting models of the effects of background selection (see below and Appendix 1 Section 4). In both cases, we take as putative targets of purifying selection the 6% of autosomal sites estimated as most likely to be under selective constraint. In one, we choose these sites using phastCons conservation scores obtained for a 99-vertebrate phylogeny that excludes humans (<xref ref-type="bibr" rid="bib99">Siepel et al., 2005</xref>). In the other, we rely on Combined Annotation-Dependent Depletion (CADD) scores, which are based primarily on phylogenetic conservation (excluding humans) but also on information from functional genomic assays (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Rentzsch et al., 2019</xref>); to avoid circularity, we use scores that were generated without the <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref> <italic>B</italic>-map as input (see Appendix 1 Section 2.5).</p><p>From these models, we obtain a map of predicted diversity levels (accounting for variation in mutation rates), which we can then compare to observed data (<xref ref-type="fig" rid="fig2">Figure 2A</xref> and <xref ref-type="fig" rid="app1fig24">Appendix 1—figure 24</xref>). We generate these maps using out-of-sample predictions in non-overlapping, contiguous 2 Mb windows (which we note is substantially greater than the scale of linkage disequilibrium in human populations; <xref ref-type="bibr" rid="bib113">Wall and Pritchard, 2003</xref>). Over-fitting has a negligible effect on our results (also see Appendix 1 Section 6.1 and <xref ref-type="fig" rid="app1fig48">Appendix 1—figure 48</xref>), as expected given that the model has few parameters and the large amount of data (7 fitted parameters in this case and 2580 Mb blocks of ~653M putatively neutral sites spread over ~2600 LD blocks; <xref ref-type="bibr" rid="bib12">Berisa and Pickrell, 2016</xref>). As a measure of the precision of our predictions, we consider the variance in diversity levels explained in non-overlapping autosomal windows (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Our predictions explain a large proportion of the variance across spatial scales: at the 1 Mb scale, the predictions based on CADD scores account for 60% of the variance in diversity levels compared to 32% explained by previous work (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>; see Appendix 1 Section 4.6).</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Comparison of diversity levels predicted by our best-fitting maps of background selection effects with observations.</title><p>(<bold>A</bold>) Predicted and observed diversity levels along chromosome 1 in the YRI sample. Diversity levels are measured in 1 Mb windows, with a 0.5 Mb overlap, with the autosomal mean set to 1. (<bold>B</bold>) The proportion of variance in YRI diversity levels explained by background selection models at different spatial scales. Shown are the results for four choices of putative targets of selection: all sites with the highest 6% of CADD or phastCons scores (denoted CADD and phastCons, respectively) and the subset of these sites that are exonic (denoted CADD<sub>e</sub> and phastCons<sub>e</sub>, respectively). The results shown for our best-fitting models (based on the 6% of sites with the highest CADD or phastCons scores) are based on out-of-sample predictions in non-overlapping, contiguous 2 Mb windows. See Appendix 1 Section 4 for similar graphs with other choices, and Appendix 1 Sections 7 and 9 for other populations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-fig2-v2.tif"/></fig></sec><sec id="s2-3"><title>Selective sweeps</title><p>Next, we examine whether incorporating selective sweeps alongside background selection improves our predictions. Our inference should be able to tease apart the effects of selective sweeps, primarily because their effects, unlike those of background selection, should be centered around the locations of substitutions. Moreover, as noted, we expect to capture the effects of selective sweeps, be they hard, partial or soft (<xref ref-type="bibr" rid="bib102">Smith and Haigh, 1974</xref>; <xref ref-type="bibr" rid="bib57">Kaplan et al., 1989</xref>; <xref ref-type="bibr" rid="bib46">Hermisson and Pennings, 2005</xref>; <xref ref-type="bibr" rid="bib86">Przeworski et al., 2005</xref>; <xref ref-type="bibr" rid="bib79">Pennings and Hermisson, 2006a</xref>; <xref ref-type="bibr" rid="bib80">Pennings and Hermisson, 2006b</xref>; <xref ref-type="bibr" rid="bib25">Coop and Ralph, 2012</xref>; <xref ref-type="bibr" rid="bib11">Berg and Coop, 2015</xref>), so long as they resulted in substitutions and substantially affected diversity levels (see <xref ref-type="bibr" rid="bib25">Coop and Ralph, 2012</xref> and SOM Section D in <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). Indeed, previous work that applied a similar methodology to data from <italic>Drosophila melanogaster</italic> was able to identify distinct effects of background selection and sweeps (<xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). To examine whether we can identify such effects in humans, we consider several choices of putatively selected substitutions along the human lineage, including any nonsynonymous substitutions or any nonsynonymous and non-coding substitutions in constrained regions, allowing each type to have its own selection parameters and considering different measures of constraint (see Appendix 1 Section 4.5). Regardless of the types of substitutions considered, incorporating sweeps does not improve our fit. In fact, in all cases, our estimates of the proportion of substitutions resulting in sweeps with discernable effects on neutral diversity is approximately 0.</p><p>Moreover, in contrast to previous attempts (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>; <xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>), our model of background selection alone provides good quantitative fits to the diversity levels observed around different genomic features and in particular around nonsynonymous and synonymous substitutions (<xref ref-type="fig" rid="fig3">Figure 3</xref> and <xref ref-type="fig" rid="app1fig49">Appendix 1—figure 49</xref>). Together, these results refute the hypothesis that reduced diversity levels around nonsynonymous substitutions in humans reflect ‘masked’ effects of selective sweeps (<xref ref-type="bibr" rid="bib33">Enard et al., 2014</xref>); more generally, they indicate that selective sweeps resulting in substitutions had little effect on diversity levels in contemporary humans.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>A background selection model predicts neutral diversity levels observed around human-specific nonsynonymous (NS) substitutions.</title><p>Shown are the results for putatively neutral sites as a function of their genetic distance to the nearest nonsynonymous substitution (in 160 bins, each spanning 0.005 cM). For observed values, we average diversity levels within each bin. For predicted values, we average diversity levels predicted by our best-fitting CADD-based model (using the out-of-sample predictions in non-overlapping, contiguous, 2 Mb windows) and correct for relative mutation rate in each bin (using substitution data; see Appendix 1 Section 3.3). Both observed and predicted diversity levels are plotted relative to the autosomal mean. See <xref ref-type="fig" rid="app1fig49">Appendix 1—figure 49</xref> and <xref ref-type="fig" rid="app1fig51">Appendix 1—figure 51</xref> for similar graphs for other genomic features and using data from other populations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-fig3-v2.tif"/></fig><p>The lack of sweeps does not imply that adaptation was rare in recent human evolution, as instead, much of it may have been driven by selection on genetically complex traits, that is, traits with heritable variation arising from many segregating loci (<xref ref-type="bibr" rid="bib24">Coop et al., 2009</xref>; <xref ref-type="bibr" rid="bib84">Pritchard et al., 2010</xref>; <xref ref-type="bibr" rid="bib83">Pritchard and Di Rienzo, 2010</xref>; <xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>; <xref ref-type="bibr" rid="bib97">Sella and Barton, 2019</xref>). Complex traits are often subject to ongoing stabilizing selection, that is, selection that acts to maintain traits near an optimal value (<xref ref-type="bibr" rid="bib122">Wright, 1935</xref>; <xref ref-type="bibr" rid="bib91">Robertson, 1966</xref>; <xref ref-type="bibr" rid="bib116">Walsh and Lynch, 2018</xref>; <xref ref-type="bibr" rid="bib97">Sella and Barton, 2019</xref>). Changes in selection pressures, that is, in optimal trait values, introduce transient directional selection on such complex traits. Under plausible conditions, we expect the adaptive response to directional selection to be highly polygenic, with phenotypic adaptation to new optima achieved rapidly, via tiny increases to the frequency of many alleles that change the traits in the direction favored by selection (<xref ref-type="bibr" rid="bib45">Hayward and Sella, 2019</xref>). Over the long run, these tiny frequency changes cause a tiny excess of fixations of the alleles that were initially favored by selection (<xref ref-type="bibr" rid="bib45">Hayward and Sella, 2019</xref>). Consequently, polygenic adaptation introduces only minor perturbations to allele trajectories compared to the case in which selection pressures on traits remain constant. In particular, the alleles that eventually fix do so extremely slowly, with trajectories that are predominated by weak selection and drift (<xref ref-type="bibr" rid="bib45">Hayward and Sella, 2019</xref>), implying that their effects on linked diversity levels should be negligible (<xref ref-type="bibr" rid="bib7">Barton, 2000</xref>; <xref ref-type="bibr" rid="bib108">Thornton, 2019</xref>).</p><p>In contrast, ongoing stabilizing selection on complex traits could have a substantial effect on linked, neutral diversity levels (<xref ref-type="bibr" rid="bib45">Hayward and Sella, 2019</xref>). Stabilizing selection induces purifying selection against minor alleles that affect complex traits (<xref ref-type="bibr" rid="bib121">Wright, 1931</xref>; <xref ref-type="bibr" rid="bib91">Robertson, 1966</xref>; <xref ref-type="bibr" rid="bib100">Simons et al., 2018</xref>), and purifying selection on these alleles could be a major source of background selection (<xref ref-type="bibr" rid="bib45">Hayward and Sella, 2019</xref>). In other words, if much of the selection in humans is driven by ongoing and changing selection pressures on complex traits, we may expect background selection to be the dominant mode of linked selection, as our results indicate.</p></sec><sec id="s2-4"><title>The source of background selection</title><p>Focusing then on models of background selection alone, we ask which genomic annotations appear to be the sources of purifying selection. Previous work found selection on non-exonic regions to contribute little, to the extent that removing conserved non-exonic sites from a model of background selection had little effect on predicted diversity levels (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>). In contrast, when we include only conserved exonic regions in our inference, our predictive ability is considerably diminished (<xref ref-type="fig" rid="fig2">Figure 2B</xref>).</p><p>Moreover, in models that include separate selection parameters for conserved exonic and non-exonic regions, purifying selection on non-exonic regions accounts for most of the reduction in linked neutral diversity (Appendix 1 Section 4.3). Our estimates suggest that ~80% of deleterious mutations affecting neutral diversity occur in non-exonic regions (e.g. in the model with the top 6% of phastCons scores, ~84% of selected sites and ~76% of deleterious mutations are non-exonic; with the top 6% of CADD scores, ~83% of selected sites and ~85% of deleterious mutations are non-exonic; see Appendix 1 Sections 4.3 and 4.6). Our estimates of the average strength of selection differ between exonic and non-exonic regions, but because the total reduction in diversity levels caused by background selection is fairly insensitive to the strength of selection (with the reduction being more localized for weakly selected mutations than for strong ones), the proportions of deleterious mutations that occur in these regions approximate their relative effects on neutral diversity levels (<xref ref-type="bibr" rid="bib52">Hudson, 1994</xref>; see Appendix 1 Sections 4.3, 4.4, and 4.6). Thus, our estimates suggest that purifying selection on non-exonic regions accounts for ~80% of the reduction in linked neutral diversity. Moreover, including separate selection parameters for conserved exonic and non-exonic regions does not improve our predictions (Appendix 1 Section 4.3 and <xref ref-type="fig" rid="app1fig19">Appendix 1—figure 19</xref>).</p><p>Incorporating additional functional genomic information also does little to improve our predictions (Appendix 1 Sections 4.2 and 4.4). Notably, when we do not incorporate information on phylogenetic conservation, but include separate selection parameters for coding regions and for each of the Encyclopedia of DNA Elements (ENCODE) classes of candidate cis-regulatory elements (cCRE) (<xref ref-type="bibr" rid="bib69">Moore et al., 2020</xref>), our predictive ability is considerably diminished (Appendix 1 Section 4.4). Moreover, using CADD scores (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Rentzsch et al., 2019</xref>), which augment information on phylogenetic conservation with functional genomic information, offers little improvement over relying on conservation alone (e.g., explaining 59.9% compared to 59.7% of the variance in diversity levels in 1 Mb windows, a difference that is not statistically significant; Appendix 1 Section 6). Thus, at present, functional annotations that do not incorporate phylogenetic conservation appear to provide poorer predictions of the effects of linked selection and those that do, offer little improvement over using conservation alone (see Appendix 1 Sections 4.1–4).</p><p>In turn, our predictions based on conservation are fairly insensitive to the phylogenetic depth of the alignments used to infer conservation levels, although we do slightly better using a 99-vertebrate alignment (excluding humans) compared to its monophyletic subsets (e.g. <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref> and <xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33</xref> and Appendix 1 Section 6.2). Our best-fitting models by a variety of metrics, are obtained using 5–7% of sites with the top CADD or phastCons scores as selection targets (<xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16</xref> and <xref ref-type="fig" rid="app1fig26">Appendix 1—figure 26</xref>). This percentage is in good accordance with more direct estimates of the proportion of the human genome subject to functional constraint (<xref ref-type="bibr" rid="bib118">Ward and Kellis, 2012</xref>; <xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>).</p></sec><sec id="s2-5"><title>Estimates of the deleterious mutation rate</title><p>Reassuringly, the deleterious mutation rates that we estimate for our best-fitting models are plausible (<xref ref-type="fig" rid="fig4">Figure 4</xref>). Current estimates of the average mutation rate per site per generation in humans, including point mutations (<xref ref-type="bibr" rid="bib64">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>), indels (<xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>), mobile element insertions (<xref ref-type="bibr" rid="bib37">Gardner et al., 2019</xref>), and structural mutations (<xref ref-type="bibr" rid="bib106">Sudmant et al., 2015</xref>; <xref ref-type="bibr" rid="bib10">Belyeu et al., 2021</xref>) lie in the range of <inline-formula><mml:math id="inf12"><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.38</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per base pair per generation (Appendix 1 Section 5). Further accounting for the length of deletions (<xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>)—whereby a deletion that starts at a neutral site and includes selected sites should contribute to our estimate of the deleterious mutation rate, but deletions that affect one or several selected sites should have the same contribution—suggests that the upper bound on estimates of the deleterious mutations rate at putatively selected sites should fall in the range of <inline-formula><mml:math id="inf13"><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.51</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per base pair per generation (Appendix 1 Section 5). The estimates for all of our best-fitting models fall well below this bound (<xref ref-type="fig" rid="fig4">Figure 4</xref>). This is expected, because not every mutation at putatively selected sites will be deleterious: some sites are misclassified as constrained and some mutations at selected sites are selectively neutral.</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Estimates of the proportion of mutations at putatively selected sites that are deleterious.</title><p>Shown are the results using 5–7% of sites with the highest phastCons scores (<bold>A</bold>) and CADD scores (<bold>B</bold>) as selection targets. For estimates based on fitting background selection models, we divide our estimates of the deleterious mutation rate per selected site by the estimate of the total mutation rate per site, where the ranges correspond to the range of estimates of the total rate, that is, <inline-formula><mml:math id="inf14"><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.51</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per base pair per generation (Appendix 1 Section 5.1). For estimates based on evolutionary rates (on the human lineage from the common ancestor of humans and chimpanzees), we take the ratio of the estimated rates at putatively selected sites and at matched sets of putatively neutral sites (see text and Appendix 1 Section 5.2 for details).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-fig4-v2.tif"/></fig><p>To test whether our estimates of the proportion of mutations that are deleterious are plausible, we compare them with independent estimates based on the relative reduction in evolutionary rates at putatively selected vs. neutral sites along the human lineage (these sets of sites were identified from an alignment that excludes humans; Appendix 1 Sections 3.1, 4.1, and 4.4). The relative reduction allows us to estimate the proportion of deleterious mutations because deleterious mutations at selected sites rarely fix in the population whereas neutral mutations fix at a much higher rate, which is the same at selected and neutral sites (<xref ref-type="bibr" rid="bib62">Kimura and Crow, 1964</xref>). In estimating the reduction at putatively selected sites, we matched the set of putatively neutral sites for the AT/GC ratio, and checked that our estimates were insensitive to the composition of other genomic features associated with mutation rates and with other non-selective processes that affect substitution rates (e.g., triplet context, methylated CpGs and recombination rates, which affect rates of biased gene conversion; Appendix 1 Section 5).</p><p>Our estimates based on evolutionary rates are closer to (and even overlap) those obtained from fitting models of background selection based on CADD scores compared to those based on phastCons scores (<xref ref-type="fig" rid="fig4">Figure 4</xref>). This is expected given that CADD scores are much better than phastCons scores at identifying constraint on a single site resolution (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Rentzsch et al., 2019</xref>), which markedly influences evolutionary rates at putatively selected sites (but not the predictions of background selection effects). We expect the two estimates to be similar but not identical, both because weak selection has a larger effect on evolutionary rates than on linked diversity levels (<xref ref-type="bibr" rid="bib67">McVean and Charlesworth, 2000</xref>; <xref ref-type="bibr" rid="bib20">Comeron and Kreitman, 2002</xref>; <xref ref-type="bibr" rid="bib41">Gordo et al., 2002</xref>; <xref ref-type="bibr" rid="bib18">Charlesworth, 2013</xref>; <xref ref-type="bibr" rid="bib39">Good et al., 2014</xref>) and because estimates based on the effects of background selection may absorb the deleterious mutation rate at selected sites that were not included in our sets but are closely linked to sites in them (Appendix 1 Section 5). In summary, given the fit to data and plausible estimates of the deleterious rates, it is natural to interpret our maps as reflecting the effects of background selection, that is, as maps of <inline-formula><mml:math id="inf15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (defined as the ratio of expected diversity levels with background selection, <inline-formula><mml:math id="inf16"><mml:mi>π</mml:mi></mml:math></inline-formula>, and in its absence, <inline-formula><mml:math id="inf17"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>; <xref ref-type="bibr" rid="bib17">Charlesworth et al., 1993</xref>).</p></sec><sec id="s2-6"><title>Background selection on autosomes</title><p>Our maps are also well calibrated (<xref ref-type="fig" rid="fig5">Figure 5</xref>). When we stratify diversity levels at putatively neutral sites by our predictions, predicted and observed diversity levels are similar throughout nearly the entire range of predicted values (e.g. <inline-formula><mml:math id="inf18"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0.96</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> when sites are in predicted percentile bins). One exception is for ~5% of sites in which background selection is predicted to be the strongest (i.e. with the lowest <inline-formula><mml:math id="inf19"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>), where our predictions are imprecise. This behavior is due to a technical approximation we employ in fitting the models (see Appendix 1 Section 1.5). The other exception is for ~2% of sites in which background selection is predicted to be the weakest (i.e. with <inline-formula><mml:math id="inf20"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> near 1), where observed diversity levels are markedly greater than expected. We observe similar behavior in all the human populations examined (<xref ref-type="fig" rid="app1fig52">Appendix 1—figure 52</xref>), and we cannot fully explain it by known mutational and recombination effects (e.g. of base composition and biased gene conversion; Appendix 1 Section 8). This behavior could reflect ancient introgression of archaic human DNA into ancestors of contemporary humans (Appendix 1 Section 8.3), indicated also in other population genetic signatures (<xref ref-type="bibr" rid="bib114">Wall and Hammer, 2006</xref>; <xref ref-type="bibr" rid="bib42">Green et al., 2010</xref>; <xref ref-type="bibr" rid="bib89">Reich et al., 2010</xref>; <xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib87">Racimo et al., 2015</xref>; <xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>). Such introgressed regions are expected to increase genetic diversity and persist the longest in regions with low functional density and high recombination, corresponding to weak background selection effects (<xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib44">Harris and Nielsen, 2016</xref>; <xref ref-type="bibr" rid="bib56">Juric et al., 2016</xref>; <xref ref-type="bibr" rid="bib95">Schumer et al., 2018</xref>).</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Observed vs. predicted neutral diversity levels across the autosomes.</title><p>Shown are the results for the best-fitting CADD-based model, using the out-of-sample predictions in non-overlapping, contiguous 2 Mb windows. Light orange scatter plot: we divide putatively neutral sites into 100 equally sized bins based on the predicted <inline-formula><mml:math id="inf21"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. For predicted values (x-axis), we average the predicted <inline-formula><mml:math id="inf22"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in each bin. For observed values (y-axis), we divide the average diversity level by the estimate of the average relative mutation rate (obtained from substitution data; see Appendix 1 Section 3.3) in each bin, and normalize by the autosomal average of <inline-formula><mml:math id="inf23"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (estimated from fitting the model; see Appendix 1 Section 1.1). Owing to a technical approximation (see Appendix 1 Section 1.5), our method forces the predictions for the 5 bins with the lowest predicted <inline-formula><mml:math id="inf24"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (open, hatched circles on the left) to be similar; we therefore also show the results for these bins grouped together (dark red circle). Dark orange curve: the LOESS fit for a similarly defined scatter plot but with 2000 rather than 100 bins (with span = 0.1). For similar graphs corresponding to other models and using data from other populations, see Appendix 1 Sections 4 and 9, respectively.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-fig5-v2.tif"/></fig><p>Setting these outlier regions aside, we can use the maps to characterize the distribution of background selection effects in human autosomes. We note that background selection effects that are not captured by our models would cause us to underestimate the range and extent of background selection effects (<xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). We find that diversity levels throughout almost all of the autosomes are affected by background selection, with a ~37% reduction in the 10% most affected sites, a non-zero (~2.1%) reduction even in the 10% least affected (after excluding outliers in the top 2% of bins; see <xref ref-type="fig" rid="fig5">Figure 5</xref>), and a mean reduction of ~17%. These conclusions are robust across our best-fitting maps and populations (Appendix 1 Section 4 and <xref ref-type="fig" rid="app1fig35">Appendix 1—figure 35</xref> and <xref ref-type="fig" rid="app1fig52">Appendix 1—figure 52</xref>). An important implication is that our maps of the effects of background selection provide a more accurate null model than currently used for other population genetic inferences that rely on diversity levels, notably inferences about demographic history (<xref ref-type="bibr" rid="bib94">Schiffels and Durbin, 2014</xref>; <xref ref-type="bibr" rid="bib107">Terhorst et al., 2017</xref>; <xref ref-type="bibr" rid="bib82">Pouyet et al., 2018</xref>).</p></sec><sec id="s2-7"><title>Conclusion</title><p>Our results indicate that background selection is the dominant mode of linked selection in human autosomes and the major determinant of neutral diversity levels on the Mb scale (after accounting for variation in mutation rates). They further reveal that background selection effects arise primarily from purifying selection at non-coding regions of the genome. Non-coding regions are known to exhibit substantial functional turnover on evolutionary timescales (<xref ref-type="bibr" rid="bib118">Ward and Kellis, 2012</xref>; <xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>), and yet we find phylogenetic conservation to be the best predictor of selected regions. Moreover, at present, augmenting measures of conservation with functional genomic information in humans offers little improvement. It therefore remains unclear how much our maps can still be improved. Even without these potential refinements, our findings demonstrate that a simple model of background selection, conceived three decades ago (<xref ref-type="bibr" rid="bib17">Charlesworth et al., 1993</xref>), provides a reliable quantitative prediction of genetic diversity levels throughout human autosomes.</p></sec></sec></body><back><sec sec-type="additional-information" id="s3"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>is affiliated with MyHeritage. The author has no financial interests to declare</p></fn><fn fn-type="COI-statement" id="conf3"><p>is affiliated with Flatiron Health Inc, The author has no financial interests to declare</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Resources, Data curation, Software, Formal analysis, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Resources, Software</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Formal analysis, Supervision, Investigation, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Formal analysis, Supervision, Funding acquisition, Investigation, Methodology, Writing – original draft, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s4"><title>Additional files</title><supplementary-material id="transrepform"><label>Transparent reporting form</label><media xlink:href="elife-76065-transrepform1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s5"><title>Data availability</title><p>Shared data can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/sellalab/HumanLinkedSelectionMaps">https://github.com/sellalab/HumanLinkedSelectionMaps</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:b177485acbb8bc94742060ab3a7a443a473b3271;origin=https://github.com/sellalab/HumanLinkedSelectionMaps;visit=swh:1:snp:c14f688b4c7fdc1e530c1b9fca0debc45f00dcb4;anchor=swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4">swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4</ext-link>). This repository includes fully documented code for: downloading and processing public datasets used, running inferences, analyzing results, and generating all figures from the manuscript. This repository also includes B-maps for all &quot;best-fitting&quot; models described in the manuscript. Customized CADD scores with bStatistic removed are available on Data Dryad at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5061/dryad.n8pk0p2x0">https://doi.org/10.5061/dryad.n8pk0p2x0</ext-link>.</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Murphy</surname><given-names>D</given-names></name><name><surname>Elyashiv</surname><given-names>E</given-names></name><name><surname>Amster</surname><given-names>G</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>CADD scores version 1.6 with bStatistic removed from inputs</data-title><source>Dryad Digital Repository</source><pub-id pub-id-type="doi">10.5061/dryad.n8pk0p2x0</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Molly Przeworski for helpful discussions throughout this work. We also thank Ipsita Agarwal, Eduardo Amorim, Peter Andolfatto, Iain Mathieson, Priya Moorjani, Itsik Pe’er, Joe Pickrell and Jonathan Pritchard for helpful discussions. We thank Yun Song for sharing unpublished results, and Lusiné Nazaretyan, Philipp Rentzsch, Max Schubach and Martin Kircher from the Kircher lab for generating CADD scores that were tailored for our purposes. We thank Ipsita Agarwal, Peter Andolfatto, Jeff Ross-Ibarra, Magnus Nordborg, Jonathan Pritchard, Molly Przeworski and one anonymous reviewer for comments on the manuscript.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abecasis</surname><given-names>GR</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>DePristo</surname><given-names>MA</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Handsaker</surname><given-names>RE</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Marth</surname><given-names>GT</given-names></name><name><surname>McVean</surname><given-names>GA</given-names></name><collab>1000 Genomes Project Consortium</collab></person-group><year iso-8601-date="2012">2012</year><article-title>An integrated map of genetic variation from 1,092 human genomes</article-title><source>Nature</source><volume>491</volume><fpage>56</fpage><lpage>65</lpage><pub-id pub-id-type="doi">10.1038/nature11632</pub-id><pub-id pub-id-type="pmid">23128226</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Andolfatto</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Hitchhiking effects of recurrent beneficial amino acid substitutions in the <italic>Drosophila melanogaster</italic> genome</article-title><source>Genome Research</source><volume>17</volume><fpage>1755</fpage><lpage>1762</lpage><pub-id pub-id-type="doi">10.1101/gr.6691007</pub-id><pub-id pub-id-type="pmid">17989248</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Apostolico</surname><given-names>A</given-names></name><name><surname>Guerra</surname><given-names>C</given-names></name><name><surname>Istrail</surname><given-names>S</given-names></name><name><surname>Pevzner</surname><given-names>PA</given-names></name><name><surname>Waterman</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2006">2006</year><source>Research in Computational Molecular Biology</source><publisher-loc>Berlin, Heidelberg</publisher-loc><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/11732990</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Garrison</surname><given-names>EP</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Korbel</surname><given-names>JO</given-names></name><name><surname>Marchini</surname><given-names>JL</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><name><surname>McVean</surname><given-names>GA</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A global reference for human genetic variation</article-title><source>Nature</source><volume>526</volume><fpage>68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature15393</pub-id><pub-id pub-id-type="pmid">26432245</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barrett</surname><given-names>T</given-names></name><name><surname>Wilhite</surname><given-names>SE</given-names></name><name><surname>Ledoux</surname><given-names>P</given-names></name><name><surname>Evangelista</surname><given-names>C</given-names></name><name><surname>Kim</surname><given-names>IF</given-names></name><name><surname>Tomashevsky</surname><given-names>M</given-names></name><name><surname>Marshall</surname><given-names>KA</given-names></name><name><surname>Phillippy</surname><given-names>KH</given-names></name><name><surname>Sherman</surname><given-names>PM</given-names></name><name><surname>Holko</surname><given-names>M</given-names></name><name><surname>Yefanov</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>H</given-names></name><name><surname>Zhang</surname><given-names>N</given-names></name><name><surname>Robertson</surname><given-names>CL</given-names></name><name><surname>Serova</surname><given-names>N</given-names></name><name><surname>Davis</surname><given-names>S</given-names></name><name><surname>Soboleva</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>NCBI GEO: archive for functional genomics data sets – update</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>D991</fpage><lpage>D995</lpage><pub-id pub-id-type="doi">10.1093/nar/gks1193</pub-id><pub-id pub-id-type="pmid">23193258</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barton</surname><given-names>NH</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>The effect of hitch-hiking on neutral genealogies</article-title><source>Genetical Research</source><volume>72</volume><fpage>123</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1017/S0016672398003462</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barton</surname><given-names>NH</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Genetic hitchhiking</article-title><source>Philosophical Transactions of the Royal Society of London Series B, Biological Sciences</source><volume>355</volume><fpage>1553</fpage><lpage>1562</lpage><pub-id pub-id-type="doi">10.1098/rstb.2000.0716</pub-id><pub-id pub-id-type="pmid">11127900</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Begun</surname><given-names>DJ</given-names></name><name><surname>Aquadro</surname><given-names>CF</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Levels of naturally occurring DNA polymorphism correlate with recombination rates in <italic>D. melanogaster</italic></article-title><source>Nature</source><volume>356</volume><fpage>519</fpage><lpage>520</lpage><pub-id pub-id-type="doi">10.1038/356519a0</pub-id><pub-id pub-id-type="pmid">1560824</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Begun</surname><given-names>DJ</given-names></name><name><surname>Holloway</surname><given-names>AK</given-names></name><name><surname>Stevens</surname><given-names>K</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Poh</surname><given-names>YP</given-names></name><name><surname>Hahn</surname><given-names>MW</given-names></name><name><surname>Nista</surname><given-names>PM</given-names></name><name><surname>Jones</surname><given-names>CD</given-names></name><name><surname>Kern</surname><given-names>AD</given-names></name><name><surname>Dewey</surname><given-names>CN</given-names></name><name><surname>Pachter</surname><given-names>L</given-names></name><name><surname>Myers</surname><given-names>E</given-names></name><name><surname>Langley</surname><given-names>CH</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Population genomics: whole-genome analysis of polymorphism and divergence in <italic>Drosophila simulans</italic></article-title><source>PLOS Biology</source><volume>5</volume><elocation-id>e310</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0050310</pub-id><pub-id pub-id-type="pmid">17988176</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Belyeu</surname><given-names>JR</given-names></name><name><surname>Brand</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Zhao</surname><given-names>X</given-names></name><name><surname>Pedersen</surname><given-names>BS</given-names></name><name><surname>Feusier</surname><given-names>J</given-names></name><name><surname>Gupta</surname><given-names>M</given-names></name><name><surname>Nicholas</surname><given-names>TJ</given-names></name><name><surname>Brown</surname><given-names>J</given-names></name><name><surname>Baird</surname><given-names>L</given-names></name><name><surname>Devlin</surname><given-names>B</given-names></name><name><surname>Sanders</surname><given-names>SJ</given-names></name><name><surname>Jorde</surname><given-names>LB</given-names></name><name><surname>Talkowski</surname><given-names>ME</given-names></name><name><surname>Quinlan</surname><given-names>AR</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title><italic>De novo</italic> structural mutation rates and gamete-of-origin biases revealed through genome sequencing of 2,396 families</article-title><source>American Journal of Human Genetics</source><volume>108</volume><fpage>597</fpage><lpage>607</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2021.02.012</pub-id><pub-id pub-id-type="pmid">33675682</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Berg</surname><given-names>JJ</given-names></name><name><surname>Coop</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A coalescent model for a sweep of a unique standing variant</article-title><source>Genetics</source><volume>201</volume><fpage>707</fpage><lpage>725</lpage><pub-id pub-id-type="doi">10.1534/genetics.115.178962</pub-id><pub-id pub-id-type="pmid">26311475</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Berisa</surname><given-names>T</given-names></name><name><surname>Pickrell</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Approximately independent linkage disequilibrium blocks in human populations</article-title><source>Bioinformatics</source><volume>32</volume><fpage>283</fpage><lpage>285</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv546</pub-id><pub-id pub-id-type="pmid">26395773</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Besenbacher</surname><given-names>S</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Helgason</surname><given-names>H</given-names></name><name><surname>Kristjansson</surname><given-names>H</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Magnusson</surname><given-names>OT</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Kong</surname><given-names>A</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Multi-nucleotide <italic>de novo</italic> mutations in humans</article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1006315</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1006315</pub-id><pub-id pub-id-type="pmid">27846220</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Black</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Mechanisms of alternative pre-messenger RNA splicing</article-title><source>Annual Review of Biochemistry</source><volume>72</volume><fpage>291</fpage><lpage>336</lpage><pub-id pub-id-type="doi">10.1146/annurev.biochem.72.121801.161720</pub-id><pub-id pub-id-type="pmid">12626338</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blanchette</surname><given-names>M</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Riemer</surname><given-names>C</given-names></name><name><surname>Elnitski</surname><given-names>L</given-names></name><name><surname>Smit</surname><given-names>AFA</given-names></name><name><surname>Roskin</surname><given-names>KM</given-names></name><name><surname>Baertsch</surname><given-names>R</given-names></name><name><surname>Rosenbloom</surname><given-names>K</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Green</surname><given-names>ED</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Miller</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Aligning multiple genomic sequences with the threaded blockset aligner</article-title><source>Genome Research</source><volume>14</volume><fpage>708</fpage><lpage>715</lpage><pub-id pub-id-type="doi">10.1101/gr.1933104</pub-id><pub-id pub-id-type="pmid">15060014</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname><given-names>JJ</given-names></name><name><surname>Macpherson</surname><given-names>JM</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Pervasive hitchhiking at coding and regulatory sites in humans</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000336</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000336</pub-id><pub-id pub-id-type="pmid">19148272</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Charlesworth</surname><given-names>B</given-names></name><name><surname>Morgan</surname><given-names>MT</given-names></name><name><surname>Charlesworth</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>The effect of deleterious mutations on neutral molecular variation</article-title><source>Genetics</source><volume>134</volume><fpage>1289</fpage><lpage>1303</lpage><pub-id pub-id-type="doi">10.1093/genetics/134.4.1289</pub-id><pub-id pub-id-type="pmid">8375663</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Charlesworth</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Background selection 20 years on: the Wilhelmine E. Key 2012 invitational lecture</article-title><source>The Journal of Heredity</source><volume>104</volume><fpage>161</fpage><lpage>171</lpage><pub-id pub-id-type="doi">10.1093/jhered/ess136</pub-id><pub-id pub-id-type="pmid">23303522</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Church</surname><given-names>DM</given-names></name><name><surname>Schneider</surname><given-names>VA</given-names></name><name><surname>Graves</surname><given-names>T</given-names></name><name><surname>Auger</surname><given-names>K</given-names></name><name><surname>Cunningham</surname><given-names>F</given-names></name><name><surname>Bouk</surname><given-names>N</given-names></name><name><surname>Chen</surname><given-names>HC</given-names></name><name><surname>Agarwala</surname><given-names>R</given-names></name><name><surname>McLaren</surname><given-names>WM</given-names></name><name><surname>Ritchie</surname><given-names>GRS</given-names></name><name><surname>Albracht</surname><given-names>D</given-names></name><name><surname>Kremitzki</surname><given-names>M</given-names></name><name><surname>Rock</surname><given-names>S</given-names></name><name><surname>Kotkiewicz</surname><given-names>H</given-names></name><name><surname>Kremitzki</surname><given-names>C</given-names></name><name><surname>Wollam</surname><given-names>A</given-names></name><name><surname>Trani</surname><given-names>L</given-names></name><name><surname>Fulton</surname><given-names>L</given-names></name><name><surname>Fulton</surname><given-names>R</given-names></name><name><surname>Matthews</surname><given-names>L</given-names></name><name><surname>Whitehead</surname><given-names>S</given-names></name><name><surname>Chow</surname><given-names>W</given-names></name><name><surname>Torrance</surname><given-names>J</given-names></name><name><surname>Dunn</surname><given-names>M</given-names></name><name><surname>Harden</surname><given-names>G</given-names></name><name><surname>Threadgold</surname><given-names>G</given-names></name><name><surname>Wood</surname><given-names>J</given-names></name><name><surname>Collins</surname><given-names>J</given-names></name><name><surname>Heath</surname><given-names>P</given-names></name><name><surname>Griffiths</surname><given-names>G</given-names></name><name><surname>Pelan</surname><given-names>S</given-names></name><name><surname>Grafham</surname><given-names>D</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Weinstock</surname><given-names>G</given-names></name><name><surname>Mardis</surname><given-names>ER</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Howe</surname><given-names>K</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Hubbard</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Modernizing reference genome assemblies</article-title><source>PLOS Biology</source><volume>9</volume><elocation-id>e1001091</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1001091</pub-id><pub-id pub-id-type="pmid">21750661</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comeron</surname><given-names>JM</given-names></name><name><surname>Kreitman</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Population, evolutionary and genomic consequences of interference selection</article-title><source>Genetics</source><volume>161</volume><fpage>389</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1093/genetics/161.1.389</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comeron</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Background selection as baseline for nucleotide variation across the <italic>Drosophila</italic> genome</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004434</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004434</pub-id><pub-id pub-id-type="pmid">24968283</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comeron</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Background selection as null hypothesis in population genomics: insights and challenges from <italic>Drosophila</italic> studies</article-title><source>Philosophical Transactions of the Royal Society of London Series B, Biological Sciences</source><volume>372</volume><elocation-id>20160471</elocation-id><pub-id pub-id-type="doi">10.1098/rstb.2016.0471</pub-id><pub-id pub-id-type="pmid">29109230</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Conn</surname><given-names>AR</given-names></name><name><surname>Gould</surname><given-names>NIM</given-names></name><name><surname>Toint</surname><given-names>PL</given-names></name></person-group><year iso-8601-date="2000">2000</year><source>Trust-Region Methods</source><publisher-name>Society for Industrial and Applied Mathematics</publisher-name><pub-id pub-id-type="doi">10.1007/978-0-387-40065-5_4</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coop</surname><given-names>G</given-names></name><name><surname>Pickrell</surname><given-names>JK</given-names></name><name><surname>Novembre</surname><given-names>J</given-names></name><name><surname>Kudaravalli</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Absher</surname><given-names>D</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Cavalli-Sforza</surname><given-names>LL</given-names></name><name><surname>Feldman</surname><given-names>MW</given-names></name><name><surname>Pritchard</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The role of geography in human adaptation</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000500</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000500</pub-id><pub-id pub-id-type="pmid">19503611</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coop</surname><given-names>G.</given-names></name><name><surname>Ralph</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Patterns of neutral diversity under general models of selective sweeps</article-title><source>Genetics</source><volume>192</volume><fpage>205</fpage><lpage>224</lpage><pub-id pub-id-type="doi">10.1534/genetics.112.141861</pub-id><pub-id pub-id-type="pmid">22714413</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Stone</surname><given-names>EA</given-names></name><name><surname>Asimenos</surname><given-names>G</given-names></name><collab>NISC Comparative Sequencing Program</collab><name><surname>Green</surname><given-names>ED</given-names></name><name><surname>Batzoglou</surname><given-names>S</given-names></name><name><surname>Sidow</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Distribution and intensity of constraint in mammalian genomic sequence</article-title><source>Genome Research</source><volume>15</volume><fpage>901</fpage><lpage>913</lpage><pub-id pub-id-type="doi">10.1101/gr.3577405</pub-id><pub-id pub-id-type="pmid">15965027</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cutter</surname><given-names>AD</given-names></name><name><surname>Payseur</surname><given-names>BA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Genomic signatures of selection at linked sites: unifying the disparity among species</article-title><source>Nature Reviews Genetics</source><volume>14</volume><fpage>262</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1038/nrg3425</pub-id><pub-id pub-id-type="pmid">23478346</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cvijović</surname><given-names>I</given-names></name><name><surname>Good</surname><given-names>BH</given-names></name><name><surname>Desai</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The effect of strong purifying selection on genetic diversity</article-title><source>Genetics</source><volume>209</volume><fpage>1235</fpage><lpage>1278</lpage><pub-id pub-id-type="doi">10.1534/genetics.118.301058</pub-id><pub-id pub-id-type="pmid">29844134</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Danecek</surname><given-names>P</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Abecasis</surname><given-names>G</given-names></name><name><surname>Albers</surname><given-names>CA</given-names></name><name><surname>Banks</surname><given-names>E</given-names></name><name><surname>DePristo</surname><given-names>MA</given-names></name><name><surname>Handsaker</surname><given-names>RE</given-names></name><name><surname>Lunter</surname><given-names>G</given-names></name><name><surname>Marth</surname><given-names>GT</given-names></name><name><surname>Sherry</surname><given-names>ST</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name><collab>1000 Genomes Project Analysis Group</collab></person-group><year iso-8601-date="2011">2011</year><article-title>The variant call format and VCFtools</article-title><source>Bioinformatics</source><volume>27</volume><fpage>2156</fpage><lpage>2158</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr330</pub-id><pub-id pub-id-type="pmid">21653522</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Biased gene conversion and the evolution of mammalian genomic landscapes</article-title><source>Annual Review of Genomics and Human Genetics</source><volume>10</volume><fpage>285</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-082908-150001</pub-id><pub-id pub-id-type="pmid">19630562</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Durvasula</surname><given-names>A</given-names></name><name><surname>Sankararaman</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Recovering signals of ghost archaic introgression in African populations</article-title><source>Science Advances</source><volume>6</volume><elocation-id>eaax5097</elocation-id><pub-id pub-id-type="doi">10.1126/sciadv.aax5097</pub-id><pub-id pub-id-type="pmid">32095519</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Elyashiv</surname><given-names>E</given-names></name><name><surname>Sattath</surname><given-names>S</given-names></name><name><surname>Hu</surname><given-names>TT</given-names></name><name><surname>Strutsovsky</surname><given-names>A</given-names></name><name><surname>McVicker</surname><given-names>G</given-names></name><name><surname>Andolfatto</surname><given-names>P</given-names></name><name><surname>Coop</surname><given-names>G</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A genomic map of the effects of linked selection in <italic>Drosophila</italic></article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1006130</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1006130</pub-id><pub-id pub-id-type="pmid">27536991</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Enard</surname><given-names>D</given-names></name><name><surname>Messer</surname><given-names>PW</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genome-Wide signals of positive selection in human evolution</article-title><source>Genome Research</source><volume>24</volume><fpage>885</fpage><lpage>895</lpage><pub-id pub-id-type="doi">10.1101/gr.164822.113</pub-id><pub-id pub-id-type="pmid">24619126</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fearnhead</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Consistency of estimators of the population-scaled recombination rate</article-title><source>Theoretical Population Biology</source><volume>64</volume><fpage>67</fpage><lpage>79</lpage><pub-id pub-id-type="doi">10.1016/s0040-5809(03)00041-8</pub-id><pub-id pub-id-type="pmid">12804872</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Frazer</surname><given-names>KA</given-names></name><name><surname>Ballinger</surname><given-names>DG</given-names></name><name><surname>Cox</surname><given-names>DR</given-names></name><name><surname>Hinds</surname><given-names>DA</given-names></name><name><surname>Stuve</surname><given-names>LL</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Belmont</surname><given-names>JW</given-names></name><name><surname>Boudreau</surname><given-names>A</given-names></name><name><surname>Hardenbol</surname><given-names>P</given-names></name><name><surname>Leal</surname><given-names>SM</given-names></name><name><surname>Pasternak</surname><given-names>S</given-names></name><name><surname>Wheeler</surname><given-names>DA</given-names></name><name><surname>Willis</surname><given-names>TD</given-names></name><name><surname>Yu</surname><given-names>F</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Zeng</surname><given-names>C</given-names></name><name><surname>Gao</surname><given-names>Y</given-names></name><name><surname>Hu</surname><given-names>H</given-names></name><name><surname>Hu</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>C</given-names></name><name><surname>Lin</surname><given-names>W</given-names></name><name><surname>Liu</surname><given-names>S</given-names></name><name><surname>Pan</surname><given-names>H</given-names></name><name><surname>Tang</surname><given-names>X</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>Q</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Zhou</surname><given-names>J</given-names></name><name><surname>Gabriel</surname><given-names>SB</given-names></name><name><surname>Barry</surname><given-names>R</given-names></name><name><surname>Blumenstiel</surname><given-names>B</given-names></name><name><surname>Camargo</surname><given-names>A</given-names></name><name><surname>Defelice</surname><given-names>M</given-names></name><name><surname>Faggart</surname><given-names>M</given-names></name><name><surname>Goyette</surname><given-names>M</given-names></name><name><surname>Gupta</surname><given-names>S</given-names></name><name><surname>Moore</surname><given-names>J</given-names></name><name><surname>Nguyen</surname><given-names>H</given-names></name><name><surname>Onofrio</surname><given-names>RC</given-names></name><name><surname>Parkin</surname><given-names>M</given-names></name><name><surname>Roy</surname><given-names>J</given-names></name><name><surname>Stahl</surname><given-names>E</given-names></name><name><surname>Winchester</surname><given-names>E</given-names></name><name><surname>Ziaugra</surname><given-names>L</given-names></name><name><surname>Altshuler</surname><given-names>D</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Yao</surname><given-names>Z</given-names></name><name><surname>Huang</surname><given-names>W</given-names></name><name><surname>Chu</surname><given-names>X</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Jin</surname><given-names>L</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Sun</surname><given-names>W</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Xiong</surname><given-names>X</given-names></name><name><surname>Xu</surname><given-names>L</given-names></name><name><surname>Waye</surname><given-names>MMY</given-names></name><name><surname>Tsui</surname><given-names>SKW</given-names></name><name><surname>Xue</surname><given-names>H</given-names></name><name><surname>Wong</surname><given-names>JTF</given-names></name><name><surname>Galver</surname><given-names>LM</given-names></name><name><surname>Fan</surname><given-names>JB</given-names></name><name><surname>Gunderson</surname><given-names>K</given-names></name><name><surname>Murray</surname><given-names>SS</given-names></name><name><surname>Oliphant</surname><given-names>AR</given-names></name><name><surname>Chee</surname><given-names>MS</given-names></name><name><surname>Montpetit</surname><given-names>A</given-names></name><name><surname>Chagnon</surname><given-names>F</given-names></name><name><surname>Ferretti</surname><given-names>V</given-names></name><name><surname>Leboeuf</surname><given-names>M</given-names></name><name><surname>Olivier</surname><given-names>JF</given-names></name><name><surname>Phillips</surname><given-names>MS</given-names></name><name><surname>Roumy</surname><given-names>S</given-names></name><name><surname>Sallée</surname><given-names>C</given-names></name><name><surname>Verner</surname><given-names>A</given-names></name><name><surname>Hudson</surname><given-names>TJ</given-names></name><name><surname>Kwok</surname><given-names>PY</given-names></name><name><surname>Cai</surname><given-names>D</given-names></name><name><surname>Koboldt</surname><given-names>DC</given-names></name><name><surname>Miller</surname><given-names>RD</given-names></name><name><surname>Pawlikowska</surname><given-names>L</given-names></name><name><surname>Taillon-Miller</surname><given-names>P</given-names></name><name><surname>Xiao</surname><given-names>M</given-names></name><name><surname>Tsui</surname><given-names>LC</given-names></name><name><surname>Mak</surname><given-names>W</given-names></name><name><surname>Song</surname><given-names>YQ</given-names></name><name><surname>Tam</surname><given-names>PKH</given-names></name><name><surname>Nakamura</surname><given-names>Y</given-names></name><name><surname>Kawaguchi</surname><given-names>T</given-names></name><name><surname>Kitamoto</surname><given-names>T</given-names></name><name><surname>Morizono</surname><given-names>T</given-names></name><name><surname>Nagashima</surname><given-names>A</given-names></name><name><surname>Ohnishi</surname><given-names>Y</given-names></name><name><surname>Sekine</surname><given-names>A</given-names></name><name><surname>Tanaka</surname><given-names>T</given-names></name><name><surname>Tsunoda</surname><given-names>T</given-names></name><name><surname>Deloukas</surname><given-names>P</given-names></name><name><surname>Bird</surname><given-names>CP</given-names></name><name><surname>Delgado</surname><given-names>M</given-names></name><name><surname>Dermitzakis</surname><given-names>ET</given-names></name><name><surname>Gwilliam</surname><given-names>R</given-names></name><name><surname>Hunt</surname><given-names>S</given-names></name><name><surname>Morrison</surname><given-names>J</given-names></name><name><surname>Powell</surname><given-names>D</given-names></name><name><surname>Stranger</surname><given-names>BE</given-names></name><name><surname>Whittaker</surname><given-names>P</given-names></name><name><surname>Bentley</surname><given-names>DR</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>de Bakker</surname><given-names>PIW</given-names></name><name><surname>Barrett</surname><given-names>J</given-names></name><name><surname>Chretien</surname><given-names>YR</given-names></name><name><surname>Maller</surname><given-names>J</given-names></name><name><surname>McCarroll</surname><given-names>S</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Pe’er</surname><given-names>I</given-names></name><name><surname>Price</surname><given-names>A</given-names></name><name><surname>Purcell</surname><given-names>S</given-names></name><name><surname>Richter</surname><given-names>DJ</given-names></name><name><surname>Sabeti</surname><given-names>P</given-names></name><name><surname>Saxena</surname><given-names>R</given-names></name><name><surname>Schaffner</surname><given-names>SF</given-names></name><name><surname>Sham</surname><given-names>PC</given-names></name><name><surname>Varilly</surname><given-names>P</given-names></name><name><surname>Altshuler</surname><given-names>D</given-names></name><name><surname>Stein</surname><given-names>LD</given-names></name><name><surname>Krishnan</surname><given-names>L</given-names></name><name><surname>Smith</surname><given-names>AV</given-names></name><name><surname>Tello-Ruiz</surname><given-names>MK</given-names></name><name><surname>Thorisson</surname><given-names>GA</given-names></name><name><surname>Chakravarti</surname><given-names>A</given-names></name><name><surname>Chen</surname><given-names>PE</given-names></name><name><surname>Cutler</surname><given-names>DJ</given-names></name><name><surname>Kashuk</surname><given-names>CS</given-names></name><name><surname>Lin</surname><given-names>S</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name><name><surname>Guan</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Munro</surname><given-names>HM</given-names></name><name><surname>Qin</surname><given-names>ZS</given-names></name><name><surname>Thomas</surname><given-names>DJ</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Bottolo</surname><given-names>L</given-names></name><name><surname>Cardin</surname><given-names>N</given-names></name><name><surname>Eyheramendy</surname><given-names>S</given-names></name><name><surname>Freeman</surname><given-names>C</given-names></name><name><surname>Marchini</surname><given-names>J</given-names></name><name><surname>Myers</surname><given-names>S</given-names></name><name><surname>Spencer</surname><given-names>C</given-names></name><name><surname>Stephens</surname><given-names>M</given-names></name><name><surname>Donnelly</surname><given-names>P</given-names></name><name><surname>Cardon</surname><given-names>LR</given-names></name><name><surname>Clarke</surname><given-names>G</given-names></name><name><surname>Evans</surname><given-names>DM</given-names></name><name><surname>Morris</surname><given-names>AP</given-names></name><name><surname>Weir</surname><given-names>BS</given-names></name><name><surname>Tsunoda</surname><given-names>T</given-names></name><name><surname>Mullikin</surname><given-names>JC</given-names></name><name><surname>Sherry</surname><given-names>ST</given-names></name><name><surname>Feolo</surname><given-names>M</given-names></name><name><surname>Skol</surname><given-names>A</given-names></name><name><surname>Zhang</surname><given-names>H</given-names></name><name><surname>Zeng</surname><given-names>C</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Matsuda</surname><given-names>I</given-names></name><name><surname>Fukushima</surname><given-names>Y</given-names></name><name><surname>Macer</surname><given-names>DR</given-names></name><name><surname>Suda</surname><given-names>E</given-names></name><name><surname>Rotimi</surname><given-names>CN</given-names></name><name><surname>Adebamowo</surname><given-names>CA</given-names></name><name><surname>Ajayi</surname><given-names>I</given-names></name><name><surname>Aniagwu</surname><given-names>T</given-names></name><name><surname>Marshall</surname><given-names>PA</given-names></name><name><surname>Nkwodimmah</surname><given-names>C</given-names></name><name><surname>Royal</surname><given-names>CDM</given-names></name><name><surname>Leppert</surname><given-names>MF</given-names></name><name><surname>Dixon</surname><given-names>M</given-names></name><name><surname>Peiffer</surname><given-names>A</given-names></name><name><surname>Qiu</surname><given-names>R</given-names></name><name><surname>Kent</surname><given-names>A</given-names></name><name><surname>Kato</surname><given-names>K</given-names></name><name><surname>Niikawa</surname><given-names>N</given-names></name><name><surname>Adewole</surname><given-names>IF</given-names></name><name><surname>Knoppers</surname><given-names>BM</given-names></name><name><surname>Foster</surname><given-names>MW</given-names></name><name><surname>Clayton</surname><given-names>EW</given-names></name><name><surname>Watkin</surname><given-names>J</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Belmont</surname><given-names>JW</given-names></name><name><surname>Muzny</surname><given-names>D</given-names></name><name><surname>Nazareth</surname><given-names>L</given-names></name><name><surname>Sodergren</surname><given-names>E</given-names></name><name><surname>Weinstock</surname><given-names>GM</given-names></name><name><surname>Wheeler</surname><given-names>DA</given-names></name><name><surname>Yakub</surname><given-names>I</given-names></name><name><surname>Gabriel</surname><given-names>SB</given-names></name><name><surname>Onofrio</surname><given-names>RC</given-names></name><name><surname>Richter</surname><given-names>DJ</given-names></name><name><surname>Ziaugra</surname><given-names>L</given-names></name><name><surname>Birren</surname><given-names>BW</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Altshuler</surname><given-names>D</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Fulton</surname><given-names>LL</given-names></name><name><surname>Rogers</surname><given-names>J</given-names></name><name><surname>Burton</surname><given-names>J</given-names></name><name><surname>Carter</surname><given-names>NP</given-names></name><name><surname>Clee</surname><given-names>CM</given-names></name><name><surname>Griffiths</surname><given-names>M</given-names></name><name><surname>Jones</surname><given-names>MC</given-names></name><name><surname>McLay</surname><given-names>K</given-names></name><name><surname>Plumb</surname><given-names>RW</given-names></name><name><surname>Ross</surname><given-names>MT</given-names></name><name><surname>Sims</surname><given-names>SK</given-names></name><name><surname>Willey</surname><given-names>DL</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name><name><surname>Han</surname><given-names>H</given-names></name><name><surname>Kang</surname><given-names>L</given-names></name><name><surname>Godbout</surname><given-names>M</given-names></name><name><surname>Wallenburg</surname><given-names>JC</given-names></name><name><surname>L’Archevêque</surname><given-names>P</given-names></name><name><surname>Bellemare</surname><given-names>G</given-names></name><name><surname>Saeki</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>An</surname><given-names>D</given-names></name><name><surname>Fu</surname><given-names>H</given-names></name><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>R</given-names></name><name><surname>Holden</surname><given-names>AL</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>McEwen</surname><given-names>JE</given-names></name><name><surname>Guyer</surname><given-names>MS</given-names></name><name><surname>Wang</surname><given-names>VO</given-names></name><name><surname>Peterson</surname><given-names>JL</given-names></name><name><surname>Shi</surname><given-names>M</given-names></name><name><surname>Spiegel</surname><given-names>J</given-names></name><name><surname>Sung</surname><given-names>LM</given-names></name><name><surname>Zacharia</surname><given-names>LF</given-names></name><name><surname>Collins</surname><given-names>FS</given-names></name><name><surname>Kennedy</surname><given-names>K</given-names></name><name><surname>Jamieson</surname><given-names>R</given-names></name><name><surname>Stewart</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>A second generation human haplotype map of over 3.1 million SNPs</article-title><source>Nature</source><volume>449</volume><fpage>851</fpage><lpage>861</lpage><pub-id pub-id-type="doi">10.1038/nature06258</pub-id><pub-id pub-id-type="pmid">17943122</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>Z</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Sasani</surname><given-names>TA</given-names></name><name><surname>Pedersen</surname><given-names>BS</given-names></name><name><surname>Quinlan</surname><given-names>AR</given-names></name><name><surname>Jorde</surname><given-names>LB</given-names></name><name><surname>Amster</surname><given-names>G</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Overlooked roles of DNA damage and maternal age in generating human germline mutations</article-title><source>PNAS</source><volume>116</volume><fpage>9491</fpage><lpage>9500</lpage><pub-id pub-id-type="doi">10.1073/pnas.1901259116</pub-id><pub-id pub-id-type="pmid">31019089</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gardner</surname><given-names>EJ</given-names></name><name><surname>Prigmore</surname><given-names>E</given-names></name><name><surname>Gallone</surname><given-names>G</given-names></name><name><surname>Danecek</surname><given-names>P</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>Handsaker</surname><given-names>J</given-names></name><name><surname>Gerety</surname><given-names>SS</given-names></name><name><surname>Ironfield</surname><given-names>H</given-names></name><name><surname>Short</surname><given-names>PJ</given-names></name><name><surname>Sifrim</surname><given-names>A</given-names></name><name><surname>Singh</surname><given-names>T</given-names></name><name><surname>Chandler</surname><given-names>KE</given-names></name><name><surname>Clement</surname><given-names>E</given-names></name><name><surname>Lachlan</surname><given-names>KL</given-names></name><name><surname>Prescott</surname><given-names>K</given-names></name><name><surname>Rosser</surname><given-names>E</given-names></name><name><surname>FitzPatrick</surname><given-names>DR</given-names></name><name><surname>Firth</surname><given-names>HV</given-names></name><name><surname>Hurles</surname><given-names>ME</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Contribution of retrotransposition to developmental disorders</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>4630</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-12520-y</pub-id><pub-id pub-id-type="pmid">31604926</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gillespie</surname><given-names>JH</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Genetic drift in an infinite population. The pseudohitchhiking model</article-title><source>Genetics</source><volume>155</volume><fpage>909</fpage><lpage>919</lpage><pub-id pub-id-type="doi">10.1093/genetics/155.2.909</pub-id><pub-id pub-id-type="pmid">10835409</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Good</surname><given-names>BH</given-names></name><name><surname>Walczak</surname><given-names>AM</given-names></name><name><surname>Neher</surname><given-names>RA</given-names></name><name><surname>Desai</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genetic diversity in the interference selection limit</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004222</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004222</pub-id><pub-id pub-id-type="pmid">24675740</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gordo</surname><given-names>I</given-names></name><name><surname>Charlesworth</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>The speed of Muller’s ratchet with background selection, and the degeneration of Y chromosomes</article-title><source>Genetical Research</source><volume>78</volume><fpage>149</fpage><lpage>161</lpage><pub-id pub-id-type="doi">10.1017/s0016672301005213</pub-id><pub-id pub-id-type="pmid">11732092</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gordo</surname><given-names>I</given-names></name><name><surname>Navarro</surname><given-names>A</given-names></name><name><surname>Charlesworth</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Muller’s ratchet and the pattern of variation at a neutral locus</article-title><source>Genetics</source><volume>161</volume><fpage>835</fpage><lpage>848</lpage><pub-id pub-id-type="doi">10.1093/genetics/161.2.835</pub-id><pub-id pub-id-type="pmid">12072478</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Green</surname><given-names>RE</given-names></name><name><surname>Krause</surname><given-names>J</given-names></name><name><surname>Briggs</surname><given-names>AW</given-names></name><name><surname>Maricic</surname><given-names>T</given-names></name><name><surname>Stenzel</surname><given-names>U</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Zhai</surname><given-names>W</given-names></name><name><surname>Fritz</surname><given-names>MHY</given-names></name><name><surname>Hansen</surname><given-names>NF</given-names></name><name><surname>Durand</surname><given-names>EY</given-names></name><name><surname>Malaspinas</surname><given-names>AS</given-names></name><name><surname>Jensen</surname><given-names>JD</given-names></name><name><surname>Marques-Bonet</surname><given-names>T</given-names></name><name><surname>Alkan</surname><given-names>C</given-names></name><name><surname>Prüfer</surname><given-names>K</given-names></name><name><surname>Meyer</surname><given-names>M</given-names></name><name><surname>Burbano</surname><given-names>HA</given-names></name><name><surname>Good</surname><given-names>JM</given-names></name><name><surname>Schultz</surname><given-names>R</given-names></name><name><surname>Aximu-Petri</surname><given-names>A</given-names></name><name><surname>Butthof</surname><given-names>A</given-names></name><name><surname>Höber</surname><given-names>B</given-names></name><name><surname>Höffner</surname><given-names>B</given-names></name><name><surname>Siegemund</surname><given-names>M</given-names></name><name><surname>Weihmann</surname><given-names>A</given-names></name><name><surname>Nusbaum</surname><given-names>C</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Russ</surname><given-names>C</given-names></name><name><surname>Novod</surname><given-names>N</given-names></name><name><surname>Affourtit</surname><given-names>J</given-names></name><name><surname>Egholm</surname><given-names>M</given-names></name><name><surname>Verna</surname><given-names>C</given-names></name><name><surname>Rudan</surname><given-names>P</given-names></name><name><surname>Brajkovic</surname><given-names>D</given-names></name><name><surname>Kucan</surname><given-names>Ž</given-names></name><name><surname>Gušic</surname><given-names>I</given-names></name><name><surname>Doronichev</surname><given-names>VB</given-names></name><name><surname>Golovanova</surname><given-names>LV</given-names></name><name><surname>Lalueza-Fox</surname><given-names>C</given-names></name><name><surname>de la Rasilla</surname><given-names>M</given-names></name><name><surname>Fortea</surname><given-names>J</given-names></name><name><surname>Rosas</surname><given-names>A</given-names></name><name><surname>Schmitz</surname><given-names>RW</given-names></name><name><surname>Johnson</surname><given-names>PLF</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Falush</surname><given-names>D</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Mullikin</surname><given-names>JC</given-names></name><name><surname>Slatkin</surname><given-names>M</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Kelso</surname><given-names>J</given-names></name><name><surname>Lachmann</surname><given-names>M</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name><name><surname>Pääbo</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A draft sequence of the Neandertal genome</article-title><source>Science</source><volume>328</volume><fpage>710</fpage><lpage>722</lpage><pub-id pub-id-type="doi">10.1126/science.1188021</pub-id><pub-id pub-id-type="pmid">20448178</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Halldorsson</surname><given-names>BV</given-names></name><name><surname>Palsson</surname><given-names>G</given-names></name><name><surname>Stefansson</surname><given-names>OA</given-names></name><name><surname>Jonsson</surname><given-names>H</given-names></name><name><surname>Hardarson</surname><given-names>MT</given-names></name><name><surname>Eggertsson</surname><given-names>HP</given-names></name><name><surname>Gunnarsson</surname><given-names>B</given-names></name><name><surname>Oddsson</surname><given-names>A</given-names></name><name><surname>Halldorsson</surname><given-names>GH</given-names></name><name><surname>Zink</surname><given-names>F</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Frigge</surname><given-names>ML</given-names></name><name><surname>Thorleifsson</surname><given-names>G</given-names></name><name><surname>Sigurdsson</surname><given-names>A</given-names></name><name><surname>Stacey</surname><given-names>SN</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Characterizing mutagenic effects of recombination through a sequence-level genetic map</article-title><source>Science</source><volume>363</volume><elocation-id>eaau1043</elocation-id><pub-id pub-id-type="doi">10.1126/science.aau1043</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname><given-names>K</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The genetic cost of Neanderthal introgression</article-title><source>Genetics</source><volume>203</volume><fpage>881</fpage><lpage>891</lpage><pub-id pub-id-type="doi">10.1534/genetics.116.186890</pub-id><pub-id pub-id-type="pmid">27038113</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hayward</surname><given-names>LK</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Polygenic adaptation after a sudden change in environment</article-title><source>eLife</source><volume>11</volume><elocation-id>e66697</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.66697</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hermisson</surname><given-names>J</given-names></name><name><surname>Pennings</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Soft sweeps: molecular population genetics of adaptation from standing genetic variation</article-title><source>Genetics</source><volume>169</volume><fpage>2335</fpage><lpage>2352</lpage><pub-id pub-id-type="doi">10.1534/genetics.104.036947</pub-id><pub-id pub-id-type="pmid">15716498</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hernandez</surname><given-names>RD</given-names></name><name><surname>Kelley</surname><given-names>JL</given-names></name><name><surname>Elyashiv</surname><given-names>E</given-names></name><name><surname>Melton</surname><given-names>SC</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name><collab>1000 Genomes Project</collab></person-group><year iso-8601-date="2011">2011</year><article-title>Classic selective sweeps were rare in recent human evolution</article-title><source>Science</source><volume>331</volume><fpage>920</fpage><lpage>924</lpage><pub-id pub-id-type="doi">10.1126/science.1198878</pub-id><pub-id pub-id-type="pmid">21330547</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname><given-names>WG</given-names></name><name><surname>Robertson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1966">1966</year><article-title>The effect of linkage on limits to artificial selection</article-title><source>Genetical Research</source><volume>8</volume><fpage>269</fpage><lpage>294</lpage><pub-id pub-id-type="doi">10.1017/S0016672300010156</pub-id><pub-id pub-id-type="pmid">5980116</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hinch</surname><given-names>AG</given-names></name><name><surname>Tandon</surname><given-names>A</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Song</surname><given-names>Y</given-names></name><name><surname>Rohland</surname><given-names>N</given-names></name><name><surname>Palmer</surname><given-names>CD</given-names></name><name><surname>Chen</surname><given-names>GK</given-names></name><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Buxbaum</surname><given-names>SG</given-names></name><name><surname>Akylbekova</surname><given-names>EL</given-names></name><name><surname>Aldrich</surname><given-names>MC</given-names></name><name><surname>Ambrosone</surname><given-names>CB</given-names></name><name><surname>Amos</surname><given-names>C</given-names></name><name><surname>Bandera</surname><given-names>EV</given-names></name><name><surname>Berndt</surname><given-names>SI</given-names></name><name><surname>Bernstein</surname><given-names>L</given-names></name><name><surname>Blot</surname><given-names>WJ</given-names></name><name><surname>Bock</surname><given-names>CH</given-names></name><name><surname>Boerwinkle</surname><given-names>E</given-names></name><name><surname>Cai</surname><given-names>Q</given-names></name><name><surname>Caporaso</surname><given-names>N</given-names></name><name><surname>Casey</surname><given-names>G</given-names></name><name><surname>Cupples</surname><given-names>LA</given-names></name><name><surname>Deming</surname><given-names>SL</given-names></name><name><surname>Diver</surname><given-names>WR</given-names></name><name><surname>Divers</surname><given-names>J</given-names></name><name><surname>Fornage</surname><given-names>M</given-names></name><name><surname>Gillanders</surname><given-names>EM</given-names></name><name><surname>Glessner</surname><given-names>J</given-names></name><name><surname>Harris</surname><given-names>CC</given-names></name><name><surname>Hu</surname><given-names>JJ</given-names></name><name><surname>Ingles</surname><given-names>SA</given-names></name><name><surname>Isaacs</surname><given-names>W</given-names></name><name><surname>John</surname><given-names>EM</given-names></name><name><surname>Kao</surname><given-names>WHL</given-names></name><name><surname>Keating</surname><given-names>B</given-names></name><name><surname>Kittles</surname><given-names>RA</given-names></name><name><surname>Kolonel</surname><given-names>LN</given-names></name><name><surname>Larkin</surname><given-names>E</given-names></name><name><surname>Le Marchand</surname><given-names>L</given-names></name><name><surname>McNeill</surname><given-names>LH</given-names></name><name><surname>Millikan</surname><given-names>RC</given-names></name><name><surname>Murphy</surname><given-names>A</given-names></name><name><surname>Musani</surname><given-names>S</given-names></name><name><surname>Neslund-Dudas</surname><given-names>C</given-names></name><name><surname>Nyante</surname><given-names>S</given-names></name><name><surname>Papanicolaou</surname><given-names>GJ</given-names></name><name><surname>Press</surname><given-names>MF</given-names></name><name><surname>Psaty</surname><given-names>BM</given-names></name><name><surname>Reiner</surname><given-names>AP</given-names></name><name><surname>Rich</surname><given-names>SS</given-names></name><name><surname>Rodriguez-Gil</surname><given-names>JL</given-names></name><name><surname>Rotter</surname><given-names>JI</given-names></name><name><surname>Rybicki</surname><given-names>BA</given-names></name><name><surname>Schwartz</surname><given-names>AG</given-names></name><name><surname>Signorello</surname><given-names>LB</given-names></name><name><surname>Spitz</surname><given-names>M</given-names></name><name><surname>Strom</surname><given-names>SS</given-names></name><name><surname>Thun</surname><given-names>MJ</given-names></name><name><surname>Tucker</surname><given-names>MA</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Wiencke</surname><given-names>JK</given-names></name><name><surname>Witte</surname><given-names>JS</given-names></name><name><surname>Wrensch</surname><given-names>M</given-names></name><name><surname>Wu</surname><given-names>X</given-names></name><name><surname>Yamamura</surname><given-names>Y</given-names></name><name><surname>Zanetti</surname><given-names>KA</given-names></name><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Ziegler</surname><given-names>RG</given-names></name><name><surname>Zhu</surname><given-names>X</given-names></name><name><surname>Redline</surname><given-names>S</given-names></name><name><surname>Hirschhorn</surname><given-names>JN</given-names></name><name><surname>Henderson</surname><given-names>BE</given-names></name><name><surname>Taylor</surname><given-names>HA</given-names><suffix>Jr</suffix></name><name><surname>Price</surname><given-names>AL</given-names></name><name><surname>Hakonarson</surname><given-names>H</given-names></name><name><surname>Chanock</surname><given-names>SJ</given-names></name><name><surname>Haiman</surname><given-names>CA</given-names></name><name><surname>Wilson</surname><given-names>JG</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name><name><surname>Myers</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The landscape of recombination in African Americans</article-title><source>Nature</source><volume>476</volume><fpage>170</fpage><lpage>175</lpage><pub-id pub-id-type="doi">10.1038/nature10336</pub-id><pub-id pub-id-type="pmid">21775986</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hsu</surname><given-names>F</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Kuhn</surname><given-names>RM</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The UCSC known genes</article-title><source>Bioinformatics</source><volume>22</volume><fpage>1036</fpage><lpage>1046</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btl048</pub-id><pub-id pub-id-type="pmid">16500937</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hudson</surname><given-names>RR</given-names></name></person-group><year iso-8601-date="1990">1990</year><source>Oxford Surveys in Evolutionary Biology</source><publisher-loc>Oxford, UK</publisher-loc><publisher-name>Oxford University Press</publisher-name><pub-id pub-id-type="doi">10.1002/ajpa.1330930314</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hudson</surname><given-names>RR</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>How can the low levels of DNA sequence variation in regions of the <italic>Drosophila</italic> genome with low recombination rates be explained?</article-title><source>PNAS</source><volume>91</volume><fpage>6815</fpage><lpage>6818</lpage><pub-id pub-id-type="doi">10.1073/pnas.91.15.6815</pub-id><pub-id pub-id-type="pmid">8041702</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hudson</surname><given-names>RR</given-names></name><name><surname>Kaplan</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Deleterious background selection with recombination</article-title><source>Genetics</source><volume>141</volume><fpage>1605</fpage><lpage>1617</lpage><pub-id pub-id-type="doi">10.1093/genetics/141.4.1605</pub-id><pub-id pub-id-type="pmid">8601498</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hudson</surname><given-names>RR</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Two-Locus sampling distributions and their application</article-title><source>Genetics</source><volume>159</volume><fpage>1805</fpage><lpage>1817</lpage><pub-id pub-id-type="doi">10.1093/genetics/159.4.1805</pub-id><pub-id pub-id-type="pmid">11779816</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jónsson</surname><given-names>H</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Kehr</surname><given-names>B</given-names></name><name><surname>Kristmundsdottir</surname><given-names>S</given-names></name><name><surname>Zink</surname><given-names>F</given-names></name><name><surname>Hjartarson</surname><given-names>E</given-names></name><name><surname>Hardarson</surname><given-names>MT</given-names></name><name><surname>Hjorleifsson</surname><given-names>KE</given-names></name><name><surname>Eggertsson</surname><given-names>HP</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Ward</surname><given-names>LD</given-names></name><name><surname>Arnadottir</surname><given-names>GA</given-names></name><name><surname>Helgason</surname><given-names>EA</given-names></name><name><surname>Helgason</surname><given-names>H</given-names></name><name><surname>Gylfason</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Rafnar</surname><given-names>T</given-names></name><name><surname>Frigge</surname><given-names>M</given-names></name><name><surname>Stacey</surname><given-names>SN</given-names></name><name><surname>Th Magnusson</surname><given-names>O</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Kong</surname><given-names>A</given-names></name><name><surname>Halldorsson</surname><given-names>BV</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Parental influence on human germline <italic>de novo</italic> mutations in 1,548 trios from Iceland</article-title><source>Nature</source><volume>549</volume><fpage>519</fpage><lpage>522</lpage><pub-id pub-id-type="doi">10.1038/nature24018</pub-id><pub-id pub-id-type="pmid">28959963</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Juric</surname><given-names>I</given-names></name><name><surname>Aeschbacher</surname><given-names>S</given-names></name><name><surname>Coop</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The strength of selection against Neanderthal introgression</article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1006340</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1006340</pub-id><pub-id pub-id-type="pmid">27824859</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaplan</surname><given-names>NL</given-names></name><name><surname>Hudson</surname><given-names>RR</given-names></name><name><surname>Langley</surname><given-names>CH</given-names></name></person-group><year iso-8601-date="1989">1989</year><article-title>The “hitchhiking effect” revisited</article-title><source>Genetics</source><volume>123</volume><fpage>887</fpage><lpage>899</lpage><pub-id pub-id-type="doi">10.1093/genetics/123.4.887</pub-id><pub-id pub-id-type="pmid">2612899</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karolchik</surname><given-names>D</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Furey</surname><given-names>TS</given-names></name><name><surname>Roskin</surname><given-names>KM</given-names></name><name><surname>Sugnet</surname><given-names>CW</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>The UCSC table browser data retrieval tool</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>D493</fpage><lpage>D496</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh103</pub-id><pub-id pub-id-type="pmid">14681465</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Wold</surname><given-names>B</given-names></name><name><surname>Snyder</surname><given-names>MP</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Marinov</surname><given-names>GK</given-names></name><name><surname>Hardison</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Defining functional DNA elements in the human genome</article-title><source>PNAS</source><volume>111</volume><fpage>6131</fpage><lpage>6138</lpage><pub-id pub-id-type="doi">10.1073/pnas.1318948111</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>Y</given-names></name><name><surname>Stephan</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Selective sweeps in the presence of interference among partially linked loci</article-title><source>Genetics</source><volume>164</volume><fpage>389</fpage><lpage>398</lpage><pub-id pub-id-type="doi">10.1093/genetics/164.1.389</pub-id><pub-id pub-id-type="pmid">12750349</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>TH</given-names></name><name><surname>Barrera</surname><given-names>LO</given-names></name><name><surname>Zheng</surname><given-names>M</given-names></name><name><surname>Qu</surname><given-names>C</given-names></name><name><surname>Singer</surname><given-names>MA</given-names></name><name><surname>Richmond</surname><given-names>TA</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>A high-resolution map of active promoters in the human genome</article-title><source>Nature</source><volume>436</volume><fpage>876</fpage><lpage>880</lpage><pub-id pub-id-type="doi">10.1038/nature03877</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kimura</surname><given-names>M</given-names></name><name><surname>Crow</surname><given-names>JF</given-names></name></person-group><year iso-8601-date="1964">1964</year><article-title>The number of alleles that can be maintained in a finite population</article-title><source>Genetics</source><volume>49</volume><fpage>725</fpage><lpage>738</lpage><pub-id pub-id-type="doi">10.1093/genetics/49.4.725</pub-id><pub-id pub-id-type="pmid">14156929</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kircher</surname><given-names>M</given-names></name><name><surname>Witten</surname><given-names>DM</given-names></name><name><surname>Jain</surname><given-names>P</given-names></name><name><surname>O’Roak</surname><given-names>BJ</given-names></name><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A general framework for estimating the relative pathogenicity of human genetic variants</article-title><source>Nature Genetics</source><volume>46</volume><fpage>310</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.1038/ng.2892</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kong</surname><given-names>A</given-names></name><name><surname>Frigge</surname><given-names>ML</given-names></name><name><surname>Masson</surname><given-names>G</given-names></name><name><surname>Besenbacher</surname><given-names>S</given-names></name><name><surname>Sulem</surname><given-names>P</given-names></name><name><surname>Magnusson</surname><given-names>G</given-names></name><name><surname>Gudjonsson</surname><given-names>SA</given-names></name><name><surname>Sigurdsson</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Jonasdottir</surname><given-names>A</given-names></name><name><surname>Wong</surname><given-names>WSW</given-names></name><name><surname>Sigurdsson</surname><given-names>G</given-names></name><name><surname>Walters</surname><given-names>GB</given-names></name><name><surname>Steinberg</surname><given-names>S</given-names></name><name><surname>Helgason</surname><given-names>H</given-names></name><name><surname>Thorleifsson</surname><given-names>G</given-names></name><name><surname>Gudbjartsson</surname><given-names>DF</given-names></name><name><surname>Helgason</surname><given-names>A</given-names></name><name><surname>Magnusson</surname><given-names>OT</given-names></name><name><surname>Thorsteinsdottir</surname><given-names>U</given-names></name><name><surname>Stefansson</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Rate of <italic>de novo</italic> mutations and the importance of father’s age to disease risk</article-title><source>Nature</source><volume>488</volume><fpage>471</fpage><lpage>475</lpage><pub-id pub-id-type="doi">10.1038/nature11396</pub-id><pub-id pub-id-type="pmid">22914163</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>R</given-names></name><name><surname>Bitoun</surname><given-names>E</given-names></name><name><surname>Altemose</surname><given-names>N</given-names></name><name><surname>Davies</surname><given-names>RW</given-names></name><name><surname>Davies</surname><given-names>B</given-names></name><name><surname>Myers</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A high-resolution map of non-crossover events reveals impacts of genetic diversity on mammalian meiotic recombination</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>3900</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-11675-y</pub-id><pub-id pub-id-type="pmid">31467277</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Macpherson</surname><given-names>JM</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Davis</surname><given-names>JC</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Genomewide spatial correspondence between nonsynonymous divergence and neutral polymorphism reveals extensive adaptation in <italic>Drosophila</italic></article-title><source>Genetics</source><volume>177</volume><fpage>2083</fpage><lpage>2099</lpage><pub-id pub-id-type="doi">10.1534/genetics.107.080226</pub-id><pub-id pub-id-type="pmid">18073425</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVean</surname><given-names>GA</given-names></name><name><surname>Charlesworth</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The effects of Hill-Robertson interference between weakly selected mutations on patterns of molecular evolution and variation</article-title><source>Genetics</source><volume>155</volume><fpage>929</fpage><lpage>944</lpage><pub-id pub-id-type="doi">10.1093/genetics/155.2.929</pub-id><pub-id pub-id-type="pmid">10835411</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVicker</surname><given-names>G</given-names></name><name><surname>Gordon</surname><given-names>D</given-names></name><name><surname>Davis</surname><given-names>C</given-names></name><name><surname>Green</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Widespread genomic signatures of natural selection in hominid evolution</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000471</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000471</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname><given-names>JE</given-names></name><name><surname>Purcaro</surname><given-names>MJ</given-names></name><name><surname>Pratt</surname><given-names>HE</given-names></name><name><surname>Epstein</surname><given-names>CB</given-names></name><name><surname>Shoresh</surname><given-names>N</given-names></name><name><surname>Adrian</surname><given-names>J</given-names></name><name><surname>Kawli</surname><given-names>T</given-names></name><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Dobin</surname><given-names>A</given-names></name><name><surname>Kaul</surname><given-names>R</given-names></name><name><surname>Halow</surname><given-names>J</given-names></name><name><surname>Van Nostrand</surname><given-names>EL</given-names></name><name><surname>Freese</surname><given-names>P</given-names></name><name><surname>Gorkin</surname><given-names>DU</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Mackiewicz</surname><given-names>M</given-names></name><name><surname>Pauli-Behn</surname><given-names>F</given-names></name><name><surname>Williams</surname><given-names>BA</given-names></name><name><surname>Mortazavi</surname><given-names>A</given-names></name><name><surname>Keller</surname><given-names>CA</given-names></name><name><surname>Zhang</surname><given-names>X-O</given-names></name><name><surname>Elhajjajy</surname><given-names>SI</given-names></name><name><surname>Huey</surname><given-names>J</given-names></name><name><surname>Dickel</surname><given-names>DE</given-names></name><name><surname>Snetkova</surname><given-names>V</given-names></name><name><surname>Wei</surname><given-names>X</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Rivera-Mulia</surname><given-names>JC</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Chhetri</surname><given-names>SB</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Victorsen</surname><given-names>A</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Visel</surname><given-names>A</given-names></name><name><surname>Yeo</surname><given-names>GW</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name><name><surname>Lécuyer</surname><given-names>E</given-names></name><name><surname>Gilbert</surname><given-names>DM</given-names></name><name><surname>Dekker</surname><given-names>J</given-names></name><name><surname>Rinn</surname><given-names>J</given-names></name><name><surname>Mendenhall</surname><given-names>EM</given-names></name><name><surname>Ecker</surname><given-names>JR</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Klein</surname><given-names>RJ</given-names></name><name><surname>Noble</surname><given-names>WS</given-names></name><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Guigó</surname><given-names>R</given-names></name><name><surname>Farnham</surname><given-names>PJ</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Graveley</surname><given-names>BR</given-names></name><name><surname>Gerstein</surname><given-names>MB</given-names></name><name><surname>Pennacchio</surname><given-names>LA</given-names></name><name><surname>Snyder</surname><given-names>MP</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name><name><surname>Wold</surname><given-names>B</given-names></name><name><surname>Hardison</surname><given-names>RC</given-names></name><name><surname>Gingeras</surname><given-names>TR</given-names></name><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name><collab>ENCODE Project Consortium</collab></person-group><year iso-8601-date="2020">2020</year><article-title>Expanded encyclopaedias of DNA elements in the human and mouse genomes</article-title><source>Nature</source><volume>583</volume><fpage>699</fpage><lpage>710</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2493-4</pub-id><pub-id pub-id-type="pmid">32728249</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Murphy</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>B maps and code for running linked selection inference on the human genome</data-title><version designator="swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4">swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:b177485acbb8bc94742060ab3a7a443a473b3271;origin=https://github.com/sellalab/HumanLinkedSelectionMaps;visit=swh:1:snp:c14f688b4c7fdc1e530c1b9fca0debc45f00dcb4;anchor=swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4">https://archive.softwareheritage.org/swh:1:dir:b177485acbb8bc94742060ab3a7a443a473b3271;origin=https://github.com/sellalab/HumanLinkedSelectionMaps;visit=swh:1:snp:c14f688b4c7fdc1e530c1b9fca0debc45f00dcb4;anchor=swh:1:rev:c09a98ac4c82e7d1c9c5d1cc7c283b13dca76db4</ext-link></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Myers</surname><given-names>S</given-names></name><name><surname>Bottolo</surname><given-names>L</given-names></name><name><surname>Freeman</surname><given-names>C</given-names></name><name><surname>McVean</surname><given-names>G</given-names></name><name><surname>Donnelly</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>A fine-scale map of recombination rates and hotspots across the human genome</article-title><source>Science</source><volume>310</volume><fpage>321</fpage><lpage>324</lpage><pub-id pub-id-type="doi">10.1126/science.1117196</pub-id><pub-id pub-id-type="pmid">16224025</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nachman</surname><given-names>MW</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Patterns of DNA variability at X-linked loci in <italic>Mus domesticus</italic></article-title><source>Genetics</source><volume>147</volume><fpage>1303</fpage><lpage>1316</lpage><pub-id pub-id-type="doi">10.1093/genetics/147.3.1303</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nelder</surname><given-names>JA</given-names></name><name><surname>Mead</surname><given-names>R</given-names></name></person-group><year iso-8601-date="1965">1965</year><article-title>A simplex method for function minimization</article-title><source>The Computer Journal</source><volume>7</volume><fpage>308</fpage><lpage>313</lpage><pub-id pub-id-type="doi">10.1093/comjnl/7.4.308</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nordborg</surname><given-names>M</given-names></name><name><surname>Charlesworth</surname><given-names>B</given-names></name><name><surname>Charlesworth</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>The effect of recombination on background selection</article-title><source>Genetical Research</source><volume>67</volume><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.1017/S0016672300033619</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nordborg</surname><given-names>M</given-names></name><name><surname>Hu</surname><given-names>TT</given-names></name><name><surname>Ishino</surname><given-names>Y</given-names></name><name><surname>Jhaveri</surname><given-names>J</given-names></name><name><surname>Toomajian</surname><given-names>C</given-names></name><name><surname>Zheng</surname><given-names>H</given-names></name><name><surname>Bakker</surname><given-names>E</given-names></name><name><surname>Calabrese</surname><given-names>P</given-names></name><name><surname>Gladstone</surname><given-names>J</given-names></name><name><surname>Goyal</surname><given-names>R</given-names></name><name><surname>Jakobsson</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Morozov</surname><given-names>Y</given-names></name><name><surname>Padhukasahasram</surname><given-names>B</given-names></name><name><surname>Plagnol</surname><given-names>V</given-names></name><name><surname>Rosenberg</surname><given-names>NA</given-names></name><name><surname>Shah</surname><given-names>C</given-names></name><name><surname>Wall</surname><given-names>JD</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Zhao</surname><given-names>K</given-names></name><name><surname>Kalbfleisch</surname><given-names>T</given-names></name><name><surname>Schulz</surname><given-names>V</given-names></name><name><surname>Kreitman</surname><given-names>M</given-names></name><name><surname>Bergelson</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The pattern of polymorphism in <italic>Arabidopsis thaliana</italic></article-title><source>PLOS Biology</source><volume>3</volume><elocation-id>e196</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0030196</pub-id><pub-id pub-id-type="pmid">15907155</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paten</surname><given-names>B</given-names></name><name><surname>Herrero</surname><given-names>J</given-names></name><name><surname>Fitzgerald</surname><given-names>S</given-names></name><name><surname>Beal</surname><given-names>K</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Holmes</surname><given-names>I</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Genome-Wide nucleotide-level mammalian ancestor reconstruction</article-title><source>Genome Research</source><volume>18</volume><fpage>1829</fpage><lpage>1843</lpage><pub-id pub-id-type="doi">10.1101/gr.076521.108</pub-id><pub-id pub-id-type="pmid">18849525</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Rohland</surname><given-names>N</given-names></name><name><surname>Zhan</surname><given-names>Y</given-names></name><name><surname>Genschoreck</surname><given-names>T</given-names></name><name><surname>Webster</surname><given-names>T</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Ancient admixture in human history</article-title><source>Genetics</source><volume>192</volume><fpage>1065</fpage><lpage>1093</lpage><pub-id pub-id-type="doi">10.1534/genetics.112.145037</pub-id><pub-id pub-id-type="pmid">22960212</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Payseur</surname><given-names>BA</given-names></name><name><surname>Nachman</surname><given-names>MW</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Gene density and human nucleotide polymorphism</article-title><source>Molecular Biology and Evolution</source><volume>19</volume><fpage>336</fpage><lpage>340</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a004086</pub-id><pub-id pub-id-type="pmid">11861892</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pennings</surname><given-names>PS</given-names></name><name><surname>Hermisson</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2006">2006a</year><article-title>Soft sweeps II — molecular population genetics of adaptation from recurrent mutation or migration</article-title><source>Molecular Biology and Evolution</source><volume>23</volume><fpage>1076</fpage><lpage>1084</lpage><pub-id pub-id-type="doi">10.1093/molbev/msj117</pub-id><pub-id pub-id-type="pmid">16520336</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pennings</surname><given-names>PS</given-names></name><name><surname>Hermisson</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2006">2006b</year><article-title>Soft sweeps III: the signature of positive selection from recurrent mutation</article-title><source>PLOS Genetics</source><volume>2</volume><elocation-id>e186</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.0020186</pub-id><pub-id pub-id-type="pmid">17173482</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Plagnol</surname><given-names>V</given-names></name><name><surname>Wall</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Possible ancestral structure in human populations</article-title><source>PLOS Genetics</source><volume>2</volume><elocation-id>e105</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.0020105</pub-id><pub-id pub-id-type="pmid">16895447</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pouyet</surname><given-names>F</given-names></name><name><surname>Aeschbacher</surname><given-names>S</given-names></name><name><surname>Thiéry</surname><given-names>A</given-names></name><name><surname>Excoffier</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Background selection and biased gene conversion affect more than 95 % of the human genome and bias demographic inferences</article-title><source>eLife</source><volume>7</volume><elocation-id>e36317</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.36317</pub-id><pub-id pub-id-type="pmid">30125248</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pritchard</surname><given-names>JK</given-names></name><name><surname>Di Rienzo</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Adaptation-not by sweeps alone</article-title><source>Nature Reviews Genetics</source><volume>11</volume><fpage>665</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1038/nrg2880</pub-id><pub-id pub-id-type="pmid">20838407</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pritchard</surname><given-names>JK</given-names></name><name><surname>Pickrell</surname><given-names>JK</given-names></name><name><surname>Coop</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The genetics of human adaptation: hard sweeps, soft sweeps, and polygenic adaptation</article-title><source>Current Biology</source><volume>20</volume><fpage>R208</fpage><lpage>R215</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2009.11.055</pub-id><pub-id pub-id-type="pmid">20178769</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Prüfer</surname><given-names>K</given-names></name><name><surname>Racimo</surname><given-names>F</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Jay</surname><given-names>F</given-names></name><name><surname>Sankararaman</surname><given-names>S</given-names></name><name><surname>Sawyer</surname><given-names>S</given-names></name><name><surname>Heinze</surname><given-names>A</given-names></name><name><surname>Renaud</surname><given-names>G</given-names></name><name><surname>Sudmant</surname><given-names>PH</given-names></name><name><surname>de Filippo</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Dannemann</surname><given-names>M</given-names></name><name><surname>Fu</surname><given-names>Q</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name><name><surname>Kuhlwilm</surname><given-names>M</given-names></name><name><surname>Lachmann</surname><given-names>M</given-names></name><name><surname>Meyer</surname><given-names>M</given-names></name><name><surname>Ongyerth</surname><given-names>M</given-names></name><name><surname>Siebauer</surname><given-names>M</given-names></name><name><surname>Theunert</surname><given-names>C</given-names></name><name><surname>Tandon</surname><given-names>A</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Pickrell</surname><given-names>J</given-names></name><name><surname>Mullikin</surname><given-names>JC</given-names></name><name><surname>Vohr</surname><given-names>SH</given-names></name><name><surname>Green</surname><given-names>RE</given-names></name><name><surname>Hellmann</surname><given-names>I</given-names></name><name><surname>Johnson</surname><given-names>PLF</given-names></name><name><surname>Blanche</surname><given-names>H</given-names></name><name><surname>Cann</surname><given-names>H</given-names></name><name><surname>Kitzman</surname><given-names>JO</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Lein</surname><given-names>ES</given-names></name><name><surname>Bakken</surname><given-names>TE</given-names></name><name><surname>Golovanova</surname><given-names>LV</given-names></name><name><surname>Doronichev</surname><given-names>VB</given-names></name><name><surname>Shunkov</surname><given-names>MV</given-names></name><name><surname>Derevianko</surname><given-names>AP</given-names></name><name><surname>Viola</surname><given-names>B</given-names></name><name><surname>Slatkin</surname><given-names>M</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name><name><surname>Kelso</surname><given-names>J</given-names></name><name><surname>Pääbo</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The complete genome sequence of a Neanderthal from the Altai Mountains</article-title><source>Nature</source><volume>505</volume><fpage>43</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1038/nature12886</pub-id><pub-id pub-id-type="pmid">24352235</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Przeworski</surname><given-names>M</given-names></name><name><surname>Coop</surname><given-names>G</given-names></name><name><surname>Wall</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The signature of positive selection on standing genetic variation</article-title><source>Evolution; International Journal of Organic Evolution</source><volume>59</volume><fpage>2312</fpage><lpage>2323</lpage><pub-id pub-id-type="doi">10.1554/05-273.1</pub-id><pub-id pub-id-type="pmid">16396172</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Racimo</surname><given-names>F</given-names></name><name><surname>Sankararaman</surname><given-names>S</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Huerta-Sánchez</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Evidence for archaic adaptive introgression in humans</article-title><source>Nature Reviews Genetics</source><volume>16</volume><fpage>359</fpage><lpage>371</lpage><pub-id pub-id-type="doi">10.1038/nrg3936</pub-id><pub-id pub-id-type="pmid">25963373</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rands</surname><given-names>CM</given-names></name><name><surname>Meader</surname><given-names>S</given-names></name><name><surname>Ponting</surname><given-names>CP</given-names></name><name><surname>Lunter</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>8.2 % of the human genome is constrained: variation in rates of turnover across functional element classes in the human lineage</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004525</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004525</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reich</surname><given-names>D</given-names></name><name><surname>Green</surname><given-names>RE</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name><name><surname>Krause</surname><given-names>J</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Durand</surname><given-names>EY</given-names></name><name><surname>Viola</surname><given-names>B</given-names></name><name><surname>Briggs</surname><given-names>AW</given-names></name><name><surname>Stenzel</surname><given-names>U</given-names></name><name><surname>Johnson</surname><given-names>PLF</given-names></name><name><surname>Maricic</surname><given-names>T</given-names></name><name><surname>Good</surname><given-names>JM</given-names></name><name><surname>Marques-Bonet</surname><given-names>T</given-names></name><name><surname>Alkan</surname><given-names>C</given-names></name><name><surname>Fu</surname><given-names>Q</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Meyer</surname><given-names>M</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Stoneking</surname><given-names>M</given-names></name><name><surname>Richards</surname><given-names>M</given-names></name><name><surname>Talamo</surname><given-names>S</given-names></name><name><surname>Shunkov</surname><given-names>MV</given-names></name><name><surname>Derevianko</surname><given-names>AP</given-names></name><name><surname>Hublin</surname><given-names>J-J</given-names></name><name><surname>Kelso</surname><given-names>J</given-names></name><name><surname>Slatkin</surname><given-names>M</given-names></name><name><surname>Pääbo</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Genetic history of an archaic hominin group from Denisova Cave in Siberia</article-title><source>Nature</source><volume>468</volume><fpage>1053</fpage><lpage>1060</lpage><pub-id pub-id-type="doi">10.1038/nature09710</pub-id><pub-id pub-id-type="pmid">21179161</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rentzsch</surname><given-names>P</given-names></name><name><surname>Witten</surname><given-names>D</given-names></name><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Cadd: predicting the deleteriousness of variants throughout the human genome</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D886</fpage><lpage>D894</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1016</pub-id><pub-id pub-id-type="pmid">30371827</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robertson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1966">1966</year><article-title>A mathematical model of the culling process in dairy cattle</article-title><source>Animal Science</source><volume>8</volume><fpage>95</fpage><lpage>108</lpage><pub-id pub-id-type="doi">10.1017/S0003356100037752</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sankararaman</surname><given-names>S</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Dannemann</surname><given-names>M</given-names></name><name><surname>Prüfer</surname><given-names>K</given-names></name><name><surname>Kelso</surname><given-names>J</given-names></name><name><surname>Pääbo</surname><given-names>S</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The genomic landscape of Neanderthal ancestry in present-day humans</article-title><source>Nature</source><volume>507</volume><fpage>354</fpage><lpage>357</lpage><pub-id pub-id-type="doi">10.1038/nature12961</pub-id><pub-id pub-id-type="pmid">24476815</pub-id></element-citation></ref><ref id="bib93"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sankararaman</surname><given-names>S.</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Reich</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The combined landscape of Denisovan and Neanderthal ancestry in present-day humans</article-title><source>Current Biology</source><volume>26</volume><fpage>1241</fpage><lpage>1247</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2016.03.037</pub-id><pub-id pub-id-type="pmid">27032491</pub-id></element-citation></ref><ref id="bib94"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schiffels</surname><given-names>S</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Inferring human population size and separation history from multiple genome sequences</article-title><source>Nature Genetics</source><volume>46</volume><fpage>919</fpage><lpage>925</lpage><pub-id pub-id-type="doi">10.1038/ng.3015</pub-id><pub-id pub-id-type="pmid">24952747</pub-id></element-citation></ref><ref id="bib95"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schumer</surname><given-names>M</given-names></name><name><surname>Xu</surname><given-names>C</given-names></name><name><surname>Powell</surname><given-names>DL</given-names></name><name><surname>Durvasula</surname><given-names>A</given-names></name><name><surname>Skov</surname><given-names>L</given-names></name><name><surname>Holland</surname><given-names>C</given-names></name><name><surname>Blazier</surname><given-names>JC</given-names></name><name><surname>Sankararaman</surname><given-names>S</given-names></name><name><surname>Andolfatto</surname><given-names>P</given-names></name><name><surname>Rosenthal</surname><given-names>GG</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Natural selection interacts with recombination to shape the evolution of hybrid genomes</article-title><source>Science</source><volume>360</volume><fpage>656</fpage><lpage>660</lpage><pub-id pub-id-type="doi">10.1126/science.aar3684</pub-id><pub-id pub-id-type="pmid">29674434</pub-id></element-citation></ref><ref id="bib96"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name><name><surname>Andolfatto</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Pervasive natural selection in the <italic>Drosophila</italic> genome?</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000495</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000495</pub-id><pub-id pub-id-type="pmid">19503600</pub-id></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sella</surname><given-names>G</given-names></name><name><surname>Barton</surname><given-names>NH</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Thinking about the evolution of complex traits in the era of genome-wide association studies</article-title><source>Annual Review of Genomics and Human Genetics</source><volume>20</volume><fpage>461</fpage><lpage>493</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-083115-022316</pub-id><pub-id pub-id-type="pmid">31283361</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siepel</surname><given-names>A.</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Phylogenetic estimation of context-dependent substitution rates by maximum likelihood</article-title><source>Molecular Biology and Evolution</source><volume>21</volume><fpage>468</fpage><lpage>488</lpage><pub-id pub-id-type="doi">10.1093/molbev/msh039</pub-id><pub-id pub-id-type="pmid">14660683</pub-id></element-citation></ref><ref id="bib99"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siepel</surname><given-names>A</given-names></name><name><surname>Bejerano</surname><given-names>G</given-names></name><name><surname>Pedersen</surname><given-names>JS</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Hou</surname><given-names>M</given-names></name><name><surname>Rosenbloom</surname><given-names>K</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Spieth</surname><given-names>J</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Richards</surname><given-names>S</given-names></name><name><surname>Weinstock</surname><given-names>GM</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Miller</surname><given-names>W</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Evolutionarily conserved elements in vertebrate, insect, worm, and yeast genomes</article-title><source>Genome Research</source><volume>15</volume><fpage>1034</fpage><lpage>1050</lpage><pub-id pub-id-type="doi">10.1101/gr.3715005</pub-id><pub-id pub-id-type="pmid">16024819</pub-id></element-citation></ref><ref id="bib100"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simons</surname><given-names>YB</given-names></name><name><surname>Bullaughey</surname><given-names>K</given-names></name><name><surname>Hudson</surname><given-names>RR</given-names></name><name><surname>Sella</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A population genetic interpretation of GWAS findings for human quantitative traits</article-title><source>PLOS Biology</source><volume>16</volume><elocation-id>e2002985</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.2002985</pub-id><pub-id pub-id-type="pmid">29547617</pub-id></element-citation></ref><ref id="bib101"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Skov</surname><given-names>L</given-names></name><name><surname>Hui</surname><given-names>R</given-names></name><name><surname>Shchur</surname><given-names>V</given-names></name><name><surname>Hobolth</surname><given-names>A</given-names></name><name><surname>Scally</surname><given-names>A</given-names></name><name><surname>Schierup</surname><given-names>MH</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Detecting archaic introgression using an unadmixed outgroup</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007641</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007641</pub-id><pub-id pub-id-type="pmid">30226838</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>JM</given-names></name><name><surname>Haigh</surname><given-names>J</given-names></name></person-group><year iso-8601-date="1974">1974</year><article-title>The hitch-hiking effect of a favourable gene</article-title><source>Genetical Research</source><volume>23</volume><fpage>23</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1017/S0016672300014634</pub-id><pub-id pub-id-type="pmid">4407212</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name><name><surname>Adzhubei</surname><given-names>I</given-names></name><name><surname>Thurman</surname><given-names>RE</given-names></name><name><surname>Kryukov</surname><given-names>GV</given-names></name><name><surname>Mirkin</surname><given-names>SM</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Human mutation rate associated with DNA replication timing</article-title><source>Nature Genetics</source><volume>41</volume><fpage>393</fpage><lpage>395</lpage><pub-id pub-id-type="doi">10.1038/ng.363</pub-id><pub-id pub-id-type="pmid">19287383</pub-id></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinrücken</surname><given-names>M</given-names></name><name><surname>Spence</surname><given-names>JP</given-names></name><name><surname>Kamm</surname><given-names>JA</given-names></name><name><surname>Wieczorek</surname><given-names>E</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Model-Based detection and analysis of introgressed neanderthal ancestry in modern humans</article-title><source>Molecular Ecology</source><volume>27</volume><fpage>3873</fpage><lpage>3888</lpage><pub-id pub-id-type="doi">10.1111/mec.14565</pub-id><pub-id pub-id-type="pmid">29603507</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stephan</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Genetic hitchhiking versus background selection: the controversy and its implications</article-title><source>Philosophical Transactions of the Royal Society of London Series B, Biological Sciences</source><volume>365</volume><fpage>1245</fpage><lpage>1253</lpage><pub-id pub-id-type="doi">10.1098/rstb.2009.0278</pub-id><pub-id pub-id-type="pmid">20308100</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sudmant</surname><given-names>PH</given-names></name><name><surname>Rausch</surname><given-names>T</given-names></name><name><surname>Gardner</surname><given-names>EJ</given-names></name><name><surname>Handsaker</surname><given-names>RE</given-names></name><name><surname>Abyzov</surname><given-names>A</given-names></name><name><surname>Huddleston</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Ye</surname><given-names>K</given-names></name><name><surname>Jun</surname><given-names>G</given-names></name><name><surname>Fritz</surname><given-names>MH-Y</given-names></name><name><surname>Konkel</surname><given-names>MK</given-names></name><name><surname>Malhotra</surname><given-names>A</given-names></name><name><surname>Stütz</surname><given-names>AM</given-names></name><name><surname>Shi</surname><given-names>X</given-names></name><name><surname>Casale</surname><given-names>FP</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Hormozdiari</surname><given-names>F</given-names></name><name><surname>Dayama</surname><given-names>G</given-names></name><name><surname>Chen</surname><given-names>K</given-names></name><name><surname>Malig</surname><given-names>M</given-names></name><name><surname>Chaisson</surname><given-names>MJP</given-names></name><name><surname>Walter</surname><given-names>K</given-names></name><name><surname>Meiers</surname><given-names>S</given-names></name><name><surname>Kashin</surname><given-names>S</given-names></name><name><surname>Garrison</surname><given-names>E</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Lam</surname><given-names>HYK</given-names></name><name><surname>Mu</surname><given-names>XJ</given-names></name><name><surname>Alkan</surname><given-names>C</given-names></name><name><surname>Antaki</surname><given-names>D</given-names></name><name><surname>Bae</surname><given-names>T</given-names></name><name><surname>Cerveira</surname><given-names>E</given-names></name><name><surname>Chines</surname><given-names>P</given-names></name><name><surname>Chong</surname><given-names>Z</given-names></name><name><surname>Clarke</surname><given-names>L</given-names></name><name><surname>Dal</surname><given-names>E</given-names></name><name><surname>Ding</surname><given-names>L</given-names></name><name><surname>Emery</surname><given-names>S</given-names></name><name><surname>Fan</surname><given-names>X</given-names></name><name><surname>Gujral</surname><given-names>M</given-names></name><name><surname>Kahveci</surname><given-names>F</given-names></name><name><surname>Kidd</surname><given-names>JM</given-names></name><name><surname>Kong</surname><given-names>Y</given-names></name><name><surname>Lameijer</surname><given-names>E-W</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Marth</surname><given-names>G</given-names></name><name><surname>Mason</surname><given-names>CE</given-names></name><name><surname>Menelaou</surname><given-names>A</given-names></name><name><surname>Muzny</surname><given-names>DM</given-names></name><name><surname>Nelson</surname><given-names>BJ</given-names></name><name><surname>Noor</surname><given-names>A</given-names></name><name><surname>Parrish</surname><given-names>NF</given-names></name><name><surname>Pendleton</surname><given-names>M</given-names></name><name><surname>Quitadamo</surname><given-names>A</given-names></name><name><surname>Raeder</surname><given-names>B</given-names></name><name><surname>Schadt</surname><given-names>EE</given-names></name><name><surname>Romanovitch</surname><given-names>M</given-names></name><name><surname>Schlattl</surname><given-names>A</given-names></name><name><surname>Sebra</surname><given-names>R</given-names></name><name><surname>Shabalin</surname><given-names>AA</given-names></name><name><surname>Untergasser</surname><given-names>A</given-names></name><name><surname>Walker</surname><given-names>JA</given-names></name><name><surname>Wang</surname><given-names>M</given-names></name><name><surname>Yu</surname><given-names>F</given-names></name><name><surname>Zhang</surname><given-names>C</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Zheng-Bradley</surname><given-names>X</given-names></name><name><surname>Zhou</surname><given-names>W</given-names></name><name><surname>Zichner</surname><given-names>T</given-names></name><name><surname>Sebat</surname><given-names>J</given-names></name><name><surname>Batzer</surname><given-names>MA</given-names></name><name><surname>McCarroll</surname><given-names>SA</given-names></name><collab>1000 Genomes Project Consortium</collab><name><surname>Mills</surname><given-names>RE</given-names></name><name><surname>Gerstein</surname><given-names>MB</given-names></name><name><surname>Bashir</surname><given-names>A</given-names></name><name><surname>Stegle</surname><given-names>O</given-names></name><name><surname>Devine</surname><given-names>SE</given-names></name><name><surname>Lee</surname><given-names>C</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Korbel</surname><given-names>JO</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>An integrated map of structural variation in 2,504 human genomes</article-title><source>Nature</source><volume>526</volume><fpage>75</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.1038/nature15394</pub-id><pub-id pub-id-type="pmid">26432246</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Terhorst</surname><given-names>J</given-names></name><name><surname>Kamm</surname><given-names>JA</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Robust and scalable inference of population history from hundreds of unphased whole genomes</article-title><source>Nature Genetics</source><volume>49</volume><fpage>303</fpage><lpage>309</lpage><pub-id pub-id-type="doi">10.1038/ng.3748</pub-id><pub-id pub-id-type="pmid">28024154</pub-id></element-citation></ref><ref id="bib108"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thornton</surname><given-names>KR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Polygenic adaptation to an environmental shift: temporal dynamics of variation under Gaussian stabilizing selection and additive effects on a single trait</article-title><source>Genetics</source><volume>213</volume><fpage>1513</fpage><lpage>1530</lpage><pub-id pub-id-type="doi">10.1534/genetics.119.302662</pub-id><pub-id pub-id-type="pmid">31653678</pub-id></element-citation></ref><ref id="bib109"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Torres</surname><given-names>R</given-names></name><name><surname>Szpiech</surname><given-names>ZA</given-names></name><name><surname>Hernandez</surname><given-names>RD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Human demographic history has amplified the effects of background selection across the genome</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007387</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007387</pub-id><pub-id pub-id-type="pmid">29912945</pub-id></element-citation></ref><ref id="bib110"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Torres</surname><given-names>R</given-names></name><name><surname>Stetter</surname><given-names>MG</given-names></name><name><surname>Hernandez</surname><given-names>RD</given-names></name><name><surname>Ross-Ibarra</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The temporal dynamics of background selection in nonequilibrium populations</article-title><source>Genetics</source><volume>214</volume><fpage>1019</fpage><lpage>1030</lpage><pub-id pub-id-type="doi">10.1534/genetics.119.302892</pub-id><pub-id pub-id-type="pmid">32071195</pub-id></element-citation></ref><ref id="bib111"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vernot</surname><given-names>B</given-names></name><name><surname>Akey</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Resurrecting surviving Neandertal lineages from modern human genomes</article-title><source>Science</source><volume>343</volume><fpage>1017</fpage><lpage>1021</lpage><pub-id pub-id-type="doi">10.1126/science.1245938</pub-id><pub-id pub-id-type="pmid">24476670</pub-id></element-citation></ref><ref id="bib112"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname><given-names>P</given-names></name><name><surname>Gommers</surname><given-names>R</given-names></name><name><surname>Oliphant</surname><given-names>TE</given-names></name><name><surname>Haberland</surname><given-names>M</given-names></name><name><surname>Reddy</surname><given-names>T</given-names></name><name><surname>Cournapeau</surname><given-names>D</given-names></name><name><surname>Burovski</surname><given-names>E</given-names></name><name><surname>Peterson</surname><given-names>P</given-names></name><name><surname>Weckesser</surname><given-names>W</given-names></name><name><surname>Bright</surname><given-names>J</given-names></name><name><surname>van der Walt</surname><given-names>SJ</given-names></name><name><surname>Brett</surname><given-names>M</given-names></name><name><surname>Wilson</surname><given-names>J</given-names></name><name><surname>Millman</surname><given-names>KJ</given-names></name><name><surname>Mayorov</surname><given-names>N</given-names></name><name><surname>Nelson</surname><given-names>ARJ</given-names></name><name><surname>Jones</surname><given-names>E</given-names></name><name><surname>Kern</surname><given-names>R</given-names></name><name><surname>Larson</surname><given-names>E</given-names></name><name><surname>Carey</surname><given-names>CJ</given-names></name><name><surname>Polat</surname><given-names>İ</given-names></name><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Moore</surname><given-names>EW</given-names></name><name><surname>VanderPlas</surname><given-names>J</given-names></name><name><surname>Laxalde</surname><given-names>D</given-names></name><name><surname>Perktold</surname><given-names>J</given-names></name><name><surname>Cimrman</surname><given-names>R</given-names></name><name><surname>Henriksen</surname><given-names>I</given-names></name><name><surname>Quintero</surname><given-names>EA</given-names></name><name><surname>Harris</surname><given-names>CR</given-names></name><name><surname>Archibald</surname><given-names>AM</given-names></name><name><surname>Ribeiro</surname><given-names>AH</given-names></name><name><surname>Pedregosa</surname><given-names>F</given-names></name><name><surname>van Mulbregt</surname><given-names>P</given-names></name><collab>SciPy 1.0 Contributors</collab></person-group><year iso-8601-date="2020">2020</year><article-title>SciPy 1.0: fundamental algorithms for scientific computing in python</article-title><source>Nature Methods</source><volume>17</volume><fpage>261</fpage><lpage>272</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0686-2</pub-id><pub-id pub-id-type="pmid">32015543</pub-id></element-citation></ref><ref id="bib113"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wall</surname><given-names>JD</given-names></name><name><surname>Pritchard</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Haplotype blocks and linkage disequilibrium in the human genome</article-title><source>Nature Reviews Genetics</source><volume>4</volume><fpage>587</fpage><lpage>597</lpage><pub-id pub-id-type="doi">10.1038/nrg1123</pub-id><pub-id pub-id-type="pmid">12897771</pub-id></element-citation></ref><ref id="bib114"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wall</surname><given-names>JD</given-names></name><name><surname>Hammer</surname><given-names>MF</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Archaic admixture in the human genome</article-title><source>Current Opinion in Genetics &amp; Development</source><volume>16</volume><fpage>606</fpage><lpage>610</lpage><pub-id pub-id-type="doi">10.1016/j.gde.2006.09.006</pub-id><pub-id pub-id-type="pmid">17027252</pub-id></element-citation></ref><ref id="bib115"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wall</surname><given-names>JD</given-names></name><name><surname>Lohmueller</surname><given-names>KE</given-names></name><name><surname>Plagnol</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Detecting ancient admixture and estimating demographic parameters in multiple human populations</article-title><source>Molecular Biology and Evolution</source><volume>26</volume><fpage>1823</fpage><lpage>1827</lpage><pub-id pub-id-type="doi">10.1093/molbev/msp096</pub-id><pub-id pub-id-type="pmid">19420049</pub-id></element-citation></ref><ref id="bib116"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Walsh</surname><given-names>B</given-names></name><name><surname>Lynch</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><source>Evolution and Selection of Quantitative Traits</source><publisher-name>Oxford University Presss</publisher-name></element-citation></ref><ref id="bib117"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>L</given-names></name><name><surname>Beissinger</surname><given-names>TM</given-names></name><name><surname>Lorant</surname><given-names>A</given-names></name><name><surname>Ross-Ibarra</surname><given-names>C</given-names></name><name><surname>Ross-Ibarra</surname><given-names>J</given-names></name><name><surname>Hufford</surname><given-names>MB</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The interplay of demography and selection during maize domestication and expansion</article-title><source>Genome Biology</source><volume>18</volume><elocation-id>215</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-017-1346-4</pub-id><pub-id pub-id-type="pmid">29132403</pub-id></element-citation></ref><ref id="bib118"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ward</surname><given-names>LD</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Evidence of abundant purifying selection in humans for recently acquired regulatory functions</article-title><source>Science</source><volume>337</volume><fpage>1675</fpage><lpage>1678</lpage><pub-id pub-id-type="doi">10.1126/science.1225057</pub-id><pub-id pub-id-type="pmid">22956687</pub-id></element-citation></ref><ref id="bib119"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wiehe</surname><given-names>TH</given-names></name><name><surname>Stephan</surname><given-names>W</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Analysis of a genetic hitchhiking model, and its application to DNA polymorphism data from <italic>Drosophila melanogaster</italic></article-title><source>Molecular Biology and Evolution</source><volume>10</volume><fpage>842</fpage><lpage>854</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a040046</pub-id><pub-id pub-id-type="pmid">8355603</pub-id></element-citation></ref><ref id="bib120"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wiuf</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Consistency of estimators of population scaled parameters using composite likelihood</article-title><source>Journal of Mathematical Biology</source><volume>53</volume><fpage>821</fpage><lpage>841</lpage><pub-id pub-id-type="doi">10.1007/s00285-006-0031-0</pub-id><pub-id pub-id-type="pmid">16960689</pub-id></element-citation></ref><ref id="bib121"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wright</surname><given-names>S</given-names></name></person-group><year iso-8601-date="1931">1931</year><article-title>Evolution in Mendelian populations</article-title><source>Genetics</source><volume>16</volume><fpage>97</fpage><lpage>159</lpage><pub-id pub-id-type="doi">10.1093/genetics/16.2.97</pub-id><pub-id pub-id-type="pmid">17246615</pub-id></element-citation></ref><ref id="bib122"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wright</surname><given-names>S</given-names></name></person-group><year iso-8601-date="1935">1935</year><article-title>The analysis of variance and the correlations between relatives with respect to deviations from an optimum</article-title><source>Journal of Genetics</source><volume>30</volume><fpage>243</fpage><lpage>256</lpage><pub-id pub-id-type="doi">10.1007/BF02982239</pub-id></element-citation></ref><ref id="bib123"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wright</surname><given-names>SI</given-names></name><name><surname>Foxe</surname><given-names>JP</given-names></name><name><surname>DeRose-Wilson</surname><given-names>L</given-names></name><name><surname>Kawabe</surname><given-names>A</given-names></name><name><surname>Looseley</surname><given-names>M</given-names></name><name><surname>Gaut</surname><given-names>BS</given-names></name><name><surname>Charlesworth</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Testing for effects of recombination rate on nucleotide diversity in natural populations of <italic>Arabidopsis lyrata</italic></article-title><source>Genetics</source><volume>174</volume><fpage>1421</fpage><lpage>1430</lpage><pub-id pub-id-type="doi">10.1534/genetics.106.062588</pub-id><pub-id pub-id-type="pmid">16951057</pub-id></element-citation></ref><ref id="bib124"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wright</surname><given-names>SI</given-names></name><name><surname>Andolfatto</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>The impact of natural selection on the genome: emerging patterns in <italic>Drosophila</italic> and Arabidopsis</article-title><source>Annual Review of Ecology and Systematics</source><volume>39</volume><fpage>193</fpage><lpage>213</lpage></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s6"><title>Methods and additional analyses</title><p>Table of Contents</p><list list-type="order"><list-item><p>Model and inference method .................................................................................................... 17</p><list list-type="simple"><list-item><p>1.1 Model and inference problem ............................................................................................. 18</p></list-item><list-item><p>1.2 Calculating lookup tables .................................................................................................... 21</p></list-item><list-item><p>1.3 Binning neutral sites ............................................................................................................ 23</p></list-item><list-item><p>1.4 Optimization ........................................................................................................................ 24</p></list-item><list-item><p>1.5 Thresholding ........................................................................................................................ 27</p></list-item><list-item><p>1.6 Software ............................................................................................................................... 30</p></list-item></list></list-item><list-item><p>Data sources and filters ............................................................................................................. 31</p><list list-type="simple"><list-item><p>2.1 Polymorphism data .............................................................................................................. 31</p></list-item><list-item><p>2.2 Multiple species alignment data ........................................................................................... 31</p></list-item><list-item><p>2.3 Genetic map ........................................................................................................................ 31</p></list-item><list-item><p>2.4 Human gene annotations .................................................................................................... 31</p></list-item><list-item><p>2.5 CADD scores ....................................................................................................................... 32</p></list-item><list-item><p>2.6 ENCODE cCRE annotations ................................................................................................ 32</p></list-item><list-item><p>2.7 Substitutions in the human lineage ..................................................................................... 32</p></list-item><list-item><p>2.8 Covariates of <inline-formula><mml:math id="inf25"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> .................................................................................................................... 32</p></list-item></list></list-item><list-item><p>Choice of exogenous parameters .............................................................................................. 33</p><list list-type="simple"><list-item><p>3.1 Choosing putatively neutral sites based on phylogenetic conservation ............................. 33</p></list-item><list-item><p>3.2 Removing sites at the telomeric ends of chromosomes ...................................................... 33</p></list-item><list-item><p>3.3 Estimating local variation in mutation rates ........................................................................ 33</p></list-item></list></list-item><list-item><p>Fitting models with different targets of selection ...................................................................... 36</p><list list-type="simple"><list-item><p>4.1 Background selection model based on phylogenetic conservation .................................... 36</p></list-item><list-item><p>4.2 Background selection model based on genic annotations ................................................... 37</p></list-item><list-item><p>4.3 Background selection models separating conserved exonic and non-exonic sites ............. 39</p></list-item><list-item><p>4.4 Background selection models based on other annotations ................................................ 42</p></list-item><list-item><p>4.5 Models with selective sweeps .............................................................................................. 46</p></list-item><list-item><p>4.6 Comparison with previous work by McVicker et al. ............................................................. 50</p></list-item></list></list-item><list-item><p>Assessing estimates of the deleterious mutation rate ............................................................... 52</p><list list-type="simple"><list-item><p>5.1 Estimates of the total mutation rate per site ........................................................................ 52</p></list-item><list-item><p>5.2 Estimating the proportion of deleterious mutations at putatively selected sites ................ 53</p></list-item><list-item><p>5.3 Interpreting the relationship between the two estimates .................................................... 54</p></list-item></list></list-item><list-item><p>Statistics ..................................................................................................................................... 54</p><list list-type="simple"><list-item><p>6.1 Estimates of explained variance ........................................................................................... 54</p></list-item><list-item><p>6.2 Comparing the fit of different maps ..................................................................................... 55</p></list-item><list-item><p>6.3 Sampling error in parameter estimates ............................................................................... 56</p></list-item></list></list-item><list-item><p>Results for other human populations ........................................................................................ 57</p></list-item><list-item><p>Diversity levels where background selection is weakest (<inline-formula><mml:math id="inf26"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula><italic>)</italic> ................................................ 60</p><list list-type="simple"><list-item><p>8.1 Covariates of <inline-formula><mml:math id="inf27"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> .................................................................................................................... 60</p></list-item><list-item><p>8.2 Mutational spectrum and biased gene conversion .............................................................. 61</p></list-item><list-item><p>8.3 A footprint of archaic introgression? ................................................................................... 64</p></list-item></list></list-item><list-item><p>Additional Figures ..................................................................................................................... 68</p></list-item></list><sec sec-type="appendix" id="s6-1"><title>1. Model and inference method</title><p>Here we detail the model and inference method used in this study. In Section 1.1, we describe our model for the effects of background selection and selective sweeps and our approach to inferring the parameters of these models. This section is adapted from Elyashiv and colleagues (<xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>), who applied a similar approach to data from <italic>Drosophila melanogaster</italic>; we reproduce it here for completeness. In Section 1.2, we describe how we calculate lookup tables for the effects of background selection and sweeps, which our inference relies upon. We introduce several changes to the methods used in previous studies (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>; <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>), which allow us to better control the precision of maps of the effects of linked selection. In Section 1.3, we describe how we represent neutral polymorphism data and maps of the effects of linked selection in our calculations in order to increase computational tractability. In Section 1.4, we describe the optimization algorithm that we use to find the selection parameters that maximize our models composite-likelihood, and we apply the optimization to simulated datasets in order to demonstrate its efficacy and robustness. In Section 1.5, we introduce a thresholding approach that contends with biases in our optimization that arise from model misspecification, and we investigate how this thresholding affects our inferences. Finally, in Section 1.6, we provide an overview of the software that we use for inference and for other key analyses in the paper. The software, its documentation, and maps of the effects of linked selection are available for download at (<ext-link ext-link-type="uri" xlink:href="https://github.com/sellalab/HumanLinkedSelectionMaps">https://github.com/sellalab/HumanLinkedSelectionMaps</ext-link>; <xref ref-type="bibr" rid="bib70">Murphy, 2021</xref>).</p><sec sec-type="appendix" id="s6-1-1"><title>1.1 Model and inference problem</title><p>We model the effects of background selection and selective sweeps on neutral heterozygosity levels (i.e. the probability of observing different alleles in a sample size of two), <inline-formula><mml:math id="inf28"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>π</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, at an autosomal position <inline-formula><mml:math id="inf29"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. In a coalescent framework, the model takes the form<disp-formula id="equ2"><label>(1)</label><mml:math id="m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf30"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local mutation rate, <inline-formula><mml:math id="inf31"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the effective population size without linked selection, <inline-formula><mml:math id="inf32"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local (multiplicative) reduction in the effective population size due to background selection and <inline-formula><mml:math id="inf33"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the local coalescence rate caused by selective sweeps (<xref ref-type="bibr" rid="bib119">Wiehe and Stephan, 1993</xref>; <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). This approximation can be derived by considering the probability that a mutation occurs (at a rate <inline-formula><mml:math id="inf34"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>2</mml:mn><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> per generation) before the pair of lineages coalesces, owing either to genetic drift (<inline-formula><mml:math id="inf35"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>), which includes the effect of background selection, or to a selective sweep (<inline-formula><mml:math id="inf36"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>). While we consider autosomes, the model can be extended to sex chromosomes with straightforward modifications.</p><fig id="app1fig1" position="float"><label>Appendix 1—figure 1.</label><caption><title>Modeling and inferring the effects of linked selection in humans.</title><p>Given the targets of selection and corresponding selection parameters (<bold>a and b</bold>), we calculate the expected neutral diversity levels along the genome (<bold>c</bold>). We infer the selection parameters by maximizing their composite-likelihood given observed diversity levels (<bold>c</bold>). Based on these parameter estimates, we calculate a map of the expected effects of linked selection on diversity levels.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig1-v2.tif"/></fig><p>The model for the effects of background selection, <inline-formula><mml:math id="inf37"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, follows <xref ref-type="bibr" rid="bib53">Hudson and Kaplan, 1995</xref> and <xref ref-type="bibr" rid="bib74">Nordborg et al., 1996</xref> (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1a</xref>). We assume a set of distinct annotations <inline-formula><mml:math id="inf38"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> under purifying selection (e.g. conserved exonic and non-exonic regions) and positions in the genome <inline-formula><mml:math id="inf39"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf40"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> denotes the set of genomic positions with annotation <inline-formula><mml:math id="inf41"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The selection parameters at these annotations are given by <inline-formula><mml:math id="inf42"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf43"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the rate of deleterious mutations and <inline-formula><mml:math id="inf44"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the distribution of selection coefficients in heterozygotes for a deleterious mutation. The reduction in the effective population size is then<disp-formula id="equ3"><label>(2)</label><mml:math id="m3"><mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo>⟮</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:munder><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>∈</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:munder><mml:mo>∫</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow><mml:mo>⟯</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf45"><mml:mi>R</mml:mi></mml:math></inline-formula> is the genetic map and <inline-formula><mml:math id="inf46"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the genetic distance between the focal position <inline-formula><mml:math id="inf47"><mml:mi>x</mml:mi></mml:math></inline-formula> and positions <inline-formula><mml:math id="inf48"><mml:mi>y</mml:mi></mml:math></inline-formula> (only positions on the same chromosome are considered). The integrand reflects the effect that a site under purifying selection at position <inline-formula><mml:math id="inf49"><mml:mi>y</mml:mi></mml:math></inline-formula> exerts on a neutral site at position <inline-formula><mml:math id="inf50"><mml:mi>x</mml:mi></mml:math></inline-formula>. This expression and its combination across sites provide a good approximation to the effect of background selection so long as selection is sufficiently strong (i.e. when <inline-formula><mml:math id="inf51"><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>t</mml:mi><mml:mo>≫</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>).</p><p>In turn, the model for the effect of selective sweeps follows from an approximation used by <xref ref-type="bibr" rid="bib6">Barton, 1998</xref> and <xref ref-type="bibr" rid="bib38">Gillespie, 2000</xref>, among others (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1b</xref>). Similarly to the model for background selection, we assume a set of distinct annotations <inline-formula><mml:math id="inf52"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> subject to sweeps, but here the specific positions at which substitutions have occurred are known, <inline-formula><mml:math id="inf53"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> with <inline-formula><mml:math id="inf54"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> denoting the set of substitution positions with annotation <inline-formula><mml:math id="inf55"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>. The selection parameters at these annotations are <inline-formula><mml:math id="inf56"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf57"><mml:mi>α</mml:mi></mml:math></inline-formula> is the fraction of substitutions that are beneficial and <inline-formula><mml:math id="inf58"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the distribution of their additive selection coefficients. For autosomes, the expected rate of coalescence per generations at position <inline-formula><mml:math id="inf59"><mml:mi>x</mml:mi></mml:math></inline-formula> due to sweeps is then approximated by<disp-formula id="equ4"><label>(3)</label><mml:math id="m4"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munder><mml:mo>∑</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:munder><mml:mi>α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>∈</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:munder><mml:mo>∫</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>τ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mi>d</mml:mi><mml:mi>s</mml:mi><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf60"><mml:mi>T</mml:mi></mml:math></inline-formula> is the length of the lineage (in generations) over which substitutions occurred, the positions of substitutions <inline-formula><mml:math id="inf61"><mml:mi>y</mml:mi></mml:math></inline-formula> are summed over the chromosome with the focal site, <inline-formula><mml:math id="inf62"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the average effective population size and <inline-formula><mml:math id="inf63"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>τ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the expected time to fixation of a beneficial substitution with selection coefficient <inline-formula><mml:math id="inf64"><mml:mi>s</mml:mi></mml:math></inline-formula> and given an effective population size <inline-formula><mml:math id="inf65"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>. We use the diffusion approximation for the fixation time<disp-formula id="equ5"><label>(4)</label><mml:math id="m5"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>τ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>4</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>γ</mml:mi><mml:mo>−</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>4</mml:mn></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub><mml:mi>s</mml:mi><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mi>s</mml:mi></mml:mfrac><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf66"><mml:mi>γ</mml:mi></mml:math></inline-formula> is the Euler constant (<xref ref-type="bibr" rid="bib46">Hermisson and Pennings, 2005</xref>). This model relies on several simplifying assumptions and approximations. In particular, the term <inline-formula><mml:math id="inf67"><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>T</mml:mi></mml:math></inline-formula> relies on an assumption of one substitution per site per lineage and neglects variation in the length of lineages across loci. In combining the effects over substitutions, we further assume that the timings of beneficial substitutions are independent and uniformly distributed along the lineage, and that they are infrequent enough such that we can ignore interference among them (<xref ref-type="bibr" rid="bib60">Kim and Stephan, 2003</xref>). The exponent approximates the probability of coalescence of two samples due to a classic sweep with additive selection coefficient <inline-formula><mml:math id="inf68"><mml:mi>s</mml:mi></mml:math></inline-formula> (where <inline-formula><mml:math id="inf69"><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>s</mml:mi><mml:mo>≫</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>) in a panmictic population of constant effective size <inline-formula><mml:math id="inf70"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>. (For the relationships between these expressions and other kinds of sweeps see SOM Section D in <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). In principle, we should use the local <inline-formula><mml:math id="inf71"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> incorporating the effects of background selection but given the logarithmic dependence of <xref ref-type="disp-formula" rid="equ5">Equation (4)</xref> on <inline-formula><mml:math id="inf72"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, we simply use the average <inline-formula><mml:math id="inf73"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><p>To infer the selection parameters <inline-formula><mml:math id="inf74"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf75"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, we use a composite-likelihood approach across sites and samples (<xref ref-type="bibr" rid="bib54">Hudson, 2001</xref>; <xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>). We denote the positions of neutral sites by <inline-formula><mml:math id="inf76"><mml:mi>X</mml:mi></mml:math></inline-formula> and the set of samples by <inline-formula><mml:math id="inf77"><mml:mi>I</mml:mi></mml:math></inline-formula>. We then summarize the observations by a set of indicator variables across sites and all pairs of samples <inline-formula><mml:math id="inf78"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>O</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:mo>∈</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>≠</mml:mo><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mi>I</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf79"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> indicates that samples <inline-formula><mml:math id="inf80"><mml:mi>i</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf81"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>j</mml:mi><mml:mtext> </mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo>≠</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> differ at position <inline-formula><mml:math id="inf82"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf83"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> indicates that they are the same. In these terms, the composite log-likelihood takes the form<disp-formula id="equ6"><label>(5)</label><mml:math id="m6"><mml:mrow><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>L</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>∈</mml:mo><mml:mi>X</mml:mi></mml:mrow></mml:munder><mml:mspace width="thickmathspace"/><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>≠</mml:mo><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mi>I</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi></mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where<disp-formula id="equ7"><mml:math id="m7"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mo movablelimits="true" form="prefix">Pr</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">Θ</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:mtext> </mml:mtext><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"/></mml:mrow><mml:mspace width="negativethinmathspace"/></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Using composite-likelihood circumvents the complications of considering linkage disequilibrium (LD) and of coalescent models for larger sample sizes. Importantly, maximizing this composite-likelihood should yield unbiased point estimates (<xref ref-type="bibr" rid="bib34">Fearnhead, 2003</xref>; <xref ref-type="bibr" rid="bib120">Wiuf, 2006</xref>). Beyond losing the information in LD patterns and in the site frequency spectrum, the main cost of this approach is the difficulty in assessing uncertainty in parameter estimates (as standard asymptotic results do not apply). We therefore use other ways to assess the reliability of our inferences.</p><p>To make the composite-likelihood calculations (i.e. the calculation of <inline-formula><mml:math id="inf84"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>) feasible genome-wide, we discretize the distribution of selection coefficients on a fixed grid. Given a grid of negative and positive selection coefficients, <inline-formula><mml:math id="inf85"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf86"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="inf87"><mml:mi>g</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mi>G</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf88"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mi>K</mml:mi></mml:math></inline-formula>, the distribution of selection coefficients for each annotation becomes a set of weights on this grid, <inline-formula><mml:math id="inf89"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>w</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf90"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>w</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>. (In principle, the grid could also be annotation-specific.) For background selection, these weights reflect the rate of deleterious mutations with a given selection coefficient and their sum should therefore be bound by the maximal deleterious mutation rate per site. For sweeps, the weights reflect the fraction of beneficial substitutions with a given selection coefficient and their sum should be bound by 1. In these terms, the effect of background selection takes the form<disp-formula id="equ8"><label>(6)</label><mml:math id="m8"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>B</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">Θ</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:munder><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>g</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:munderover><mml:mi>w</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">)</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf91"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is the proportional reduction in the effective population size induced by having one deleterious mutation per generation per site with selection coefficient <inline-formula><mml:math id="inf92"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> at all the positions in annotation <inline-formula><mml:math id="inf93"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> . By the same token, the effects of sweeps take the form<disp-formula id="equ9"><label>(7)</label><mml:math id="m9"><mml:mrow><mml:mi>S</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">Θ</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munder><mml:mo>∑</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:munder><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mi>w</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf94"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the probability of coalescence per generation induced by sweeps in annotation <inline-formula><mml:math id="inf95"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, if all the substitutions in this annotation are beneficial with selection coefficient <inline-formula><mml:math id="inf96"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. By using a grid, we can calculate a lookup table of <inline-formula><mml:math id="inf97"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf98"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> once and then use it repeatedly to calculate the likelihood of different sets of weights. Moreover, the interpretation of estimated distributions on a grid is arguably simpler than that of the continuous parametric distributions commonly used (e.g. gamma and exponential), which impose rigid interdependencies between the densities associated with different selection coefficients with little justification and while the data is only informative about a subset of the domain. In the next section, we describe additional simplifications in the calculation of <inline-formula><mml:math id="inf99"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf100"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><p>Other parameters are estimated as follows. Consider <xref ref-type="disp-formula" rid="equ2">Equation (1)</xref> rewritten as<disp-formula id="equ10"><label>(8)</label><mml:math id="m10"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>π</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>π</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mi>π</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="thinmathspace"/><mml:mo>+</mml:mo><mml:mspace width="thinmathspace"/><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>B</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace width="thinmathspace"/><mml:mo>+</mml:mo><mml:mspace width="thinmathspace"/><mml:mn>2</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>;</mml:mo><mml:msub><mml:mover><mml:mi>N</mml:mi><mml:mo accent="false">¯</mml:mo></mml:mover><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>to clearly specify all the additional parameters required for inference. <inline-formula><mml:math id="inf101"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>π</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>≡</mml:mo><mml:mn>4</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is (approximately) the average neutral heterozygosity, given the effective population size in the absence of linked selection and the average mutation rate per site (<inline-formula><mml:math id="inf102"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>); <inline-formula><mml:math id="inf103"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is estimated through the likelihood maximization. The local variation in mutation rate <inline-formula><mml:math id="inf104"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is estimated based on substitution rates at putatively neutral sites in an eight-primate phylogeny (excluding humans) in nonoverlapping windows, with a window size chosen to balance true variation in mutation rates and measurement error (see Section 3.3). Finally, <inline-formula><mml:math id="inf105"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is estimated based on the average genome-wide heterozygosity at putatively neutral sites, after dividing out by a direct estimate of the spontaneous point mutation rate of <inline-formula><mml:math id="inf106"><mml:mn>1.2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per site per generation (<xref ref-type="bibr" rid="bib64">Kong et al., 2012</xref>), and <inline-formula><mml:math id="inf107"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>T</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is estimated by <inline-formula><mml:math id="inf108"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover><mml:mi>K</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>π</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf109"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>K</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is the average number of point substitutions per putatively neutral site on the human lineage (see Section 2.7).</p></sec><sec sec-type="appendix" id="s6-1-2"><title>1.2 Calculating lookup tables</title><p>Here we describe how we calculate the lookup tables for<disp-formula id="equ11"><label>(9)</label><mml:math id="m11"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>≡</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>∈</mml:mo><mml:mi>a</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:munder><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>τ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>N</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>and<disp-formula id="equ12"><label>(10)</label><mml:math id="m12"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>≡</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>∈</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:munder><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>at all putatively neutral autosomal positions (<inline-formula><mml:math id="inf110"><mml:mi>x</mml:mi></mml:math></inline-formula>), given annotations (<inline-formula><mml:math id="inf111"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf112"><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) and selection coefficients (<inline-formula><mml:math id="inf113"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf114"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>). We focus on one annotation and selection coefficient at a time and therefore simplify the notation to <inline-formula><mml:math id="inf115"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf116"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, and omit the variables in <inline-formula><mml:math id="inf117"><mml:mi>τ</mml:mi></mml:math></inline-formula> and the subscripts of the selection coefficients. When we refer to accuracy in this section, we assume that there is no model misspecification (e.g., that putatively neutral sites are neutral, that sets of selected sites and selection parameter values are accurate, that genetic maps are accurate, etc.); once we control the accuracy in this sense, the main sources of error in our predictions will be due to model misspecification.</p><p><italic>Our general approach is to calculate</italic> <inline-formula><mml:math id="inf118"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> <italic>and</italic> <inline-formula><mml:math id="inf119"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> <italic>with high accuracy at a subset of positions and to use linear interpolation between them</italic>. The distances between these positions are chosen such that maps built using the lookup tables maintain a preset level of accuracy <inline-formula><mml:math id="inf120"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>. Specifically, we require that our approximation <inline-formula><mml:math id="inf121"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">~</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf122"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>b</mml:mi><mml:mo stretchy="false">~</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> at any position <inline-formula><mml:math id="inf123"><mml:mi>x</mml:mi></mml:math></inline-formula> satisfy<disp-formula id="equ13"><mml:math id="m13"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">~</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>ϵ</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mrow><mml:mover><mml:mi>b</mml:mi><mml:mo stretchy="false">~</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>ϵ</mml:mi><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf124"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is an upper bound on the deleterious mutation rate per site per generation. When these conditions are met one can show (based on <xref ref-type="disp-formula" rid="equ8 equ9">Equations 6; 7</xref>) that the relative accuracy of <inline-formula><mml:math id="inf125"><mml:mi>S</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf126"><mml:mi>B</mml:mi></mml:math></inline-formula>, and consequently of the expected neutral diversity level <inline-formula><mml:math id="inf127"><mml:mi>π</mml:mi></mml:math></inline-formula> (based on <xref ref-type="disp-formula" rid="equ2">Equation 1</xref>), are also bound by <inline-formula><mml:math id="inf128"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>.</p></sec><sec sec-type="appendix" id="s6-1-3"><title>Sweeps</title><p>Assume that we have calculated <inline-formula><mml:math id="inf129"><mml:mi>s</mml:mi></mml:math></inline-formula> accurately at position <inline-formula><mml:math id="inf130"><mml:mi>x</mml:mi></mml:math></inline-formula> and consider the distance <inline-formula><mml:math id="inf131"><mml:mi mathvariant="normal">Δ</mml:mi></mml:math></inline-formula> at which the relative change in <inline-formula><mml:math id="inf132"><mml:mi>s</mml:mi></mml:math></inline-formula> is bound by <inline-formula><mml:math id="inf133"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>, i.e., where<disp-formula id="equ14"><label>(11)</label><mml:math id="m14"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mo>≤</mml:mo><mml:mi>ϵ</mml:mi><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>From <xref ref-type="disp-formula" rid="equ14">Equation 11</xref>, we find that<disp-formula id="equ15"><mml:math id="m15"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mo>≤</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mi>y</mml:mi></mml:munder><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>=</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mi>y</mml:mi></mml:munder><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>≈</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mi>y</mml:mi></mml:munder><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>≈</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi>τ</mml:mi><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where the approximations assume <inline-formula><mml:math id="inf134"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>≪</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>. Consequently, by solving for <inline-formula><mml:math id="inf135"><mml:mi>Δ</mml:mi></mml:math></inline-formula> such that<disp-formula id="equ16"><label>(12)</label><mml:math id="m16"><mml:mrow><mml:mi>r</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>ϵ</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>τ</mml:mi></mml:mrow></mml:math></disp-formula></p><p>we assure that the relative accuracy between <inline-formula><mml:math id="inf136"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf137"><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi></mml:math></inline-formula> is bound by <inline-formula><mml:math id="inf138"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>. We therefore calculate <inline-formula><mml:math id="inf139"><mml:mi>s</mml:mi></mml:math></inline-formula> at the selected set of positions on a chromosome beginning at one end and choosing our step sizes according to <xref ref-type="disp-formula" rid="equ16">Equation 12</xref> until we reach the other end.</p></sec><sec sec-type="appendix" id="s6-1-4"><title>Background selection</title><p>Our calculation for background selection is based on the algorithm developed by <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref> (their calc_bkgd program) with several important modifications (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>). The problems that require these modifications are most pronounced for small selection coefficients, whose background selection effects are localized at short genetic distances from selected segments where they can be quite strong. First, McVicker et al. used an additional lookup table to integrate over the effects of background selection exerted by a contiguous selected segment (SI of <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>). This lookup table had poor resolution for small selection coefficients at short genetic distances from selected segments, and we have increased the resolution accordingly to fix the problem. Second, the algorithm for choosing the step size <inline-formula><mml:math id="inf140"><mml:mi mathvariant="normal">Δ</mml:mi></mml:math></inline-formula> is designed to control the absolute error, such that<disp-formula id="equ17"><mml:math id="m17"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mover><mml:mi>b</mml:mi><mml:mo>∼</mml:mo></mml:mover><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>ϵ</mml:mi><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>rather than the relative error (<xref ref-type="disp-formula" rid="equ18">Equation 13</xref>), which results in large relative errors when background selection effects are the strongest (which is with small selection coefficients). Third, the choice of step size <inline-formula><mml:math id="inf141"><mml:mi mathvariant="normal">Δ</mml:mi></mml:math></inline-formula> is based on the local behavior of background selection at the previous position, and consequently it sometimes skips over selected segments largely ignoring their highly localized effects (which are due to small selection coefficients). We describe how we resolve the last two problems in turn.</p><fig id="app1fig2" position="float"><label>Appendix 1—figure 2.</label><caption><title>Distribution of relative errors in predictions before and after modifying calc_bkgd.</title><p>We consider the model in which autosomal sites with the top 6% of CADD scores are chosen as selection targets, the deleterious mutation rate is <inline-formula><mml:math id="inf142"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mn>7.4</mml:mn><mml:mo>⋅</mml:mo><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per bp per generation and the selection coefficient is the lowest in our grid (<inline-formula><mml:math id="inf143"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>4.5</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>), because this is the case most prone to errors (see text). We calculate <inline-formula><mml:math id="inf144"><mml:mi>B</mml:mi></mml:math></inline-formula>-values accurately (using <xref ref-type="disp-formula" rid="equ12">Equation 10</xref>) at a million positions picked randomly from the 22 autosomes and use these values to calculate the relative errors based on the McVicker et al. algorithm (<bold>a</bold>) and on our modified algorithm (<bold>b</bold>). The side panel shows the proportion of sites in which the error exceeds <inline-formula><mml:math id="inf145"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> (below), as well as its breakdown in multiples of <inline-formula><mml:math id="inf146"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig2-v2.tif"/></fig><p>Assume that we have calculated <inline-formula><mml:math id="inf147"><mml:mi>b</mml:mi></mml:math></inline-formula> accurately at position <inline-formula><mml:math id="inf148"><mml:mi>x</mml:mi></mml:math></inline-formula> and consider the distance <inline-formula><mml:math id="inf149"><mml:mi mathvariant="normal">Δ</mml:mi></mml:math></inline-formula> at which the relative change in <inline-formula><mml:math id="inf150"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is bound by <inline-formula><mml:math id="inf151"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> (see <xref ref-type="disp-formula" rid="equ18">Equation 13</xref>), that is, where<disp-formula id="equ18"><label>(13)</label><mml:math id="m18"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mo>≤</mml:mo><mml:mi>ϵ</mml:mi><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Rearranging the left-hand side, we find that<disp-formula id="equ19"><mml:math id="m19"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">p</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>≤</mml:mo><mml:mi>ϵ</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>and assuming that <inline-formula><mml:math id="inf152"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>≪</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> we find that this requirement is well approximated by<disp-formula id="equ20"><mml:math id="m20"><mml:mrow><mml:mrow><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow></mml:mrow><mml:mo>≈</mml:mo><mml:mrow><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow><mml:msup><mml:mi>b</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mi>b</mml:mi><mml:mo>″</mml:mo></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:msup><mml:mi mathvariant="normal">Δ</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mo maxsize="1.623em" minsize="1.623em">|</mml:mo></mml:mrow></mml:mrow><mml:mo>≤</mml:mo><mml:mi>ϵ</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mi>M</mml:mi></mml:msub><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>As our putative step size, we therefore take the (smallest) solution of the quadratic<disp-formula id="equ21"><label>(14)</label><mml:math id="m21"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msup><mml:mi>b</mml:mi><mml:mrow><mml:mo>′</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:msup><mml:mi>b</mml:mi><mml:mrow><mml:msup><mml:mi/><mml:mo>″</mml:mo></mml:msup></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>ϵ</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>As in the case of sweeps, we calculate <inline-formula><mml:math id="inf153"><mml:mi>b</mml:mi></mml:math></inline-formula> at a selected set of positions on a chromosome, beginning on one end and choosing our step sizes in a way that maintains the preset relative accuracy <inline-formula><mml:math id="inf154"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> until we reach the other end. Assuming that we have calculated <inline-formula><mml:math id="inf155"><mml:mi>b</mml:mi></mml:math></inline-formula> accurately at position <inline-formula><mml:math id="inf156"><mml:mi>x</mml:mi></mml:math></inline-formula>, our algorithm for choosing the step size consists of the following steps:</p><list list-type="order"><list-item><p>If <inline-formula><mml:math id="inf157"><mml:mi>x</mml:mi></mml:math></inline-formula> is at the end of the chromosome, stop.</p></list-item><list-item><p>Calculate a candidate step size <inline-formula><mml:math id="inf158"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> by solving <xref ref-type="disp-formula" rid="equ21">Equation 14</xref>.</p></list-item><list-item><p>If <inline-formula><mml:math id="inf159"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is greater than a preset maximal step size <inline-formula><mml:math id="inf160"><mml:msub><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> then set <inline-formula><mml:math id="inf161"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p></list-item><list-item><p>If there is a selected segment between positions <inline-formula><mml:math id="inf162"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf163"><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> then set <inline-formula><mml:math id="inf164"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> such that <inline-formula><mml:math id="inf165"><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the midpoint between <inline-formula><mml:math id="inf166"><mml:mi>x</mml:mi></mml:math></inline-formula> and the beginning of the (closest) selected segment. This step assures that we do not ‘skip’ selected segments.</p></list-item><list-item><p>Convert <inline-formula><mml:math id="inf167"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> from Morgans to base-pairs, rounding downwards. But if the step ≤ 1 bp then set it to 1 bp, calculate <inline-formula><mml:math id="inf168"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, set <inline-formula><mml:math id="inf169"><mml:mi>x</mml:mi></mml:math></inline-formula> to <inline-formula><mml:math id="inf170"><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, and return to step 1.</p></list-item><list-item><p>Calculate <inline-formula><mml:math id="inf171"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> . If <inline-formula><mml:math id="inf172"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&gt;</mml:mo><mml:mi>ϵ</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> then set the step size in Morgans to <inline-formula><mml:math id="inf173"><mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> and return to step 4. Otherwise, set <inline-formula><mml:math id="inf174"><mml:mi>x</mml:mi></mml:math></inline-formula> to <inline-formula><mml:math id="inf175"><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">*</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and return to step 1.</p></list-item></list></sec><sec sec-type="appendix" id="s6-1-5"><title>Interpolation and representation of lookup tables</title><p>We calculate <inline-formula><mml:math id="inf176"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> or <inline-formula><mml:math id="inf177"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> at every autosomal position <inline-formula><mml:math id="inf178"><mml:mi>x</mml:mi></mml:math></inline-formula> (for a given selection coefficient and selected annotation) by linear interpolation between adjacent positions at which we calculated <inline-formula><mml:math id="inf179"><mml:mi>s</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf180"><mml:mi>b</mml:mi></mml:math></inline-formula> accurately. We then discretize the values of <inline-formula><mml:math id="inf181"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>b</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> or <inline-formula><mml:math id="inf182"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> on a linear grid of values corresponding to the preset accuracy <inline-formula><mml:math id="inf183"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>, and group together contiguous autosomal segments with the same discrete value. We intersect these segments with our list of putatively neutral sites (Section 3.1) to obtain lookup tables consisting of contiguous segments of putatively neutral sites with the same coarse-grained <inline-formula><mml:math id="inf184"><mml:mi>s</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf185"><mml:mi>b</mml:mi></mml:math></inline-formula> values for our sets of selected annotations and selection coefficients.</p></sec><sec sec-type="appendix" id="s6-1-6"><title>1.3 Binning neutral sites</title><p>A direct calculation of the composite log-likelihood function for given sets of selected annotations and selection coefficients and parameters (<xref ref-type="disp-formula" rid="equ6">Equation 5</xref>) requires that we store and access lookup tables and calculate the log-likelihood function at ~<inline-formula><mml:math id="inf186"><mml:mn>6.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> putatively neutral autosomal sites (see Section 2.1). Doing so would entail high computation and memory demands in the search for selection parameters that maximize the composite-likelihood. For example, our best-fitting models of background selection (see Main Text) with a grid of 6 selection coefficients would require storing and repeatedly accessing lookup tables that amount to <inline-formula><mml:math id="inf187"><mml:mn>6.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mi> </mml:mi><mml:mo>×</mml:mo><mml:mn>8</mml:mn><mml:mi> </mml:mi><mml:mo>×</mml:mo><mml:mn>6</mml:mn><mml:mo>≈</mml:mo><mml:mn>32</mml:mn></mml:math></inline-formula> GB (given a precision of <inline-formula><mml:math id="inf188"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>ϵ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>), and models involving multiple annotations for background selection and sweeps push the memory requirement to hundreds of GBs.</p><p>We reduce the computational and memory demands by dividing the set of putatively neutral sites into bins in which all the effects of background selection and sweeps predicted by the lookup tables and our estimates of the local (relative) mutation rate (<inline-formula><mml:math id="inf189"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="equ10">Equation 8</xref>; Section 3.3) are identical. The composite log-likelihood function can then be calculated by summing over log-likelihood functions corresponding to bins, where the calculation per bin requires only the bin-specific parameters and bin-specific summaries of polymorphism. The number and identity of bins varies with the sets of selected annotations and selection coefficients and parameters and with the precision (<inline-formula><mml:math id="inf190"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>). For our best-fitting models, the average number of sites per bin is ~100, implying a ~100 fold reduction in demands on memory and in the number of log-likelihood calculations. For our most complex selection models (Section 4), the binning reduces memory and computational demands tenfold.</p></sec><sec sec-type="appendix" id="s6-1-7"><title>1.4 Optimization</title><p>Here we describe how we developed and tested the algorithm we use in order to find the selection parameters that maximize the composite-likelihood of our different models. The high dimensional parameter space (including up to 55 parameters in the most complex model in Section 4) potentially makes this optimization problem non-trivial.</p></sec><sec sec-type="appendix" id="s6-1-8"><title>One step optimization</title><p>First, we tested the performance of standard optimization algorithms from the SciPy minimization toolkit (<xref ref-type="bibr" rid="bib112">Virtanen et al., 2020</xref>). To this end, we generated polymorphism datasets based on our best-fitting model of background selection based on phastCons conservation scores, as follows:</p><list list-type="order"><list-item><p>We fixed the total deleterious mutation rate to <inline-formula><mml:math id="inf191"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula> per base pair per generation, and randomly divided it among the 6 selection coefficients of the model by sampling from a Dirichlet distribution (with <inline-formula><mml:math id="inf192"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>). We set the expected neutral diversity level in the absence of background selection to <inline-formula><mml:math id="inf193"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mi>Y</mml:mi><mml:mi>R</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, where <inline-formula><mml:math id="inf194"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mi>Y</mml:mi><mml:mi>R</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is a value of <inline-formula><mml:math id="inf195"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> from an iteration of our best-fitting phastCons-based model using polymorphism data from the Yoruba (YRI) population (Section 2.1).</p></list-item><list-item><p>We generated the map of expected neutral diversity levels in autosomes given the chosen parameters. The map was represented in terms of the expected levels at each bin of putatively neutral sites (see Section 1.3).</p></list-item><list-item><p>We generated a polymorphism dataset corresponding to a sample size <inline-formula><mml:math id="inf196"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>108</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> pairs of (haploid) autosomes by picking the number of pairwise differences in each bin such that the average diversity level in it most closely matched the level predicted by the map. The discretization step introduces small differences between average and expected diversity levels in bins.</p></list-item></list><p>We tested each algorithm by applying it to <italic>10 simulated datasets</italic>, with <italic>3 sets of initial conditions</italic> for each dataset, corresponding to weak, intermediate and strong background selection (with <inline-formula><mml:math id="inf197"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="inf198"><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>9</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula><mml:math id="inf199"><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per base pair per generation, respectively), and <italic>5 randomly chosen initial conditions</italic> in each set (with the total rate divided among the 6 selection coefficients by sampling from a Dirichlet distribution with <inline-formula><mml:math id="inf200"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>) amounting to <italic>150 runs</italic>. The initial value of <inline-formula><mml:math id="inf201"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> was always set to the average diversity level in the dataset <inline-formula><mml:math id="inf202"><mml:mover accent="true"><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mo>-</mml:mo></mml:mover></mml:math></inline-formula>.</p><p>None of the algorithms closely converged to the ground truth parameters in all cases. Nelder-Mead downhill simplex minimization (<xref ref-type="bibr" rid="bib73">Nelder and Mead, 1965</xref>) (NM) and Constrained Trust Region minimization (<xref ref-type="bibr" rid="bib23">Conn et al., 2000</xref>) (CTR) performed the best overall, closely recovering the true parameters in ~2/3 of cases. While CTR was slightly more reliable, it was also up to ten times slower than NM. We therefore decide to combine them in order to leverage the relative strengths.</p></sec><sec sec-type="appendix" id="s6-1-9"><title>Two-step minimization algorithm</title><p>After some experimentation we converged on the following two-step algorithm (<xref ref-type="fig" rid="app1fig3">Appendix 1—figure 3</xref>):</p><list list-type="order"><list-item><p>We apply NM with multiple initial conditions. For models of background selection with a single selected annotation we generate 3 sets of initial conditions, with 5 randomly chosen initial conditions per set, as we described above. For models of sweeps with a single annotation we generate the initial conditions analogously. Namely, we generate 3 sets of initial conditions corresponding to a low, intermediate and high proportion of beneficial substitutions (with <inline-formula><mml:math id="inf203"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.0125</mml:mn><mml:mo>,</mml:mo><mml:mn>0.125</mml:mn><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, respectively) with 5 randomly chosen initial conditions per set (with the total proportion divided among selection coefficients by sampling from a Dirichlet distribution with <inline-formula><mml:math id="inf204"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>). For models with background selection and sweeps and/or multiple annotations, we generate 3 sets of initial conditions, corresponding to the weak/low, intermediate, and strong/high categories, with 5 random initial conditions per set that are chosen similarly for each mode and annotation. In all cases, the initial value of <inline-formula><mml:math id="inf205"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is set to the average diversity level in the dataset <inline-formula><mml:math id="inf206"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>π</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p></list-item><list-item><p>We apply the CTR algorithm with a single initial condition that is chosen based on the output of the previous step. Specifically, we focus on the sets of selection parameters inferred in the 3 out of 15 initial runs that yielded the highest composite-likelihood, and use their average as our initial condition.</p></list-item></list><fig id="app1fig3" position="float"><label>Appendix 1—figure 3.</label><caption><title>Illustration of the two-step algorithm.</title><p>In this example, the optimization is applied to a model of background selection with a single selected annotation and a grid of 6 selection coefficients. See text for details.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig3-v2.tif"/></fig><p>We tested the two-step algorithm under a variety of scenarios. When we applied it to the aforementioned ‘deterministically’ simulated datasets corresponding to the best-fitting model of background selection, it always closely recovered the ground truth parameters (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4</xref>). The tiny differences between predicted and simulated diversity levels introduced by discretizing sometimes caused tiny differences between the inferred and ground-truth parameter values (see e.g. <xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4c</xref>), but the composite log-likelihood of the inferred parameters was always higher, indicating that the algorithm is working well. Moreover, the runtime of the CTR algorithm in step 2 was typically short, presumably because its initial conditions were close to the true maximum.</p><fig id="app1fig4" position="float"><label>Appendix 1—figure 4.</label><caption><title>Comparison of inferred and ground-truth parameters for datasets simulated ‘deterministically’ under the best-fitting background selection model.</title><p>Panels <bold>a-d</bold> correspond to different simulated datasets. Boxed region in (<bold>c</bold>) highlights the small differences between inferred and ground truth parameters introduced by discretization.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig4-v2.tif"/></fig><p>We also tested the algorithm on simulated datasets that include substantial noise in diversity levels. We generated the datasets for a sample size <inline-formula><mml:math id="inf207"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> by sampling the number of pairwise differences in a bin of neutral sites from a Binomial distribution with a probability of success that equals the predicted diversity level (replacing step 3 in the simulations described above). The parameters inferred by our optimization algorithm were always similar to those used in the corresponding simulations, but with noticeable differences (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5</xref>). In all cases, however, the composite-likelihood of the inferred parameters was greater than that of the ground-truth parameters indicating that the differences were due to overfitting (which is expected given the noise we introduced in the simulations) rather than a problem in the optimization.</p><fig id="app1fig5" position="float"><label>Appendix 1—figure 5.</label><caption><title>Comparison of inferred and ground-truth parameters for datasets simulated with noise under the best-fitting background selection model.</title><p>Panels <bold>a-d</bold> correspond to different simulated datasets.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig5-v2.tif"/></fig><p>Lastly, we tested the optimization algorithm on datasets simulated under a joint model of background selection and selective sweeps. We modeled the effects of sweeps driven by nonsynonymous substitutions, assuming that they made up <inline-formula><mml:math id="inf208"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.25</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> of the nonsynonymous substitutions on the human lineage since divergence from the common ancestor with chimpanzees (see Section 2.7), and randomly dividing this proportion among 6 selection coefficients of by sampling from a Dirichlet distribution (with <inline-formula><mml:math id="inf209"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>). We modeled background selection as we detailed above, and generated the dataset using the ‘noisy’ simulation scheme corresponding to a sample size of <inline-formula><mml:math id="inf210"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>. The parameters inferred by our optimization algorithm were always similar to those used in the simulations, with greater composite-likelihood of inferred than of ground-truth parameters indicative of overfitting (<xref ref-type="fig" rid="app1fig6">Appendix 1—figure 6</xref>) as we observed in the case with background selection alone. We obtained similar results when we simulated datasets under a variety of scenarios corresponding to the combinations weak, intermediate and strong background selection (<inline-formula><mml:math id="inf211"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="inf212"><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>9</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="inf213"><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per base pair per generation, respectively) with low, intermediate, and high proportions of beneficial substitutions (<inline-formula><mml:math id="inf214"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.0125</mml:mn><mml:mo>,</mml:mo><mml:mn>0.125</mml:mn><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, respectively).</p><fig id="app1fig6" position="float"><label>Appendix 1—figure 6.</label><caption><title>Comparison of inferred and ground-truth parameters for datasets simulated with noise under a joint model of background selection and selective sweeps.</title><p>Panels <bold>a and b</bold> correspond to different simulated datasets.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig6-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-1-10"><title>1.5 Thresholding</title><p>Our inference is strongly affected by forms of model misspecification that cause erroneous predictions of strong background selection effects (i.e., low values of <inline-formula><mml:math id="inf215"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>) and thus of low diversity levels at a relatively small proportion of neutral sites in our dataset. (We refer to neutral rather than putatively neutral sites for brevity and because low error in the identification of neutral sites is irrelevant to the problem at hand). These kinds of erroneous predictions can occur, for example, at neutral sites near regions that are incorrectly annotated as conserved or that are truly conserved but have proportionally fewer weakly deleterious mutations than most similarly annotated regions (because weakly deleterious mutations have strong localized effects on diversity levels). Even when neutral sites near such regions make up a small proportion of the dataset, having more of them be polymorphic than predicted can substantially reduce the composite-likelihood of models that may otherwise fit the data well (see <xref ref-type="disp-formula" rid="equ6">Equation 5</xref>), potentially biasing our inference. Here, we present evidence for this problem, show how we modify our inference to solve it – by imposing a lower threshold for the value of <inline-formula><mml:math id="inf216"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in the lookup tables or in the optimization, and address the consequences of this modification.</p><p>In <xref ref-type="fig" rid="app1fig7">Appendix 1—figures 7</xref>–<xref ref-type="fig" rid="app1fig9">9</xref>, we compare the results of our inference with and without thresholding for our best-fitting CADD-based model (the results for other models are qualitatively similar). Under the aforementioned forms of model misspecification, we might expect excess neutral diversity in regions where background selection is predicted to be strongest. Accordingly, when we apply the inference with little or no thresholding and focus on 1% of neutral sites where background selection is predicted to be the strongest, we find that observed diversity levels are up to twofold higher than our predictions (<xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7a</xref>). Additionally, we expect this form of model misspecification to bias the inferred distribution of selection effects toward larger selection coefficients, because smaller selection effects cause a more localized reduction in diversity levels and are therefore expected to be heavily penalized by having even relatively few misspecified regions. Accordingly, we find that the inferred distribution without thresholding is shifted toward greater selection coefficients (<inline-formula><mml:math id="inf217"><mml:mi>t</mml:mi><mml:mo>≥</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>2.5</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) compared to the distributions with thresholding (<xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7c(i)</xref>).</p><fig id="app1fig7" position="float"><label>Appendix 1—figure 7.</label><caption><title>Comparison of inference results with and without thresholding.</title><p>The results shown correspond to our best-fitting CADD-based model (see Main Text), with threshold values of <inline-formula><mml:math id="inf218"><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> (without threshold, labeled ‘none’), <inline-formula><mml:math id="inf219"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.2</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, <inline-formula><mml:math id="inf220"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.5</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, and <inline-formula><mml:math id="inf221"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.6</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> applied in the lookup tables. (<bold>a</bold>) Observed vs. predicted neutral diversity levels across the autosomes. The graph was generated as detailed in <xref ref-type="fig" rid="fig5">Figure 5</xref>. Note that the division of neutral sites among bins varies with the choices of thresholds because it is based on corresponding maps. (<bold>b</bold>) Observed vs. predicted neutral diversity levels as a function of genetic distance from human-specific nonsynonymous (NS) substitutions. The graph was generated as detailed in <xref ref-type="fig" rid="fig3">Figure 3</xref>, using a narrower range of genetic distances to NS substitutions to highlight differences among thresholds. (<bold>c</bold>) Parameter estimates and summaries of the inferences. From left to right: (i) The estimated distribution of fitness effects, described in terms of the rate of mutation per generation with a given selection coefficient. Mutation rates (throughout) are measured relative to the estimate of the total mutation rate in humans, <inline-formula><mml:math id="inf222"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1.4</mml:mn><mml:mo>∙</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per bp per generation (see Section 5). (ii) The total deleterious mutation rate (<inline-formula><mml:math id="inf223"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) measured in units of <inline-formula><mml:math id="inf224"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. (iii) Our prediction of the mean reduction in neutral diversity level due to background selection, measured as the ratio of the average predicted level across the genome, <inline-formula><mml:math id="inf225"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>π</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, to the predicted level in the absence of selection at linked sites, <inline-formula><mml:math id="inf226"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. (iv) The reduction in composite log-likelihood (CLL) per site relative to the model with the highest CLL. Differences in CLL should be interpreted with caution, as this measure does not account for linkage disequilibrium. (<bold>d</bold>) The proportion of variance in diversity levels explained (<inline-formula><mml:math id="inf227"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) on different spatial scales (measured in non-overlapping windows).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig7-v2.tif"/></fig><p>Importantly, the map of background selection effects generated without thresholding fits the data more poorly than the maps with thresholding. Notably, when we compare observed and predicted diversity levels around nonsynonymous substitutions, we find that the predictions generated without thresholding underestimate the reduction in diversity levels near nonsynonymous substitutions (inset in <xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7b</xref>). This can be explained by the bias toward larger selection coefficients, which causes the inference without thresholding to underestimate the reduction in diversity levels near conserved regions that are specified correctly (in order to avoid the reduction in diversity levels near misspecified regions). Additionally, when we compare the fit of maps with and without thresholding, we find that without thresholding the composite-likelihood is lower (<xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7c(iv)</xref>), the variance in diversity levels explained throughout the range of window sizes is lower (<xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7d</xref> and <xref ref-type="fig" rid="app1fig8">Appendix 1—figure 8</xref>) and the calibration of our predictions is poorer (<xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7a</xref>; this remains the case when we exclude the top and bottom 5% of our predicted values, such that the predictions with and without thresholding span the same ranges of values; for example, Pearson <inline-formula><mml:math id="inf228"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> of <inline-formula><mml:math id="inf229"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.99</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf230"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.97</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> with a threshold of <inline-formula><mml:math id="inf231"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>0.6</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and without thresholding, respectively).</p><p>We considered two ways of thresholding, where in both we set any value of <inline-formula><mml:math id="inf232"><mml:mi>B</mml:mi></mml:math></inline-formula> that is below the threshold to the threshold value: (1) applying the threshold in the lookup tables, that is, <italic>before</italic> the composite-likelihood maximization step, and (2) applying the threshold <italic>at each step</italic> of the maximization, when <inline-formula><mml:math id="inf233"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> values are calculated for a given distribution of selection effects (see <xref ref-type="disp-formula" rid="equ8">Equation 6</xref>). The two approaches yield similar improvements in fit at equivalent threshold levels, and even applying a relatively low threshold improves fits markedly compared to <italic>B-</italic>maps without thresholding (<xref ref-type="fig" rid="app1fig8">Appendix 1—figure 8</xref>). Based on our metrics of fit, we find that applying a threshold of <inline-formula><mml:math id="inf234"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>0.6</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> in the lookup tables yields the best fits (<xref ref-type="fig" rid="app1fig7">Appendix 1—figures 7</xref>–<xref ref-type="fig" rid="app1fig9">9</xref>), although thresholds within the range <inline-formula><mml:math id="inf235"><mml:mn>0.45</mml:mn><mml:mo>≤</mml:mo><mml:mi>B</mml:mi><mml:mo>≤</mml:mo><mml:mn>0.65</mml:mn></mml:math></inline-formula> yield comparable results. Nonetheless, lower thresholds yield better fits to data in regions of the genome where selection is particularly strong (e.g. <xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7a–b</xref>, red box in <xref ref-type="fig" rid="app1fig9">Appendix 1—figure 9</xref>). It may therefore be useful to use a lower threshold when considering regions of the genome that are subject to especially strong background selection. We provide <italic><bold>B</bold></italic>-maps for a range of thresholds that can be downloaded at <ext-link ext-link-type="uri" xlink:href="https://github.com/sellalab/HumanLinkedSelectionMaps">https://github.com/sellalab/HumanLinkedSelectionMaps</ext-link> (in addition to the ‘best-fitting <italic><bold>B</bold></italic>-maps’ presented in the Main Text).</p><fig id="app1fig8" position="float"><label>Appendix 1—figure 8.</label><caption><title>The proportion of variance explained in 10 kb, 100 kb, and 1 Mb windows using a range of <inline-formula><mml:math id="inf236"><mml:mi>B</mml:mi></mml:math></inline-formula> thresholds applied to lookup tables (‘lookup’) or during maximization (‘maximization’).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig8-v2.tif"/></fig><fig id="app1fig9" position="float"><label>Appendix 1—figure 9.</label><caption><title>Predicted and observed diversity levels along chromosome 1 in the YRI sample.</title><p>Diversity levels are measured in 1 Mb windows, with a 0.5 Mb overlap, with the autosomal mean set to 1. Thresholds were applied in the lookup tables. Lower thresholds yield better predictions in regions with low diversity levels, for example, near 50 Mb (red box).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig9-v2.tif"/></fig><p>While thresholding largely resolves the aforementioned problem of model misspecification, it also introduces some problems. First, as we already noted, it leads to an underestimation of background selection effects at ~5% of the genome in which background selection effects is predicted to be the strongest. Second, thresholding potentially biases our estimates of the distribution of selection effects. While this bias is probably smaller than the bias without thresholding, its form and magnitude are not obvious. This is why we decided not to report the inferred distributions of selection effects in the Main Text. We are working on more principled ways of resolving the problems introduced by model misspecification, but these fall beyond the scope of the current paper.</p></sec><sec sec-type="appendix" id="s6-1-11"><title>1.6 Software</title><p>We provide a set of Python programs to download and format the genomic data that we use (see Section 2), infer maps of the effects of linked selection and reproduce all of the analyses and figures described in this study (<ext-link ext-link-type="uri" xlink:href="https://github.com/sellalab/HumanLinkedSelectionMaps">https://github.com/sellalab/HumanLinkedSelectionMaps</ext-link>). We rely on publicly available software for some steps, including the PHAST package (<xref ref-type="bibr" rid="bib98">Siepel and Haussler, 2004</xref>; <xref ref-type="bibr" rid="bib99">Siepel et al., 2005</xref>), which we use to identify conserved regions and to estimate substitution rates (see Sections 3 and 5), and a modified version of the calc_bkgd program from <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>, which we use to generate lookup tables of the effects of background selection (see Section 1.2).</p></sec><sec sec-type="appendix" id="s6-1-12"><title>Running the inference pipeline</title><p>The inference pipeline is controlled by a data structure called RunStruct, which is initialized with information about input/output file paths used, model parameters and other control variables, such as the precision <inline-formula><mml:math id="inf237"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> of lookup tables (see Section 1.2) and the <inline-formula><mml:math id="inf238"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> threshold (see Section 1.5). Once RunStruct has been initialized, the pipeline proceeds through the following steps:</p><list list-type="order"><list-item><p>Download and organize input files (annotations, genetic maps, etc.).</p></list-item><list-item><p>Create lookup tables of the effects of background selection and/or selective sweeps (Section 1.2) for the given set of selected annotations and grid of selection coefficients.</p></list-item><list-item><p>Organize polymorphism dataset that includes polymorphism data at putatively neutral sites (Section 2.1), corresponding estimates of substitution rates (Section 3.3) and corresponding values of lookup table into our compressed bins format (Section 1.3).</p></list-item><list-item><p>Run the two-step optimization algorithm to obtain estimates of model parameters, a map of the predicted effects of linked selection, and summary statistics including, for example, the estimated deleterious mutation rate (<inline-formula><mml:math id="inf239"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>) and proportion of beneficial substitutions (<inline-formula><mml:math id="inf240"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>α</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>) associated with different annotations and the average reduction in diversity levels (<inline-formula><mml:math id="inf241"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>π</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>π</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>).</p></list-item></list></sec><sec sec-type="appendix" id="s6-1-13"><title>Parallelization and runtimes</title><p>The composite-likelihood calculations during optimization can be partitioned into sums over subsets of bins of neutral sites, which in turn allows us to parallelize the optimization. The number of processing cores used in optimization is controlled by RunStruct. For our best-fitting models of background selection, loading lookup tables and neutral polymorphism data and running the two-step optimization requires ~1 GB of memory for each of the 15 processes in step 1 and the single process in step 2. Running each process on a single core takes ~12–24 hr or ~200–400 CPU × GB hours. The computing cluster we used allows up to 12 cores per process and thus using parallelization we were able to run the optimization for the best-fitting models in 1–2 hr. Our most complex models (see Section 4) required up to 10 GB of memory per process and took up to 60 hr with using 12 cores (i.e. ~ 10<sup>4</sup> CPU × GB hours).</p></sec></sec><sec sec-type="appendix" id="s6-2"><title>2. Data sources and filters</title><sec sec-type="appendix" id="s6-2-1"><title>2.1 Polymorphism data</title><p>We download 1000 Genomes Project phase 3 VCF files for all 26 populations from across the world (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>). Unless otherwise noted, results in the Main Text and Appendix 1 are based on autosomal data from Yoruba (YRI); the results for other populations are reported in Sections 7 and 9 of this Appendix 1.</p><p>We apply several filters to these data. First, we restrict our analysis to bases that pass all filters, denoted ‘P’ in the 1000 Genomes Project strictMask accessibility mask (<xref ref-type="bibr" rid="bib1">Abecasis et al., 2012</xref>; <xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>). In addition, we remove low-complexity, simple repeats, duplications, and hg19 build gaps using repeatMasker files downloaded from UCSC (<xref ref-type="bibr" rid="bib58">Karolchik et al., 2004</xref>). For each population, we restrict polymorphic sites to that population’s subset of biallelic SNPs from VCF files, excluding indels and other variants using VCFTools (<xref ref-type="bibr" rid="bib29">Danecek et al., 2011</xref>). Remaining sites are treated as monomorphic.</p><p>We apply additional filters to restrict our analyses to putatively neutral sites. First, we remove the union of genic regions, as detailed in section 2.4. Second, we remove all remaining sites with phastCons conservation scores greater than 0.001 as described in section 3.1. Third, we remove putatively neutral sites at the telomeric ends of autosomes, near the edges of our genetic maps (Section 2.3), as detailed in Section 3.2. Accessibility and repeat masks remove ~33.3% of all autosomal sites; excluding genic regions removes an additional ~3.3%; filtering based on phastCons scores removes another ~40.5%; and filtering sites at the telomeric ends removes ~1.2% more. We are left with a set of ~653 M putatively neutral sites, which correspond to ~23% of autosomal sites (based on hg19 build).</p></sec><sec sec-type="appendix" id="s6-2-2"><title>2.2 Multiple species alignment data</title><p>We rely on multiple sequence alignments to identify phylogenetically conserved and non-conserved regions of the genome, as well as for estimating local variation in neutral substitution rates (see Sections 3 and 4). To this end, we download mutation annotation format (MAF) files containing 99 vertebrate genomes aligned to the human genome (build hg19), using the Multiz software from UCSC (<xref ref-type="bibr" rid="bib15">Blanchette et al., 2004</xref>).</p></sec><sec sec-type="appendix" id="s6-2-3"><title>2.3 Genetic map</title><p>We use the (<xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref>) genetic map, which was inferred from ancestry switches in African-Americans. At the &gt;10 kb scale, it is highly correlated to other fine-scale maps (e.g. <xref ref-type="bibr" rid="bib35">Frazer et al., 2007</xref>; <xref ref-type="bibr" rid="bib43">Halldorsson et al., 2019</xref>). Among high-resolution genetic maps in humans, however, this one is likely the least confounded by diversity levels along the genome.</p></sec><sec sec-type="appendix" id="s6-2-4"><title>2.4 Human gene annotations</title><p>We use genic annotations from the UCSC knownGene database (<xref ref-type="bibr" rid="bib50">Hsu et al., 2006</xref>) to identify putative targets of selection as well as regions that should be removed from our set of putative neutral sites. To this end, we rely on exon coordinates from knownGene transcripts to identify four kinds of annotations: (1) upstream/downstream regions, defined as 1 kb upstream of a transcript start and 1 kb downstream of a transcript end; (2) untranslated region (UTR), both 5’ and 3’; (3) protein coding sequences (CDSs); and (4) splice regions, defined as 200 bp from the start and end of each intron.</p><p>For the purpose of identifying putative targets of selection, we rely on a non-overlapping subset of these four annotations. For genes with multiple splice variants, we keep only the set of exons within the longest isoform. In rare cases of two overlapping gene predictions, we retain the gene with the longer exonic sequence. For the purpose of removing putatively functional regions from our set of putative neutral sites, however, we remove the union of all four annotations for all gene transcripts.</p></sec><sec sec-type="appendix" id="s6-2-5"><title>2.5 CADD scores</title><p>We use CADD scores (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Rentzsch et al., 2019</xref>) in order to annotate putative targets of selection in a couple of models (Sections 4.4 and 4.5). The standard CADD scores rely on the map of background selection effects generated by <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref> as one of their inputs. While this input has minor effects on CADD scores (i.e. the top 1–10% of scores; see Table S3 in <xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>), in order to avoid any measure of circularity we approached the Kircher Lab (Martin Kircher, Lusiné Nazaretyan, Philip Rentzsch and Max Schubach), who manage the development of CADD scores, and who kindly agreed to generate and share a version of CADD score without the background selection map as input (this set of CADD scores is available on request from either the Kircher or Sella labs). For each site in the genome, we retain the highest of the three CADD scores (corresponding to the three possible point mutations). We use the distribution of scores across the autosomes to determine cutoffs for our annotations (e.g. sites within the top 6% of scores) and use sites with scores that exceed these cutoffs as putative targets of selection (sometimes in conjunction with another annotation, e.g. exons).</p></sec><sec sec-type="appendix" id="s6-2-6"><title>2.6 ENCODE cCRE annotations</title><p>In two of our models (Section 4.4), we consider regulatory elements identified by the ENCODE project as putative targets of selection (<xref ref-type="bibr" rid="bib69">Moore et al., 2020</xref>). To this end, we download ENCODE candidate cis-regulatory elements (cCREs) from the Tier 1 a group of biosamples, which include experimental support from all relevant assays used to define elements: high DNase signal and high H3K4me3, H3K27ac or CTCF signal (<xref ref-type="bibr" rid="bib69">Moore et al., 2020</xref>). The resulting cCREs are categorized as (1) enhancer-like signatures (ELS), (2) promoter-like signatures (PLS), (3) CTCF-bound (CTCF) and (4) poised elements marked by DNase and H3K4me3 (H3K4me3). cCRE annotations were downloaded for each individual Tier 1 a biosample using the SCREEN tool (<xref ref-type="bibr" rid="bib69">Moore et al., 2020</xref>) and lifted over from hg38 to hg19 coordinates.</p></sec><sec sec-type="appendix" id="s6-2-7"><title>2.7 Substitutions in the human lineage</title><p>We rely on an estimate of the human-chimpanzee ancestor inferred using the Enredo-Pecan-Ortheus (EPO) 6-species alignment pipeline (<xref ref-type="bibr" rid="bib76">Paten et al., 2008</xref>) to identify likely substitutions on the human lineage. We use subsets of these substitutions that arose in putative targets of positive selection as candidate substitutions resulting in classic sweeps (Section 4.5). We derive sets of likely substitutions in a couple of different ways. First, we compare the reconstructed ancestral genome with the human hg19 reference, taking the differences as putative substitutions. In this case and others, we do not differentiate between low and high confidence calls (lower and upper case, respectively) in the estimated ancestor. Because the hg19 reference genome is a composite of genomes with different ancestries (<xref ref-type="bibr" rid="bib19">Church et al., 2011</xref>), we also consider population-specific inferences of substitutions for YRI and CEU. To this end, we compare the reconstructed ancestral genome with the polymorphism data collected in the 1000 Genome Project for a given population. If a site is monomorphic in the population and differs from the HC ancestor, we include the site in our set of substitutions. For biallelic sites where one of the two alleles is ancestral, we randomly choose one of the alleles with probabilities that are weighted by allele frequency; if the chosen allele is the derived one, the site is considered a substitution. We generate two such samples for a given population to see whether different choices of substitutions affect our results. In practice, each of these sets differs from the set based on the hg19 reference at fewer than 1% of sites, the differences between the two samples for a given population are even smaller, and the results of our inference end up being insensitive to these differences (Section 4.6).</p></sec><sec sec-type="appendix" id="s6-2-8"><title>2.8 Covariates of <inline-formula><mml:math id="inf242"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula></title><p>In Section 8, we ask whether genomic features that covary with <inline-formula><mml:math id="inf243"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> could account for the divergence between observed and predicted diversity levels in the ~2% of sites in which background selection is predicted to be the weakest. In addition to annotations of features whose sources were already mentioned, we also use the following datasets: (1) BED files of CpG islands downloaded from the UCSC Table Browser <xref ref-type="bibr" rid="bib58">Karolchik et al., 2004</xref>; (2) BED files of testis CpG methylation levels in human males downloaded from the GEO database (GEO accession: GSM1127119; <xref ref-type="bibr" rid="bib5">Barrett et al., 2013</xref>); (3) coordinates of C&gt;G hypermutable regions, given at 1 Mb resolution, taken from the Supplemental Information of <xref ref-type="bibr" rid="bib55">Jónsson et al., 2017</xref>; (4) coordinates of centromeres and telomeres taken from the hg19 gaps track in the UCSC Table Browser <xref ref-type="bibr" rid="bib58">Karolchik et al., 2004</xref>; (5) inferred proportions of archaic ancestry in European (CEU) and East-Asian (CHB/CHS) populations based on estimates from <xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>.</p></sec></sec><sec sec-type="appendix" id="s6-3"><title>3. Choice of exogenous parameters</title><p>Fitting our model to data requires several choices beyond those of datasets and filters. Here, we describe how we chose our set of putatively neutral sites and estimate the substitution rate at these sites. In Section 4, we describe how our results depend on the choice of targets of selection.</p><sec sec-type="appendix" id="s6-3-1"><title>3.1 Choosing putatively neutral sites based on phylogenetic conservation</title><p>Our main source of information for choosing the set of putatively neutral sites is the degree of conservation in multiple species alignments. To this end, we rely on running phastCons (<xref ref-type="bibr" rid="bib99">Siepel et al., 2005</xref>) on subsets of the 99-vertebrate alignment (from which we exclude the human genome). PhastCons fits a phylogenetic hidden Markov Model (phylo-HMM) with two states, neutral and conserved, to multiple species alignments of contiguous sites along the genome using the relative substitution rates in the alignment columns to infer conservation. The phastCons score is the posterior probability that any given site is conserved. In principle, including more species in the alignment increases the power to distinguish between conserved and neutral sites (<xref ref-type="fig" rid="app1fig10">Appendix 1—figure 10a</xref>). However, as the phylogenetic distance from humans increases, sequence conservation might become less informative about conservation in humans because of functional turnover (<xref ref-type="bibr" rid="bib118">Ward and Kellis, 2012</xref>; <xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>). In practice, the latter effect is ameliorated by the fact that phastCons only uses information at aligned sites and the proportion of the genome that aligns to the human reference decreases with phylogenetic distance (<xref ref-type="fig" rid="app1fig10">Appendix 1—figure 10b</xref>), especially in regions with considerable turnover.</p><fig id="app1fig10" position="float"><label>Appendix 1—figure 10.</label><caption><title>The distribution of phastCons scores across autosomes for varying phylogenetic distances from humans.</title><p>(<bold>a</bold>) The number of species included at each phylogenetic depth is noted in the caption. As the number of species in the alignment increases, the ability to distinguish between conserved and neutral sites increases. (<bold>b</bold>) The proportion of a species’ sequenced genome that aligns to the human reference (hg19) decreases with their phylogenetic distance from humans. The decrease is not monotonic because of other factors, for example, the quality of the sequencing. The proportion is not 1 for humans because of missing information in the reference genome.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig10-v2.tif"/></fig><p>In relying on phastCons scores to identify a set of putatively neutral sites, we need to choose two parameters: the phylogenetic depth of species included in the alignment and the cutoff conservation score below which a site will be considered neutral. In both cases, we pick the parameter values that maximize the variance in diversity levels explained by our best-fitting models (<xref ref-type="fig" rid="app1fig11">Appendix 1—figure 11</xref>). Given these criteria, we chose to base our set of neutral sites on the alignment of supra-primates (<xref ref-type="fig" rid="app1fig11">Appendix 1—figure 11a</xref>), and use the 35% of sites (in the set remaining after filters and removing genic regions; see Sections 2.1 and 2.4) with the lowest phastCons scores in this alignment, which includes sites with scores ≤ 0.001 (<xref ref-type="fig" rid="app1fig11">Appendix 1—figure 11b</xref>). These choices are robust to the phylogenetic depth used to specify the selection targets (see Section 4) and to the window size in which we measure the variance explained by our model (we show the results for windows of 1 Mb in <xref ref-type="fig" rid="app1fig11">Appendix 1—figure 11</xref>).</p><fig id="app1fig11" position="float"><label>Appendix 1—figure 11.</label><caption><title>The variance in diversity levels explained by our two best-fitting models using different choices of putatively neutral sites.</title><p>In (<bold>a</bold>) we vary the phylogenetic depth of the multi-species alignment (i.e. the maximal phylogenetic distance from humans to any/all of the other species) and in (<bold>b</bold>) we vary the cutoff phastCons score for the least conserved sites included in our set. The best fit corresponds to the least conserved 35% of sites (phastCons scores ≤ 0.001) in the supra-primate alignment (euar).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig11-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-3-2"><title>3.2 Removing sites at the telomeric ends of chromosomes</title><p>The Hinch et al. genetic map (<xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref>) does not include recombination rate estimates for ~0.5–1 Mb at the 5’ and 3’ ends of autosomes. Consequently, we are unable to describe background selection effects of putatively selected regions that lie in these telomeric regions, and our inferences and predictions at putatively neutral sites near the telomeres are less accurate. We therefore exclude putatively neutral sites in telomeric regions not covered by the genetic map. Similar to our approach in the previous section, we choose the map size of the region to remove based on how the choice affects the model fit to diversity levels across autosomes (<xref ref-type="fig" rid="app1fig12">Appendix 1—figure 12a</xref>). We find that filtering putatively neutral sites in 0.1 cM from the edge of the genetic map, which amounts to ~0.8% of neutral sites, largely removes this ‘edge effect’. This genetic distance makes sense, as it is roughly one at which background selection effects of deleterious mutations with <inline-formula><mml:math id="inf244"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> – the strongest selection effects inferred to contribute substantially (<xref ref-type="fig" rid="app1fig12">Appendix 1—figure 12b</xref>) – become negligible. Moreover, our estimates of model parameters are fairly insensitive to the removal of larger regions (<xref ref-type="fig" rid="app1fig12">Appendix 1—figure 12b</xref>).</p><fig id="app1fig12" position="float"><label>Appendix 1—figure 12.</label><caption><title>The effect of removing putatively neutral sites near telomeres on model fit and parameter estimates.</title><p>We show the result for our best-fitting CADD-based model; results for phastCons scores are highly similar (not shown). (<bold>a</bold>) The proportion of variance in diversity levels explained for different window sizes, as a function of the size of the removed region (in cM). (<bold>b</bold>) (<bold>i-iii</bold>) Estimates of model parameters as a function of the size of the removed region (in cM).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig12-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-3-3"><title>3.3 Estimating local variation in mutation rates</title><p>We rely on estimates of substitution rates at putatively neutral sites along the genome to control for the effect of variation in mutation rates on neutral diversity levels (see <xref ref-type="disp-formula" rid="equ2">Equation 1</xref> in Section 1.1). To this end, we use phyloFit (<xref ref-type="bibr" rid="bib98">Siepel and Haussler, 2004</xref>) to estimate the substitution rate in a phylogeny, in windows of putatively neutral sites across the genome. We choose the species to include in the phylogeny based on the following considerations. The number of substitutions in a given window can be approximated by a Poisson random variable with expectation <inline-formula><mml:math id="inf245"><mml:mi>λ</mml:mi></mml:math></inline-formula>, which is proportional to the total branch length of the phylogeny, <inline-formula><mml:math id="inf246"><mml:mi>T</mml:mi></mml:math></inline-formula>, and the number of putatively neutral sites in the window, <inline-formula><mml:math id="inf247"><mml:mi>n</mml:mi></mml:math></inline-formula>. Consequently, the precision of our estimates of the relative mutation rate increase with <inline-formula><mml:math id="inf248"><mml:mi>λ</mml:mi><mml:mo>∝</mml:mo><mml:mi>n</mml:mi><mml:mo>∙</mml:mo><mml:mi>T</mml:mi></mml:math></inline-formula>. Including more species in the phylogeny increase <inline-formula><mml:math id="inf249"><mml:mi>T</mml:mi></mml:math></inline-formula> but reduces <inline-formula><mml:math id="inf250"><mml:mi>n</mml:mi></mml:math></inline-formula>, because it reduces the fraction of putatively neutral sites that align to the human reference in all the species included. <xref ref-type="fig" rid="app1fig13">Appendix 1—figure 13a</xref> shows the trade-off between the two factors, for all subsets of 9 primate species included in the 99-vertebrate alignment (see Section 2.2). We chose the subset that maximizes <inline-formula><mml:math id="inf251"><mml:mi>n</mml:mi><mml:mo>∙</mml:mo><mml:mi>T</mml:mi></mml:math></inline-formula>, which includes 8 of the 9 species (gibbon is removed) with an average of ~0.135 substitutions per putatively neutral site.</p><fig id="app1fig13" position="float"><label>Appendix 1—figure 13.</label><caption><title>Choosing the parameters used in estimating the relative mutation rate at putatively neutral sites.</title><p>(<bold>a</bold>) The trade-off between the fraction of aligned sites and total branch length for subsets of the primate phylogeny. The fraction of aligned sites is estimated for our set of putatively neutral sites, and the total branch length is measured in terms of the average number of substitutions per site on the phylogeny, estimated by phyloFit. The maximum product of the fraction and branch length is attained by including all primates included in the 99-vertebrate alignment other than gibbon. (<bold>b</bold>) The variance in diversity levels explained by our best-fitting models across 1 Mb windows, for different choices of window sizes (i.e. the number of putatively neutral sites) used to control for variation in mutation rates at putatively neutral sites.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig13-v2.tif"/></fig><p>We estimate relative mutation rates along the genome based on the estimated substitution rates in the 8-primate phylogeny in windows with a fixed number of contiguous putatively neutral sites. Using windows with a greater number of sites decreases the sampling error but reduces the spatial resolution of our estimates. We use the variance in diversity levels explained by our best-fitting models as a criterion for choosing the window size, finding that a window with 6000 putatively neutral sites performs best among the options we examined (<xref ref-type="fig" rid="app1fig13">Appendix 1—figure 13b</xref>). This choice corresponds to mean physical window sizes of 26,454 bp (with a S.D. of 18,455 bp) and to a mean relative error of ~3.3% in our estimates of the relative mutation rate per window. We also examined other ways of estimating the relative mutation rate, including using windows of fixed physical length and sliding windows with varying degrees of overlap, but none of these approaches yielded better results.</p><p>In the analyses in which we bin neutral sites, either by their distance to genomic elements (e.g. <xref ref-type="fig" rid="fig3">Figure 3</xref>) or by predicted <inline-formula><mml:math id="inf252"><mml:mi>B</mml:mi></mml:math></inline-formula> (e.g. <xref ref-type="fig" rid="fig5">Figure 5</xref>), we estimate the relative mutation rate in each bin. To this end, we use phyloFit (<xref ref-type="bibr" rid="bib98">Siepel and Haussler, 2004</xref>) to estimate the substitution rate in the 8-primate phylogeny on all sites in that bin jointly and then normalize this estimate by the average across bins.</p></sec></sec><sec sec-type="appendix" id="s6-4"><title>4. Fitting models with different targets of selection</title><p>Our framework allows us to fit models of background selection, selective sweeps, or both, based on different choices of putative targets of negative and/or positive selection. Here we detail the analysis of the models and choices that are described in the Main Text. We use several criteria to evaluate how well the models fit the data; these indicate that models of background selection alone in which the targets of selection are chosen based on constrained elements annotated by either phastCons or CADD scores are best supported by the data. We also compare the predictions of these models with those of <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>.</p><sec sec-type="appendix" id="s6-4-1"><title>4.1. Background selection model based on phylogenetic conservation</title><p>We first consider a model of background selection in which targets of selection are chosen based on phylogenetic conservation. We identify conserved genomic elements using phastCons scores (<xref ref-type="bibr" rid="bib99">Siepel et al., 2005</xref>) calculated on monophyletic subsets of the 99-vertebrate alignment to the human genome (<xref ref-type="bibr" rid="bib15">Blanchette et al., 2004</xref>), all of which exclude the human genome itself (see Section 2.2). We vary the phylogenetic depth of the subset of species considered (i.e. the maximal distance from humans). For a given depth, we obtain targets of selection by specifying a proportion of selected sites (i.e. of the total autosomal length in hg19) and choosing those sites that have the highest phastCons scores in the alignment (after excluding some sites, e.g. from up to 5% of the four-ape alignment to less than 0.1% of the 99-vertebrate alignment, that are in our putatively neutral set). As we have done for previous choices (e.g. Section 3.1), we examine how our choices of phylogenetic depth and of proportion of selected sites affect the models’ fit to autosomal diversity levels.</p><p>We find the fit to be largely insensitive to the choice of phylogenetic depth, with models based on conservation in the full 99-vertebrate alignment fitting slightly better than other choices of depth (<xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>). Notably, the explained variance in diversity levels (in windows of different sizes) is similar across depths, increasing slightly with the number of species included, other than for the four-ape phylogeny (<xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14b and c</xref>). The fits of predicted diversity levels along the genome (e.g. <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14d</xref>) and around genomic features (e.g. <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14e</xref>) are similar, with none of the choices of depth clearly outperforming others. Moreover, for all choices, the predicted diversity levels are well calibrated (<xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14f</xref>), with the exception of regions in which background selection is predicted to be very weak, that is, <inline-formula><mml:math id="inf253"><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> (see Section 8). When we restrict each annotation to the top 6% of scores in sites for which all phylogenetic depths include phastCons scores (~98% of sites satisfy this criterion), our results are unchanged.</p><p>Distantly related species, such as those added when we move from supra-primates (<inline-formula><mml:math id="inf254"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>25</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>) to vertebrates out to lamprey (<inline-formula><mml:math id="inf255"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>99</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>), have little effect on phastCons scores and thus on our models, because only a small proportion of their genomes align with humans (<xref ref-type="fig" rid="app1fig10">Appendix 1—figure 10b</xref>). This can be seen in the high correlations between the number of conserved sites based on different depths across windows of different sizes (<xref ref-type="fig" rid="app1fig15">Appendix 1—figure 15a</xref>). The spatial distribution of conserved sites is even fairly insensitive to varying the species included from four apes to 99 vertebrates (<xref ref-type="fig" rid="app1fig15">Appendix 1—figure 15a</xref>). Interestingly, we later show that the improvement in fit across 1 Mb windows of the model based on conservation in 99 vertebrates compared with models based on conservation in shallower phylogenies is statistically significant, except for the model based on four-apes (<xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33</xref>), whereas the spatial distributions of conserved sites in the 99-vertebrate and four-ape models are the least correlated (<xref ref-type="fig" rid="app1fig15">Appendix 1—figure 15</xref>). The v-shaped dependence on phylogenetic depth may reflect a tradeoff in which phastCons scores based on deeper alignments have greater power to identify long-lived selected regions (see, e.g. <xref ref-type="fig" rid="app1fig10">Appendix 1—figure 10a</xref>), whereas those based on apes are better at identifying regions that are selected in humans but exhibited functional turnover in the deeper phylogeny (<xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>; see also Section 6.2).</p><p>The model fit is also fairly insensitive to the cutoff conservation score used in choosing selection targets, although choosing 5–7% of autosomal sites as targets does appear to yield slightly better fits than other choices (<xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16</xref>). Notably, the variance explained for different window sizes is maximized between 5–7% (<xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16b and c</xref>); at the higher end of the range of cutoffs from 2% to 9%, the fits of diversity levels along the genome (e.g. <xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16d</xref>) and around genomic features (e.g., <xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16e</xref>) appear to be slightly worse, and the stratification of observed values by predicted ones spans a smaller range (<xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16f</xref>). Among comparisons between models based on 6% and all other cutoffs in the range of 2–9%, only 8 and 9% lead to a statistically significant reduction of fit in windows of 1 Mb (<xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33</xref>). Based on these analyses, we use the model with the 6% of autosomal sites with the highest phastCons scores based on the 99-vertebrate alignment in many of our analyses, and refer to this as our <italic>best-fitting phastCons-based model</italic> in both the Main Text and throughout Appendix 1.</p><fig id="app1fig14" position="float"><label>Appendix 1—figure 14.</label><caption><title>Comparison of background selection models based on phastCons conservation scores in phylogenies of difference depths.</title><p>Shown are results of models based on conservation in four apes, eight primates, 12 prosimians, 25 supra-primates, 50 laurasiatherians, 61 mammals, and 99 vertebrates extending out to lamprey. In all cases, we take the 6% of autosomal sites with the highest phastCons scores (excluding putatively neutral sites) as our targets of selection. Throughout Appendix 1, with the exception of Section 7, we show results using data from YRI. <bold><italic>The panels describe</italic>:</bold> (<bold>a</bold>) Parameters and summaries of models (from left to right): (<bold>i</bold>) Estimated distribution of fitness effects, described in terms of the rate of mutations with given selection coefficients. Mutation rates throughout are measured relative to the estimate of the estimated average mutation rate per bp per generation in humans, <inline-formula><mml:math id="inf256"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1.4</mml:mn><mml:mo>∙</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> (see Section 5). As detailed in Section 1.5, the inferred distribution of selection coefficients should be interpreted with caution. (<bold>ii</bold>) Estimated total deleterious mutation rate per selected site (<inline-formula><mml:math id="inf257"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) measured in units of <inline-formula><mml:math id="inf258"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. (<bold>iii</bold>) Estimated autosomal average fold-reduction in neutral diversity levels due to selection at linked sites, i.e., the ratio of average predicted heterozygosity, <inline-formula><mml:math id="inf259"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>π</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, to average predicted heterozygosity in the absence of selection at linked sites, <inline-formula><mml:math id="inf260"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. (<bold>iv</bold>) The reduction in composite log-likelihood (CLL) per site relative to the model with the highest CLL. Differences in CLL should be interpreted with caution, as diversity levels at putatively neutral sites are not independent. (<bold>b</bold>) The proportion of variance in diversity levels explained (<inline-formula><mml:math id="inf261"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) on different spatial scales (measured in non-overlapping contiguous windows). (<bold>c</bold>) Close-up on the variance explained for several window sizes. (<bold>d</bold>) Predicted and observed diversity levels along chromosome 1. Diversity levels are measured in 1 Mb windows, with 0.5 Mb overlap, and are normalized by the mean level (as detailed in <xref ref-type="fig" rid="fig2">Figure 2</xref>). The results here and in subsequent panels are shown for a subset of depths, including four apes, 25 supra-primates and 99 vertebrates. (<bold>e</bold>) Predicted and observed diversity levels as a function of genetic distance to the nearest human-specific nonsynonymous (NS) substitutions. The plot was generated as detailed in <xref ref-type="fig" rid="fig3">Figure 3</xref>. Inset shows closeup between –0.05 and 0.05 cM. (<bold>f</bold>) Observed vs. predicted neutral diversity levels across the autosomes. The plot was generated as detailed in <xref ref-type="fig" rid="fig5">Figure 5</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig14-v2.tif"/></fig><fig id="app1fig15" position="float"><label>Appendix 1—figure 15.</label><caption><title>The spatial distribution of putatively selected sites remains similar when we vary the phylogenetic depth of the alignment used to infer conservation (shown in <bold>a</bold>), and the proportion of sites with the highest conservation scores included (in <bold>b</bold>).</title><p>We compare two choices of selection targets at a time, and show the Pearson correlations (<inline-formula><mml:math id="inf262"><mml:mi>ρ</mml:mi></mml:math></inline-formula>) between the numbers of putatively selected sites among windows of different genetic lengths (measured in Morgans). The range of window sizes roughly corresponds to the spatial scales over which selection affects linked neutral diversity for the estimated range of selection effects. When we vary the phylogenetic depth, we use the 6% of autosomal sites with the highest phastCons scores, and when we vary the conservation cutoff, we use phastCons scores based on the 99-vertebrate alignment. (<bold>c</bold>) The deleterious mutation rate per gamete per generation inferred as a function of assumed proportion of selected sites in autosomes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig15-v2.tif"/></fig><p>The insensitivity of our fits to varying the conservation cutoff can be understood as follows. phastCons estimates the probability that runs of sites belong to conserved segments (<xref ref-type="bibr" rid="bib99">Siepel et al., 2005</xref>). When we reduce the conservation cutoff, shorter segments with high scores tend to expand to include adjacent, lower scoring sites. This results in a high spatial correlation between the conserved sites corresponding to different cutoffs (<xref ref-type="fig" rid="app1fig15">Appendix 1—figure 15b</xref>). Given a lower conservation cutoff and longer ‘selected’ segments, we infer a lower deleterious mutation rate per site (<xref ref-type="fig" rid="app1fig16">Appendix 1—figure 16a(ii)</xref>) but a similar deleterious mutation rate per segment (see, e.g. <xref ref-type="fig" rid="app1fig15">Appendix 1—figure 15c</xref>), thereby producing similar troughs in diversity around such segments and similar fits overall.</p><fig id="app1fig16" position="float"><label>Appendix 1—figure 16.</label><caption><title>Comparison of background selection models based on phastCons scores using different proportions of autosomal sites as selection targets.</title><p>In all cases considered, we rely on conservation in 99 vertebrates. Otherwise, all panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig16-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-4-2"><title>4.2 Background selection model based on genic annotations</title><p>Next, we consider a model of background selection in which selection targets are chosen based on simple genic annotations, i.e., the exons divided into UTRs and protein coding sequences (CDSs), as well as regions in the immediate vicinity of these sequences controlling transcript regulation: regions 1 kb up- and downstream of transcript start/end, and splice regions 200 bp at the start and end of introns (<xref ref-type="bibr" rid="bib14">Black, 2003</xref>; <xref ref-type="bibr" rid="bib61">Kim et al., 2005</xref>; see Section 2.4 for details). We allow selection parameters to vary among annotations, but find that in the best-fitting model only protein coding and splice regions have non-negligible deleterious mutation rates (for other annotations, <inline-formula><mml:math id="inf263"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula>).</p><p>We also find that this model fits much worse than our best-fitting phastCons-based model (<xref ref-type="fig" rid="app1fig17">Appendix 1—figure 17</xref>): the variance in diversity levels it explains is substantially lower across different window sizes (<xref ref-type="fig" rid="app1fig17">Appendix 1—figure 17b</xref>), its fit to diversity levels along the genome is discernably worse (e.g. <xref ref-type="fig" rid="app1fig17">Appendix 1—figure 17c</xref>), and when observed diversity levels are stratified by the model’s predictions, they are less calibrated (<xref ref-type="fig" rid="app1fig17">Appendix 1—figure 17e</xref>). The genic model does do reasonably well at predicting how diversity levels drop with genetic distance around nonsynonymous substitutions (e.g. <xref ref-type="fig" rid="app1fig17">Appendix 1—figure 17d</xref>). The generally poorer fit as well as the reasonably good fit around nonsynonymous substitutions can be understood in terms of the overlap between our simple genic annotations and direct measures of constraint (<xref ref-type="fig" rid="app1fig18">Appendix 1—figure 18</xref>). Namely, the genic annotations miss most constrained sites, which are intronic or intergenic (<xref ref-type="fig" rid="app1fig18">Appendix 1—figure 18b</xref>), but most protein coding regions (CDSs) are constrained (<xref ref-type="fig" rid="app1fig18">Appendix 1—figure 18a</xref>) explaining why models including them as an annotation perform well near them.</p><fig id="app1fig17" position="float"><label>Appendix 1—figure 17.</label><caption><title>The background selection model based on simple genic annotations fits worse than our best-fitting phastCons-based model.</title><p>All the panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref> (but with the hatch-marked blue bars in <bold>a </bold>(<bold>i</bold>) and (<bold>ii</bold>) corresponding to different annotations of the genic model).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig17-v2.tif"/></fig><fig id="app1fig18" position="float"><label>Appendix 1—figure 18.</label><caption><title>The relationship between simple genic annotations and our main measures of constraint.</title><p>Specifically, we examine the overlap of the 6% of autosomal sites with the highest phastCons or CADD scores with the genic annotation detailed in the text; we added intronic (INTRON) and intergenic (INTERG) annotations for completeness. (<bold>a</bold>) The fraction of each genic annotation within the 6% most constrained sites. (<bold>b</bold>) The fraction of the 6% most constrained within each genic annotation. (<bold>c</bold>) Enrichment of genic annotations in the 6% most constrained sites, i.e., the ratio of their proportion among constrained and all autosomal sites.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig18-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-4-3"><title>4.3 Background selection models separating conserved exonic and non-exonic sites</title><p>While background selection models based on simple genic annotations do worse than those based on phylogenetic conservation, using such annotations in conjunction with conservation could allow for improved fits. Notably, it is often argued that purifying selection in protein coding regions is stronger than in functional non-coding regions (<xref ref-type="bibr" rid="bib59">Kellis et al., 2014</xref>; <xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>); if this were true, then allowing them to have different selection parameters could result in better fits. To examine this possibility, we fit a model with two types of selection target: exonic (i.e. segments combining CDSs and UTRs) and non-exonic conserved sites (see details in Section 2.4).</p><p>We infer a higher deleterious mutation rate and stronger selection in exonic compared to non-exonic sites (<xref ref-type="fig" rid="app1fig19">Appendix 1—figure 19a</xref>), although we note that our estimates of selection parameters could be affected by thresholding (see Section 1.5). The total deleterious mutation rate per gamete is similar in models with and without the exonic/non-exonic division (<inline-formula><mml:math id="inf264"><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mn>1.6</mml:mn></mml:math></inline-formula> and <inline-formula><mml:math id="inf265"><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mn>1.73</mml:mn></mml:math></inline-formula> per gamete per generation, respectively), but the (weighted) average selection effect is greater in the model with the division (<inline-formula><mml:math id="inf266"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mn>1.71</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula> vs. <inline-formula><mml:math id="inf267"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>s</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mn>6.8</mml:mn><mml:mtext> </mml:mtext><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula> for the models with and without division, respectively), primarily due to stronger selection in conserved exonic sites. Overall, despite affording additional parameters, dividing conserved sites into exonic and non-exonic has little effect on our fits (<xref ref-type="fig" rid="app1fig19">Appendix 1—figure 19b–e</xref>).</p><p>Regardless of whether we separate exonic and non-exonic conserved sites, most of the reduction in diversity levels is caused by selection in non-exonic regions. Weakly selected mutations cause a large reduction in neutral diversity levels over short genetic distances, whereas strongly selected mutations cause a weak reduction over long genetic distances; but the integral reduction in diversity levels due to weak and strong selection on a given set of deleterious mutations end up roughly equivalent (<xref ref-type="bibr" rid="bib52">Hudson, 1994</xref>). This property allows us to use estimates of the total deleterious mutation rates in conserved exonic and non-exonic regions as a rough measure of their proportional effects on neutral diversity levels, despite differences in selection effects in these regions. These estimates suggest that ~80% of deleterious mutations occur in non-exonic regions, indicating that they account for most of the reduction in linked neutral diversity (e.g. in the model with the top 6% of phastCons scores, ~84% of selected sites and ~76% of deleterious mutations are non-exonic; with the top 6% of CADD scores, ~83% of selected sites and ~85% of deleterious mutations are non-exonic; also see discussion in Section 4.6).</p><p>Given that the bulk of deleterious mutations exerting background selection occur in non-exonic regions, it is not surprising that a model including only conserved non-exonic sites fits the data only slightly worse than a model including all conserved sites as targets of selection (<xref ref-type="fig" rid="app1fig20">Appendix 1—figure 20</xref>). By the same token, it is not surprising that a model including only conserved exonic sites fits the data substantially worse than models with either conserved non-exonic or all conserved sites as targets of selection (<xref ref-type="fig" rid="app1fig20">Appendix 1—figure 20</xref>). Moreover, the estimate of the deleterious mutation rate per site in the exonic model is much higher than in the other two (<xref ref-type="fig" rid="app1fig20">Appendix 1—figure 20a(ii)</xref>).</p><fig id="app1fig19" position="float"><label>Appendix 1—figure 19.</label><caption><title>Dividing conserved sites into exonic and non-exonic sets leads to different estimates of selection parameters in each, but to little improvement in fit compared to the model based on conservation alone.</title><p>Our set of conserved sites consists of the 6% of sites with the highest phastCons scores in the 99-vertebrate alignment (see Section 4.1). All panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>. Because of thresholding (Section 1.5), the model based on conservation alone is not formally nested in the one with the division into exonic and non-exonic sets, explaining how its maximum composite-likelihood can be slightly greater.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig19-v2.tif"/></fig><fig id="app1fig20" position="float"><label>Appendix 1—figure 20.</label><caption><title>Comparison of background models using exonic, non-exonic and all conserved sites as targets of selection.</title><p>Our set of conserved sites consists of the 6% of sites with the highest phastCons scores in the 99-vertebrate alignment (see Section 4.1). All panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig20-v2.tif"/></fig><p>It is somewhat surprising that the model based on conserved exonic sites alone fits the data as well as it does (<xref ref-type="fig" rid="app1fig20">Appendix 1—figure 20b and c</xref>). This can be understood by noting that the spatial distribution of conserved exonic sites and of all conserved sites are fairly highly correlated (<xref ref-type="fig" rid="app1fig21">Appendix 1—figure 21</xref>). Given similar spatial distributions of selected sites, the distribution of background selection effects in the model with all conserved sites can be approximated by having a higher deleterious mutation rate per site at the fewer selected sites in the exonic model. These considerations explain why we infer a similar (albeit lower) average reduction in diversity levels but a substantially higher deleterious mutation rate in the exonic model (<xref ref-type="fig" rid="app1fig20">Appendix 1—figure 20a(ii) and (iii)</xref>). They also help to explain differences between our inferences and those of <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>, notably their implausibly high estimate of the deleterious mutation rate given that their main model assumes selection only at conserved exonic sites (see Main Text and Section 4.6).</p><fig id="app1fig21" position="float"><label>Appendix 1—figure 21.</label><caption><title>The spatial correlations of exonic, non-exonic and all conserved sites for varying window sizes (‘con<sub>e</sub>’, ‘con<sub>n</sub>’ and ‘con<sub>a</sub>‘, respectively).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig21-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-4-4"><title>4.4 Background selection models based on other annotations</title><p>We consider two additional widely-used functional annotations as putative background selection targets. First, we rely on the expanded encyclopedias of DNA elements (ENCODE) annotations of candidate cis-regulatory elements (cCREs), including enhancer-like signatures (ELS), promoter-like signatures (PLS), CTCF-bound (CTCF) and poised/DNAse-hypersensitive (H3K4me3) assayed in 25 Tier 1 a biosamples (<xref ref-type="bibr" rid="bib69">Moore et al., 2020</xref>), alongside protein coding sequences (CDSs) (see Sections 2.4 and 2.6 for data sources and definition of elements). ENCODE cCREs attempt to capture the diverse repertoire of regulatory elements across cell types that control gene expression in different cellular and biological contexts. They are based on a large set of epigenomic assays, including ChIP-seq measuring the occupancy of histone marks associated with both activation and repression of gene expression, pulldown of DNA-bound transcription factors, and DNA accessibility measured in terms of DNAse sensitivity. Since we infer the majority of autosomal sites under purifying selection to be non-exonic (see Section 4.3), we reason that some combination of cCREs may substantially overlap these sites. Importantly, cCRE annotations may allow us to better partition non-exonic regions into sub-classes of sites experiencing different selection strengths. We define our choices of selection targets (other than CDSs) by grouping cCRE in two alternative ways. In one, we take the union of cCREs of a given type over all 25 biosamples. In the other, we divide cCREs of a given type into those identified in few (≤ median number) or in many (&gt; median number) biosamples (in practice, most cCREs included in the first set are cell-type specific whereas most of those in the second are found in a few to all cell-types). The model in which cCREs of a given type are split performs slightly better, presumably because of the additional degrees of freedom. Both models, however, fit the data substantially worse than either of our best-fitting models (<xref ref-type="fig" rid="app1fig22">Appendix 1—figure 22</xref>). The poor fit accords with the modest overlap between cCREs and our estimates of constraint sites (<xref ref-type="fig" rid="app1fig23">Appendix 1—figure 23</xref>). Moreover, PLSs, the cCREs that are most highly enriched in constrained sites (<xref ref-type="fig" rid="app1fig23">Appendix 1—figure 23a and c</xref>) are inferred to have a negligible deleterious mutation rate.</p><fig id="app1fig22" position="float"><label>Appendix 1—figure 22.</label><caption><title>The model based on the ENCODE annotations of cCRE fit the data substantially worse than our best-fitting phastCons-based model using conservation in the 99-vertebrate alignment (conserved).</title><p>The results shown correspond to the model in which we split each type of cCREs into those that occur in few (subscript 1) and many biosamples (subscript 2). We infer a non-negligible deleterious mutation rate (i.e. <inline-formula><mml:math id="inf268"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&gt;</mml:mo><mml:mn>0.01</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>) in 2 of the 8 cCRE-based putative selection targets: enhancer like sequences and CTCF binding sites identified in few biosamples, ELS<sub>1</sub> and CTCF<sub>1</sub> respectively, as well as in protein coding regions (CDS). All the panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig22-v2.tif"/></fig><fig id="app1fig23" position="float"><label>Appendix 1—figure 23.</label><caption><title>The relationship between ENCODE cCRE annotations and our main measures of constraint.</title><p>Specifically, we examine the overlap of the 6% of autosomal sites with the highest phastCons or CADD scores with promoter like sequences (PLS), enhancer like sequences (ELS), CTCF-bound (CTCF), poised/DNAse-hypersensitive (H3K4me3), as well as sites that are not in any of these annotations (NONE). (<bold>a</bold>) The fraction of each cCRE annotation within the 6% most constrained sites. (<bold>b</bold>) The fraction of the 6% most constrained within each cCRE annotation. (<bold>c</bold>) Enrichment of cCRE annotations in the 6% most constrained sites, that is, the ratio of their proportion among constrained and all autosomal sites.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig23-v2.tif"/></fig><p>Next, we consider Combined Annotation-Dependent Depletion (CADD) scores (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Rentzsch et al., 2019</xref>). CADD scores predict the ‘deleteriousness’ of every point mutation in the genome. They are generated by using machine learning to integrate information from a diverse set of annotations (122 annotations in version 1.6), such as measures of phylogenetic conservation (including phastCons scores based on the 99-vertebrate alignment), predictions of regulatory elements (including many of the assays used for constructing the ENCODE cCREs), genic annotations (including those described in Sections 2.7 and 4.2) and predicted functional consequences of variants in protein coding sequences. The algorithm is trained using the depletion of 14.7 million high-frequency (&gt;95%) derived alleles (based on 1000 Genomes Data) relative to 14.7 million simulated variants with the same genomic distribution as the criterion for ‘deleteriousness’. While the standard CADD scores (version 1.6) incorporate the <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref> map of background selection effects as one of the annotations, we use a version in which this annotation was excluded in order to avoid circularity (see Section 2.5). We use the maximal score at each site (corresponding to the most deleterious of three possible point mutations), and, for comparison with our best-fitting phastCons-based model (Section 4.1), we begin by considering the 6% of autosomal sites with the highest CADD scores (excluding putatively neutral sites) as targets of selection.</p><p>Despite incorporating many sources of information beyond phylogenetic conservation, and doing better than phastCons scores at predicting functional consequences of variants at a single site resolution (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>), the model based on CADD scores offers only a minor improvement over our best-fitting phastCons-based model (<xref ref-type="fig" rid="app1fig24">Appendix 1—figure 24</xref>). For example, the model based on CADD scores explains 59.9% of the variance in diversity levels in 1 Mb windows compared to 59.7% for the model based on phastCons scores, although this difference and differences in other window sizes are not statistically significant (see <xref ref-type="fig" rid="app1fig32">Appendix 1—figure 32</xref> and Section 6.2). The little improvement is not that surprising, given that phylogenetic conservation is the annotation most correlated with CADD scores genome-wide (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>), and that the spatial distributions of sites with top CADD and phastCons scores are highly correlated on the spatial scales that impact background selection effects (<xref ref-type="fig" rid="app1fig25">Appendix 1—figure 25</xref>).</p><fig id="app1fig24" position="float"><label>Appendix 1—figure 24.</label><caption><title>The model based on CADD scores offer little improvement over the model based on phastCons scores (based on the 99-vertebrate alignment).</title><p>In both cases, we take the 6% of sites with the highest scores. All the panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig24-v2.tif"/></fig><fig id="app1fig25" position="float"><label>Appendix 1—figure 25.</label><caption><title>The spatial correlation between the 6% of sites with the highest CADD and phastCons scores.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig25-v2.tif"/></fig><p>The fit of models based on CADD scores is fairly insensitive to the proportion of sites included as selection targets, with proportions of 5–7% yielding slightly better fits than other choices (<xref ref-type="fig" rid="app1fig26">Appendix 1—figure 26</xref>). This insensitivity and the increase in estimates of the deleterious mutation rate per site with decreasing proportion of sites used as selection targets (<xref ref-type="fig" rid="app1fig26">Appendix 1—figure 26a(ii)</xref>) can be explained in the same way that we explained similar observations for models based on phastCons scores (Section 4.1).</p><fig id="app1fig26" position="float"><label>Appendix 1—figure 26.</label><caption><title>Comparison of background selection models based on CADD scores using different proportions of autosomal sites as selection targets.</title><p>All panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig26-v2.tif"/></fig><p>Based on the analyses in <xref ref-type="fig" rid="app1fig24">Appendix 1—figure 24</xref> and <xref ref-type="fig" rid="app1fig26">Appendix 1—figure 26</xref>, we refer to the model with the 6% of autosomal sites with the highest CADD scores as our <italic>best-fitting CADD-based model</italic>, and use it in most of our analyses here and in the Main Text. While the differences in fit of our best-fitting CADD-based and phastCons-based models are minor, the improved predictions of CADD compared to phastCons scores at the single site resolution substantially affects our estimates of the deleterious mutation rate based on evolutionary rates and thus their agreement with estimates based on the effects of background selection (see Main Text and Section 5).</p></sec><sec sec-type="appendix" id="s6-4-5"><title>4.5 Models with selective sweeps</title><p>Next, we examine whether models that include both background selection and selective sweeps fit the data better than models with background selection alone. Our inference should be able to tease apart the effects of sweeps, primarily because these effects, unlike those of background selection, are centered around the locations of substitutions. This feature should hold true for hard, partial or soft sweeps (<xref ref-type="bibr" rid="bib46">Hermisson and Pennings, 2005</xref>; <xref ref-type="bibr" rid="bib86">Przeworski et al., 2005</xref>; <xref ref-type="bibr" rid="bib79">Pennings and Hermisson, 2006a</xref>; <xref ref-type="bibr" rid="bib80">Pennings and Hermisson, 2006b</xref>; <xref ref-type="bibr" rid="bib25">Coop and Ralph, 2012</xref>; <xref ref-type="bibr" rid="bib11">Berg and Coop, 2015</xref>), so long as they result in substitutions and have a substantial effect on diversity levels (SOM Section D in <xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>). Indeed, previous work that applied a similar methodology to data from <italic>Drosophila melanogaster</italic> was able to identify and quantify distinct signatures of background selection and sweeps alongside one another (<xref ref-type="bibr" rid="bib32">Elyashiv et al., 2016</xref>).</p><p>We consider a variety of models characterized by different sets of putatively selected sites. For background selection, we consider the two sets used in our best-fitting models based on phastCons and CADD scores. We also consider several choices for targets of <italic>positive selection</italic>, i.e., for sweeps, corresponding to different kinds of substitutions that we infer to have occurred on the human lineage from the common ancestor with chimpanzees (see Section 2.7). Notably, we consider models that include the set of all nonsynonymous substitutions paired with either of the two sets for background selection. We also consider models with the substitutions that have occurred at sites with the top 2%, 3%, …, 9% of phastCons or CADD scores, where in each case we separate substitutions into sets of nonsynonymous and other, and pair that choice with the corresponding set for background selection (i.e. based on phastCons or CADD scores). For each of these choices, we infer the set of substitutions on the human lineage in two ways, either comparing the estimated human-chimpanzee ancestral genome (<xref ref-type="bibr" rid="bib76">Paten et al., 2008</xref>) with the human reference genome (hg19) or with a population (YRI or CEU) sample of human genomes (see Section 2.7). We perform the inference for all of these models (18 in total) using the same grid of selection coefficients for each of the sets of selected sites, and data from either YRI or CEU. <italic>In all cases, our estimate of the fraction of beneficial substitutions,</italic> <inline-formula><mml:math id="inf269"><mml:mi>α</mml:mi></mml:math></inline-formula><italic>, is essentially 0</italic> (&lt; 10<sup>-9</sup>). We do not show the results because they are indistinguishable from those for the corresponding models with background selection alone (i.e., see <xref ref-type="fig" rid="app1fig16">Appendix 1—figures 16</xref> and <xref ref-type="fig" rid="app1fig26">26</xref>).</p><p>We also consider models with sweeps alone. <xref ref-type="fig" rid="app1fig27">Appendix 1—figure 27</xref> shows the results for a subset of these models, including the best-fitting one (e.g. based on variance explained). These models fit the data substantially worse than those with background selection alone, as seen by each of our measures (interestingly, even when considering the reduction in diversity levels around nonsynonymous substitutions; <xref ref-type="fig" rid="app1fig27">Appendix 1—figure 27d</xref>). Sweep models do account for substantial variance in diversity levels, but given that they add nothing to a model of background selection alone yet fit much worse, this is plausibly because they approximate some of the effects of background selection. Notably, both background selection and sweeps cause reductions in diversity levels near selected sites, and the densities of sites that give rise to background selection and sweeps in the corresponding models are spatially correlated along the genome (<xref ref-type="fig" rid="app1fig28">Appendix 1—figure 28</xref>). Moreover, the sweep models that fit the data best are those that rely on substitutions whose spatial distributions are the most highly correlated with the distributions of selection targets in our best-fitting background selection models (e.g. compare the fits and correlations for the models based on substitutions in the most conserved 2% and 9% of sites in <xref ref-type="fig" rid="app1fig27">Appendix 1—figures 27</xref> and <xref ref-type="fig" rid="app1fig28">28</xref>). Taken together, the evidence presented here supports previous studies (<xref ref-type="bibr" rid="bib24">Coop et al., 2009</xref>; <xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>) indicating that sweeps had little effect on current diversity levels and that background selection is the dominant mode of linked selection in humans.</p><fig id="app1fig27" position="float"><label>Appendix 1—figure 27.</label><caption><title>Models with sweeps alone fit substantially worse than models with background selection alone.</title><p>Shown are the results for sweep models based on either: all nonsynonymous substitutions (NS); nonsynonymous and other substitutions at sites within the top 9% of phastCons scores (9%: NS/other); or nonsynonymous and other substitutions at sites within the top 2% of phastCons scores (2%: NS/other). For comparison, we also show the results of our best-fitting phastCons-based background selection model (conserved). The panels are as described in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>, with the exception of the bottom halves of panels <bold>a</bold> (<bold>i</bold>) and (<bold>ii</bold>), which show the proportions of substitutions that are estimated to be adaptive for a given selection coefficient (<bold>i</bold>) or in total (<bold>ii</bold>), for different annotations and sweeps models.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig27-v2.tif"/></fig><fig id="app1fig28" position="float"><label>Appendix 1—figure 28.</label><caption><title>The spatial correlation between targets of selection in sweep models and in our best-fitting phastCons-based background selection model.</title><p>Results shown for sweep models based on human-specific substitutions at sites within the top 2% and 9% of phastCons scores (see text for details).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig28-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-4-6"><title>4.6 Comparison with previous work by McVicker et al</title><p>For completeness, we conclude by comparing our inferences about the effects of background selection with those of <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>. The McVicker et al. study was done more than a decade ago, before genome-wide resequencing polymorphism data were available. Instead, they ingeniously used a five-primate alignment of ~4.7 million putatively neutral sites, relying on incomplete lineage sorting between human, chimpanzee and gorilla in order to learn about variation in the effective population size along the genome of the common ancestor of humans and chimpanzees. We rely on diversity levels in samples of 108 individuals at ~653 million putatively neutral sites (Section 2.1). Similar to this study, they relied on conservation scores and estimates of neutral substitution rates based on multiple sequence alignments, but they based themselves on the genomes of 15 placental mammals when we have 99 aligned vertebrate genomes at our disposal (Section 2.2). Lastly, they used a genetic map based on LD patterns (<xref ref-type="bibr" rid="bib71">Myers et al., 2005</xref>), whereas we rely on genetic maps based on ancestry switches in African Americans (<xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref>). The McVicker study also differed in several aspects of the methodology. Notably, McVicker et al. did not incorporate selective sweeps into their models, and were therefore unable to exclude the possibility that sweeps had made a substantial contribution to their inferred effects of background selection (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>). Also, McVicker et al. assumed that selection coefficients are distributed exponentially, whereas we assumed a more flexible (non-parametric) distribution on a grid. Despite limitations, the McVicker et al. maps of the effects of background selection capture substantial variation in diversity levels along the human genome (<xref ref-type="fig" rid="app1fig29">Appendix 1—figure 29a</xref> and Figure 7 in <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>).</p><fig id="app1fig29" position="float"><label>Appendix 1—figure 29.</label><caption><title>Our maps of the effects of background selection fit the data much better than the maps from <xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>.</title><p>Shown are the results for our best-fitting CADD-based model. All panels are as described for the corresponding ones in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig29-v2.tif"/></fig><p>Nonetheless, our maps of the effects of background selection fit the data substantially better than the map from McVicker et al., both quantitatively and qualitatively (<xref ref-type="fig" rid="app1fig29">Appendix 1—figure 29</xref>). They explain considerably greater proportions of the variance in diversity levels across window sizes (<xref ref-type="fig" rid="app1fig29">Appendix 1—figure 29c</xref>); for example, they explain ~60% compared to ~32% of the variance on the 1 Mb scale. Our predictions are well calibrated, whereas those of McVicker et al. are not (<xref ref-type="fig" rid="app1fig29">Appendix 1—figure 29d</xref>). Our predictions also do substantially better at capturing diversity patterns near specific genomic features, as illustrated by the fit to diversity levels around nonsynonymous substitutions (<xref ref-type="fig" rid="app1fig29">Appendix 1—figure 29b</xref>). The relatively poor quantitative fit of the McVicker et al. predictions around synonymous and nonsynonymous substitutions (<xref ref-type="bibr" rid="bib47">Hernandez et al., 2011</xref>) was used to argue that the effects of background selection could be more pronounced around synonymous than nonsynonymous substitutions, thereby masking the effects of selective sweeps (<xref ref-type="bibr" rid="bib33">Enard et al., 2014</xref>). In this regard, the close fit of our predictions helps to refute one of two arguments for a residual, important role of selective sweeps.</p><p>We turn to the second argument, regarding estimates of the deleterious mutation rate, next. Our work and that of McVicker et al. differ markedly in our inferences about the rate and genomic distribution of deleterious mutations causing background selection in humans. In fact, the main problem in interpreting the McVicker et al. findings in terms of background selection alone is that they are based on an estimated deleterious mutation rate of <inline-formula><mml:math id="inf270"><mml:mn>7.4</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per generation at their ‘conserved exonic’ sites (defined as sites within the top 5.3% of conservation scores in segments that overlap exons, accounting for ~1.1% of euchromatic autosomal sites) – more than fivefold higher than current estimates of the total mutation rate per site (see next Section). In contrast, as we detail in the next section, our estimates of the deleterious mutation rate per selected site are quite plausible (<inline-formula><mml:math id="inf271"><mml:mn>1.00</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per generation for both of our best-fitting models based on phastCons and CADD scores; <xref ref-type="fig" rid="fig4">Figure 4</xref> in Main Text). The results of McVicker et al. further suggest that background selection arises predominantly from deleterious mutations in the ‘conserved exonic’ regions covering ~1.1% of euchromatic autosomal sites (i.e. they estimate ~2.3 mutations per gamete per generation in such regions in exons compared to ~0.1 elsewhere). In contrast, our results suggest that background selection arises mostly from deleterious mutations at non-exonic sites (i.e. from ~1.22 and~1.27 mutations per gamete per generation in non-exonic compared to ~0.38 and~0.23 mutations in exonic sites in the models based on phastCons and CADD scores, respectively). Notably, in our best-fitting models, these deleterious mutations occur in 6% of autosomal sites as opposed to only ~1% in the McVicker et al. model. Having the effects of background selection arise from deleterious mutations in a substantially greater fraction of the genome largely explains why our estimates of the deleterious mutation rate are much lower and much more plausible (<xref ref-type="fig" rid="fig4">Figure 4</xref> and <xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30</xref>).</p></sec></sec><sec sec-type="appendix" id="s6-5"><title>5. Assessing estimates of the deleterious mutation rate</title><p>Here, we consider the plausibility of the deleterious mutation rate that we estimated by fitting models of background selection. First, we consider the total mutation rate per site in humans, which provides an upper bound on the deleterious mutation rate. Second, we rely on the reduction in substitution rates at our selection targets relative to putative neutral sites to obtain estimates of the proportion of mutations at selected sites that are deleterious. These estimates should be largely independent of those that we obtained by fitting background selection models, and can therefore be used to evaluate the plausibility of the latter. Lastly, we briefly consider to what extent we should expect the two kinds of estimates to line up.</p><sec sec-type="appendix" id="s6-5-1"><title>5.1 Estimates of the total mutation rate per site</title><p>The total mutation rate per site includes contributions from point mutations, indels, mobile element insertion (MEIs) and copy variants such as inversions. Current estimates of mutation rates per site per generation in humans are <inline-formula><mml:math id="inf272"><mml:mn>1.2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for point mutations (<xref ref-type="bibr" rid="bib64">Kong et al., 2012</xref>; <xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>), <inline-formula><mml:math id="inf273"><mml:mn>8.79</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>9.82</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for indels (<xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>), whereas the rate for MEIs and other structural variants (including inversions and duplications) are more than two orders of magnitude lower than the point mutation rate (<xref ref-type="bibr" rid="bib106">Sudmant et al., 2015</xref>; <xref ref-type="bibr" rid="bib37">Gardner et al., 2019</xref>; <xref ref-type="bibr" rid="bib10">Belyeu et al., 2021</xref>), making their contribution to our calculations below negligible. Adding up point mutation and indel rates results in a per site per generation estimate of <inline-formula><mml:math id="inf274"><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.38</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. In estimating an upper bound on the rate of deleterious mutations at selected sites, we may consider weighting deletions by their length. For instance, we would like to count a deletion that begins at a neutral site but includes a selected site yet avoid counting one that includes multiple selected sites more than once. Counting deletions, which account for ~0.725 of indels, between once and up to their mean size of ~2.88 bp (<xref ref-type="bibr" rid="bib13">Besenbacher et al., 2016</xref>), yields estimates of the total mutation rate in the range of <inline-formula><mml:math id="inf275"><mml:mn>1.29</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>1.51</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per site per generation. Throughout the paper, we use the middle of this range, that is, <inline-formula><mml:math id="inf276"><mml:mn>1.4</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> per site per generation, as our estimate for the total mutation rate (<inline-formula><mml:math id="inf277"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>). The estimates of the deleterious rate per putatively selected site for our best-fitting models fall well below the estimated total mutation rate (<xref ref-type="fig" rid="fig4">Figure 4</xref> and Sections 4.1 and 4.4), as one would hope.</p></sec><sec sec-type="appendix" id="s6-5-2"><title>5.2 Estimating the proportion of deleterious mutations at putatively selected sites</title><p>Next, we estimate the proportional reduction of the substitution rate at selection targets relative to that at putatively neutral sites. We apply phyloFit (<xref ref-type="bibr" rid="bib98">Siepel and Haussler, 2004</xref>) to the human-chimp-gorilla-orangutan (HCGO) alignment (based on the HCGO sequences from the 99-vertebrate alignment described in Section 2.2) in order to estimate the substitution rate per site on the human lineage from the ancestor with chimpanzee, for sets of selected and neutral sites (Section 3.1). To control for differences in base composition between the two sets, we estimate the reduction in substitution rates separately for each type of ancestral nucleotide (e.g. substitutions from G&gt;X), and weight the proportional reductions by the proportions of each nucleotide in the set of selected sites. Controlling for the composition of triplets rather than single nucleotides produces similar estimates. Note that in choosing our sets of neutral and selected sites based on phylogenetic conservation (Sections 3.1 and 4.1), we excluded the human genome from the alignments and therefore our estimates of the reduction in substitution rates on the human branch should be minimally confounded with the choice of sites. Similarly, the conservation scores that serve as input for calculating CADD scores are based on the same 99-vertebrate alignment excluding the human reference genome (see Supplementary Table 1 in <xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>).</p><p>The estimates of the proportion of mutations that are deleterious are shown in <xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30</xref> (and <xref ref-type="fig" rid="fig4">Figure 4</xref>), along with their comparison to estimates from background selection models. Expectedly, estimates based on substitution rates decline slightly as the cutoff phastCons or CADD score decreases (i.e. as the percentage of sites included in the selected set increases; <xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30a and b</xref>). Importantly, estimates based on substitution rates are substantially greater for the sets chosen based on CADD than on phastCons scores (<xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30a and b</xref>), whereas estimates based on background selection effects are similar in both cases (<xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30d and e</xref>). We interpret this finding as reflecting the greater ability of CADD scores to identify selection on a single site resolution (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>), plausibly because CADD scores incorporate measures of phylogenetic conservation based on one site at a time (e.g. phyloP, GERP; <xref ref-type="bibr" rid="bib26">Cooper et al., 2005</xref>; <xref ref-type="bibr" rid="bib3">Apostolico et al., 2006</xref>) in addition to measures that rely on runs of sites, such as phastCons scores. Consequently, the two estimates of the deleterious mutation rate are within a factor of 2 for our best-fitting phastCons-based model whereas they overlap for our best-fitting CADD-based model, while the estimates based on background selection effects are similar in both cases (<xref ref-type="fig" rid="app1fig30">Appendix 1—figure 30d and e</xref>).</p><fig id="app1fig30" position="float"><label>Appendix 1—figure 30.</label><caption><title>Different estimates of the deleterious mutation rate at putatively selected sites, measured relative to total mutation rates per site (<inline-formula><mml:math id="inf278"><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>).</title><p>Estimates based on evolutionary rates are shown for sets of selected sites chosen based on either: (<bold>a</bold>) the top 4–8% of phastCons scores for the 99-vertebrate alignment, (<bold>b</bold>) the top 4–8% of CADD scores, or (<bold>c</bold>) the top 6% of phastCons scores for alignments of varying phylogenetic depths. As expected, (see Section 4.1), the estimates are fairly insensitive to the phylogenetic depth (<bold>c</bold>). (<bold>d and e</bold>) Estimates based on evolutionary rates vs. those based on the effects of background selection, for sets of putatively selected sites based on phastCons scores for the 99-vertebrate alignment (<bold>d</bold>) and on CADD scores (<bold>e</bold>). (<bold>f</bold>) Estimates for different sets of putatively selected sites based on evolutionary rates (ER) and background selection effects (BS). The range of estimates based on background selection effects (in <bold>d-f</bold>) is due to the uncertainty about the total mutation rate per site (Section 5.1).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig30-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-5-3"><title>5.3 Interpreting the relationship between the two estimates</title><p>We would expect the two estimators of the deleterious mutation rate to yield similar but not identical answers. For one, the range of selection coefficients that cause a substantial reduction in evolutionary rates, e.g., <inline-formula><mml:math id="inf279"><mml:mn>4</mml:mn><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>s</mml:mi><mml:mo>≳</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib62">Kimura and Crow, 1964</xref>), is greater than the range of effects that cause a substantial reduction in diversity levels via classic background selection, e.g., <inline-formula><mml:math id="inf280"><mml:mn>4</mml:mn><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mi>s</mml:mi><mml:mo>≳</mml:mo><mml:mn>10</mml:mn></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib17">Charlesworth et al., 1993</xref>; <xref ref-type="bibr" rid="bib67">McVean and Charlesworth, 2000</xref>; <xref ref-type="bibr" rid="bib40">Gordo and Charlesworth, 2001</xref>). This consideration suggests that estimates based on evolutionary rates should be greater than those based on the effects of background selection (although non-equilibrium demographic history, notably changes in population size, might complicate quantitative expectations). On the other hand, we cannot expect to identify all selected sites and only those by our criteria. Estimates based on the effects of background selection plausibly soak up much of the contribution of missing selected sites, because their spatial distribution is likely to be highly correlated with sites that are included in our sets (see Section 4.1). In contrast, estimates based on evolutionary rates are affected only by the sites in our sets and would be biased downwards by the accidental inclusion of effectively neutral sites. For these reasons, we do not expect the two estimates of the deleterious mutation rate to align perfectly. Nonetheless, it is encouraging that when we rely on selected sites that amount to current estimates of the proportion the human genome under selection, that is,~5–9% (<xref ref-type="bibr" rid="bib59">Kellis et al., 2014</xref>; <xref ref-type="bibr" rid="bib88">Rands et al., 2014</xref>), our two estimates of the deleterious mutation rate are quite similar. Moreover, the similarity is highest when we use CADD scores, which are better than phastCons scores at identifying selection on a single site resolution (<xref ref-type="bibr" rid="bib63">Kircher et al., 2014</xref>). Thus, our results resolve the issues raised by the substantial overestimation of the deleterious mutation rate in past work (<xref ref-type="bibr" rid="bib68">McVicker et al., 2009</xref>).</p></sec></sec><sec sec-type="appendix" id="s6-6"><title>6. Statistics</title><sec sec-type="appendix" id="s6-6-1"><title>6.1 Estimates of explained variance</title><p>Our main quantitative measure for the fit of our models is the variance in diversity levels explained by our predictions, <inline-formula><mml:math id="inf281"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, for different window sizes. A concern in using <inline-formula><mml:math id="inf282"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> as a measure of fit is that it not be inflated by overfitting. To avoid this problem, we exclude the data in a given window from the inference used to predict diversity levels in that window. Specifically, we divide the autosomal polymorphism data into contiguous non-overlapping blocks of 2 Mb, and repeat the inference using the data excluding one block at a time. As the autosomes cover just over 2.88 Gb, this amounts to repeating our inference 1440 times. When we calculate the contribution of a given window (of size ≤ 2 Mb) to <inline-formula><mml:math id="inf283"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, we use the prediction based on excluding the 2 Mb block containing that window. In Section 6.3, we use the same datasets and inferences to calculate jackknife estimates for the sampling error of our estimates of model parameters.</p><p><xref ref-type="fig" rid="app1fig31">Appendix 1—figure 31</xref> shows the relative difference between our <inline-formula><mml:math id="inf284"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> estimates using all the data and this exclusion approach for our two best-fitting models. Additionally, in <xref ref-type="fig" rid="app1fig48">Appendix 1—figure 48</xref> we compare the results of our main analyses of the best-fitting CADD-based model, using all the data, out-of-sample predictions in contiguous, non-overlapping 2 MB blocks, and out-of-sample predictions for each autosome. These results suggest that overfitting has a tiny effect, which is not surprising given the large amounts of data used in our inferences. Given the negligible effect and computational burden of these analyses, we do not repeat it for each of the models we examine, and use the <inline-formula><mml:math id="inf285"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> estimates based on predictions using all the data instead.</p><fig id="app1fig31" position="float"><label>Appendix 1—figure 31.</label><caption><title>The relative difference between <inline-formula><mml:math id="inf286"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> estimates using all the data and the exclusion approach described in the text, for our two best-fitting models.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig31-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-6-2"><title>6.2 Comparing the fit of different maps</title><p>We use permutations of paired maps to test whether differences in <inline-formula><mml:math id="inf287"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> between two maps are statistically significant. Assume without loss of generality that <inline-formula><mml:math id="inf288"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for a given window size is greater for map I than for map II. We divide autosomes into 2881 contiguous non-overlapping blocks of 1 Mb and generate a new map (map A) by picking each 1 Mb block from map I or II at random; we generate the complementary map (map B) by picking the alternative 1 Mb blocks throughout. This way, we generate <inline-formula><mml:math id="inf289"><mml:mi>n</mml:mi></mml:math></inline-formula> paired maps and calculate the difference in <inline-formula><mml:math id="inf290"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> between each pair, <inline-formula><mml:math id="inf291"><mml:mi mathvariant="normal">Δ</mml:mi><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, to obtain a distribution for the expected differences in <inline-formula><mml:math id="inf292"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> between maps I and II under the null hypothesis that their fit to polymorphism data is roughly equivalent. Having <inline-formula><mml:math id="inf293"><mml:mi>r</mml:mi></mml:math></inline-formula> denote the number of permutations with <inline-formula><mml:math id="inf294"><mml:mi mathvariant="normal">Δ</mml:mi><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> greater than or equal to the observed difference <inline-formula><mml:math id="inf295"><mml:mi mathvariant="normal">Δ</mml:mi><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula>, we estimate the p-value for <inline-formula><mml:math id="inf296"><mml:mi mathvariant="normal">Δ</mml:mi><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> under the null by <inline-formula><mml:math id="inf297"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>r</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><p>We illustrate this procedure by comparing our two best-fitting models (<xref ref-type="fig" rid="app1fig32">Appendix 1—figure 32</xref>). The fit of these models is very similar, with, for example, <inline-formula><mml:math id="inf298"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0.599</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf299"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.597</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> at the 1 Mb scale for the models based on CADD and phastCons scores, respectively. We find that the difference between the fits is not statistically significant, supporting our claim that the functional annotations incorporated in CADD offer little or no improvement in predictive power (see Main Text and Section 4.4). Using this procedure to compare our best-fitting phastCons-based model with phastCons-based models with alternative phylogenetic depths or conservation thresholds, we only find significant differences at the 1 Mb scale in a small subset of cases (<xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33a and b</xref>). The same is true for comparisons of our best-fitting CADD-based model with CADD-based models with alternative thresholds (<xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33c</xref>). We note that even when the fits are significantly worse, they are still far closer to the fits of our best-fitting models than any of the models based on other choices of selection targets discussed in Section 4.</p><fig id="app1fig32" position="float"><label>Appendix 1—figure 32.</label><caption><title>Assessing the differences in fit between our best-fitting models on three spatial scales.</title><p>We show the distribution of differences in explained variance (<inline-formula><mml:math id="inf300"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mstyle></mml:math></inline-formula>) for 10,000 paired permutations of the best-fitting CADD and phastCons based maps. The part of the distributions with <inline-formula><mml:math id="inf301"><mml:msup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>≥</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> is in red, and the corresponding p-value is shown above.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig32-v2.tif"/></fig><fig id="app1fig33" position="float"><label>Appendix 1—figure 33.</label><caption><title>Significance level of differences in fit between our best-fitting models and variations on these models, on the 1 Mb spatial scale.</title><p>(<bold>a and b</bold>) Comparison of our best-fitting phastCons-based model with phastCons-based models with alternative phylogenetic depths (<bold>a</bold>) and conservation thresholds (<bold>b</bold>). (<bold>c</bold>) Comparisons of our best-fitting CADD-based model with CADD-based models with alternative thresholds.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig33-v2.tif"/></fig><p>Interestingly, the difference in fit between the model based on conservation in 99-vertebrates and four-apes is the only non-significant comparison across phylogenetic depths (<xref ref-type="fig" rid="app1fig33">Appendix 1—figure 33a</xref>), despite the fact that other phylogenetic depths have <inline-formula><mml:math id="inf302"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> values closer to the 99-vertebrate result across various window sizes (<xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14c</xref>). We believe this is due to the fact that the four-ape alignment may actually better capture some recent targets of selection, but with greater noise, reflecting a tradeoff in the power to detect conservation vs. functional turnover. As a result, a non-negligible subset of 1 Mb windows in the four-ape yield better fits to the data than the 99-vertebrate map. In contrast, maps from deeper in the phylogeny are essentially highly correlated to the 99-vertebrate map, and the small differences in fit are uniformly biased in favor of the 99-vertebrate map across 1 Mb windows due to its greater power to resolve the boundaries of conserved elements.</p></sec><sec sec-type="appendix" id="s6-6-3"><title>6.3 Sampling error in parameter estimates</title><p>We use a jackknife resampling approach to estimate the sampling errors of our parameter estimates (see, e.g. <xref ref-type="bibr" rid="bib77">Patterson et al., 2012</xref>). To this end, we perform the inference on datasets excluding 2 Mb blocks as described in Section 6.1. Specifically, denoting the parameter of interest by <inline-formula><mml:math id="inf303"><mml:mi>θ</mml:mi></mml:math></inline-formula>, and the estimate based on the data set excluding block <inline-formula><mml:math id="inf304"><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:math></inline-formula> by <inline-formula><mml:math id="inf305"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>θ</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, our jackknife estimates of the mean and variance are <inline-formula><mml:math id="inf306"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mi>θ</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>θ</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf307"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mover><mml:mi>θ</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:msub><mml:mi>θ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>−</mml:mo><mml:mrow><mml:mover><mml:mrow><mml:mi>θ</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> respectively. We use the standard deviation <inline-formula><mml:math id="inf308"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msqrt><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mover><mml:mi>θ</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:msqrt></mml:mrow></mml:mstyle></mml:math></inline-formula> as our measure of sampling error (SE). <xref ref-type="fig" rid="app1fig34">Appendix 1—figure 34</xref> shows the SEs for the parameter estimates of our two best-fitting models. As these examples illustrate, these errors are quite small and do not affect the conclusions of our analyses. Consequently, and given the computational cost of obtaining them, we do not calculate these SEs for most models.</p><fig id="app1fig34" position="float"><label>Appendix 1—figure 34.</label><caption><title>Estimates of the sampling errors of parameter estimates for our two best-fitting models.</title><p>The bars denote <inline-formula><mml:math id="inf309"><mml:mo>±</mml:mo><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:math></inline-formula> estimated using jackknife as described in the text.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig34-v2.tif"/></fig></sec></sec><sec sec-type="appendix" id="s6-7"><title>7. Results for other human populations</title><p>Here, we examine whether the maps of the effects of background selection that we infer and evaluate using polymorphism data from YRI provide a good fit to data from other populations. To this end, we use data from each of the other 25 populations collected in Phase III of the 1000 genomes project (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>), which span a wide geographic range and have had different demographic histories (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>), to infer the maps corresponding to our best-fitting models. The population-specific maps can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/sellalab/HumanLinkedSelectionMaps">https://github.com/sellalab/HumanLinkedSelectionMaps</ext-link>.</p><p>Overall, we find that the maps and main parameters inferred in different populations are remarkably similar (<xref ref-type="fig" rid="app1fig35">Appendix 1—figures 35, 50</xref>–<xref ref-type="fig" rid="app1fig51">52</xref>). When we compare the predictions of relative diversity levels along autosomes (i.e. relative to the mean in each population) we find nearly perfect correlations across window sizes (<xref ref-type="fig" rid="app1fig35">Appendix 1—figure 35a and b</xref>). The distributions of selection effects of deleterious mutations, estimates of the total deleterious mutation rate per selected site, and the mean reduction in diversity levels, are all quite close among populations (<xref ref-type="fig" rid="app1fig35">Appendix 1—figure 35c</xref>). The similarity among maps implies that we can use the maps of relative diversity levels inferred in YRI to predict diversity levels in other populations without loss of accuracy. Specifically, we multiply predictions based on YRI by a constant, chosen such that the predicted and observed mean diversity level in the focal population match. <xref ref-type="fig" rid="app1fig36">Appendix 1—figure 36</xref> illustrates that the adjusted YRI maps predict diversity levels as well as the population specific ones.</p><fig id="app1fig35" position="float"><label>Appendix 1—figure 35.</label><caption><title>The maps and parameter estimates for different populations are remarkably similar.</title><p>Shown are the results for our best-fitting models using data for one of 1000 Genomes Project populations from each continental group: Africa – Yoruba (YRI), Europe – North-Western European (CEU), South Asia – Gujrati Indian (GIH), East Asia – Japanese (JPT), and Americas – Mexican (MXL). (<bold>a and b</bold>) The Pearson correlations between predictions of relative diversity levels (compared to the population mean) in YRI vs. the other populations for the models based on phastCons (<bold>a</bold>) and CADD (<bold>b</bold>) scores. (<bold>c</bold>) Comparison of parameter estimates using data from these populations (panels <bold>c</bold> (<bold>i-iii</bold>) as described in panels <bold>a</bold> (<bold>i-iii</bold>) in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig35-v2.tif"/></fig><p>While the maps inferred in different populations are highly similar, the proportion of variance explained differs substantially among populations (<xref ref-type="fig" rid="app1fig36">Appendix 1—figure 36</xref>). These differences can be explained by the effects of different demographic histories (e.g. historical changes in effective population sizes) on variation in diversity levels across the genome. To make this more concrete, we consider a simple model for the variance in neutral diversity levels in non-overlapping windows of a given size; for simplicity, we ignore variation in mutation rates across windows. We denote the relative (average) diversity level in window <inline-formula><mml:math id="inf310"><mml:mi>i</mml:mi></mml:math></inline-formula> by <inline-formula><mml:math id="inf311"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>π</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>π</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, the predicted relative diversity level in that window by <inline-formula><mml:math id="inf312"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>B</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, and the corresponding residual by <inline-formula><mml:math id="inf313"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, where <inline-formula><mml:math id="inf314"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mover><mml:mi>f</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf315"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>e</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>. We can now decompose the variance in relative diversity levels across windows, as<disp-formula id="equ22"><label>(15)</label><mml:math id="m22"><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>f</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>e</mml:mi><mml:mo>,</mml:mo><mml:mi>f</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><fig id="app1fig36" position="float"><label>Appendix 1—figure 36.</label><caption><title>The proportion of variance in diversity levels explained (<inline-formula><mml:math id="inf316"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) in different populations, using the population specific map vs. the YRI map.</title><p>We show the results for three window sizes (10 kb, 100 kb, and 1 Mb) based on our best-fitting CADD-based model. Each point corresponds to one of the 26 populations sampled in the 1000 genomes project and is colored based on continental origin, i.e., African (AFR), European (EUR), South Asian (SAS), East Asian (EAS), and American (AMR).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig36-v2.tif"/></fig><p>where <inline-formula><mml:math id="inf317"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> corresponds to the variance due background selection; <inline-formula><mml:math id="inf318"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, the variance of residuals, can be thought of as reflecting the effects of drift and demographic history; and the covariance, <inline-formula><mml:math id="inf319"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, can be thought of as reflecting the interaction between background selection and demographic history. Recasting the proportion of variance explained in these terms, we find that<disp-formula id="equ23"><label>(16)</label><mml:math id="m23"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>≡</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mfrac><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>f</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>V</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>β</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width="negativethinmathspace"/><mml:mspace width="negativethinmathspace"/><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf320"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>⋅</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the slope of the linear regression of the residuals against the predictions, which reflects the effects of interactions between background selection and demographic history on diversity levels.</p><p>Given that we found the predicted effects of background selection to be highly similar across populations, this modeling exercise sets up a testable prediction: if the interaction terms were nil, the difference in <inline-formula><mml:math id="inf321"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> among populations should come from the total variance in the denominator, <inline-formula><mml:math id="inf322"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, and specifically from the contribution of demographic history to this variance, <inline-formula><mml:math id="inf323"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>. <xref ref-type="fig" rid="app1fig37">Appendix 1—figure 37a–c</xref> suggest that while most of the differences in <inline-formula><mml:math id="inf324"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> among populations are indeed explained by differences in total variance due to demographic history, the interaction terms (the <inline-formula><mml:math id="inf325"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>β</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>s) are non-zero. To examine these interactions further, we look at the relationship between residuals, <inline-formula><mml:math id="inf326"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and predictions, <inline-formula><mml:math id="inf327"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, in several populations (<xref ref-type="fig" rid="app1fig37">Appendix 1—figure 37d–h</xref>). We find a strong apparent dependency at the low and high ends of our predicted range, presumably reflecting artifacts due to thresholding at the low end (see Section 1.5) and possibly the effects of ancient introgression at the high end (see Section 8); removing 1.5% of the windows at each of these extremes appears to largely remove these effects. In the rest of the range, we find a weak negative correlation between residuals and predictions (which becomes somewhat stronger when we remove the ends). We also find that this correlation varies substantially among populations, for example, –0.06 to –0.1 for 1 Mb windows, which is what we would expect given differences in demographic history. Thus, our analysis suggests that interactions between demographic history and background selection also contribute to the differences in <inline-formula><mml:math id="inf328"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> among populations.</p><fig id="app1fig37" position="float"><label>Appendix 1—figure 37.</label><caption><title>Differences among populations in the variance in diversity levels explained by our map of the effects background selection.</title><p>(<bold>a–c</bold>) The variance explained (<inline-formula><mml:math id="inf329"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) as a function of <inline-formula><mml:math id="inf330"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> (where <inline-formula><mml:math id="inf331"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the total variance), for three choices of window size. If the interaction terms (<inline-formula><mml:math id="inf332"><mml:mi>β</mml:mi></mml:math></inline-formula>) were 0, we would expect populations to fall on the dashed line <inline-formula><mml:math id="inf333"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>⋅</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, with slope <inline-formula><mml:math id="inf334"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and differences in <inline-formula><mml:math id="inf335"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> due to demographic history. The distances from the dashed line reflect the interaction terms (specifically, <inline-formula><mml:math id="inf336"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>−</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>β</mml:mi><mml:mo>⋅</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>; <xref ref-type="disp-formula" rid="equ22">Equation 15</xref>). The points correspond to the 26 populations sampled in the 1000 genomes project (<xref ref-type="bibr" rid="bib4">Auton et al., 2015</xref>) and are colored by continental origin as described in <xref ref-type="fig" rid="app1fig36">Appendix 1—figure 36</xref>. Here we base our predictions on our best-fitting CADD-based map in YRI, but using other population-specific maps yields almost identical results. (<bold>d-h</bold>) The relationship between the residuals (<inline-formula><mml:math id="inf337"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) and predictions (<inline-formula><mml:math id="inf338"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) in representative populations (same as in <xref ref-type="fig" rid="app1fig36">Appendix 1—figure 36</xref>) on the 1 Mb scale. The 1.5% of windows with lowest and highest predicted values, where our predictions are likely off for various reasons (see Sections 1.5 and 8), are marked in blue. The <inline-formula><mml:math id="inf339"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>β</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>s for each population, with and without extreme points, are shown on the graph (denoted ‘a’ for ‘all’ and ‘t’ for ‘trimmed’, respectively). As above, we use the predictions in YRI, but other population maps yield qualitatively similar results (<inline-formula><mml:math id="inf340"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>β</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>s obtained using the corresponding population specific maps are shown in parenthesis).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig37-v2.tif"/></fig><p>In summary, our findings suggest that the effects of background selection are similar across human populations, and that differences among populations in the proportion of variance in diversity levels that our predictions explain are likely due to differences in population demographic history. Interestingly, there appears to be an interaction between the effects of background selection and demography on diversity levels, which varies among populations, as recently suggested by several studies (<xref ref-type="bibr" rid="bib22">Comeron, 2017</xref>; <xref ref-type="bibr" rid="bib117">Wang et al., 2017</xref>; <xref ref-type="bibr" rid="bib109">Torres et al., 2018</xref>; <xref ref-type="bibr" rid="bib110">Torres et al., 2020</xref>). Our maps of the effects of background selection have enabled us to identify evidence for these interaction effects and should facilitate a better understanding of these effects in the future.</p></sec><sec sec-type="appendix" id="s6-8"><title>8. Diversity levels where background selection is weakest (<inline-formula><mml:math id="inf341"><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>)</title><p>Our maps of background selection effects are well calibrated throughout the range of predicted effects, with two exceptions. One is in the ~5% of sites in which background selection is predicted to be strongest, where predictions are imprecise; this arises from the thresholding approximation we apply in fitting, and is discussed in Section 1.5. The other exception is for sites in which background selection is predicted to be the weakest, where observed diversity levels are markedly greater than expected (<xref ref-type="fig" rid="fig5">Figure 5</xref> and <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38</xref>). A close up on this region shows that observed values depart from predictions in the ~2% of sites where <inline-formula><mml:math id="inf342"><mml:mi>B</mml:mi><mml:mo>≳</mml:mo><mml:mn>0.98</mml:mn></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38b</xref>). Similar behavior is seen in all 26 populations sampled in the 1000 Genomes Project (<xref ref-type="fig" rid="app1fig52">Appendix 1—figure 52</xref>). We consider possible explanations for it here.</p><fig id="app1fig38" position="float"><label>Appendix 1—figure 38.</label><caption><title>Observed vs. predicted neutral diversity levels across the autosomes (similar to <xref ref-type="fig" rid="fig5">Figure 5</xref>).</title><p>(<bold>a</bold>) Light orange scatter plot: We divide putatively neutral sites into 100 equally sized bins based on the predicted effect of background selection, <inline-formula><mml:math id="inf343"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, from the best-fitting CADD-based model. For predicted values (x-axis), we average the predicted <inline-formula><mml:math id="inf344"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in each bin. For observed values (y-axis), we divide the average diversity level in each bin by the average predicted diversity level in the absence of background selection, <inline-formula><mml:math id="inf345"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, after scaling each in bin by its estimated local (relative) mutation rate (<inline-formula><mml:math id="inf346"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>u</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="equ10">Equation 8</xref>; Section 3.3). Dark orange curve: the LOESS curve for a similarly defined scatter plot but with 2000 rather than 100 bins (with span = 0.1). (<bold>b</bold>) A close-up near <inline-formula><mml:math id="inf347"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> corresponding to the boxed region in (<bold>a</bold>). Here, the LOESS curve has span = 0.033 and the scatter plot corresponds to 2000 bins (showing the top 30%).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig38-v2.tif"/></fig><p>First, we characterize the main covariates of the strength of background selection (quantified by <inline-formula><mml:math id="inf348"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>), such as recombination rate, base composition, and chromosomal position. Beyond the inherent interest in these covariates, they point toward processes that may explain the departure from our predictions near <inline-formula><mml:math id="inf349"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>. Second, we investigate whether differences in rates of different kinds of mutations and of biased gene conversion associated with the covariates of <inline-formula><mml:math id="inf350"><mml:mi>B</mml:mi></mml:math></inline-formula> can explain the departure near <inline-formula><mml:math id="inf351"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>; our analysis suggests that they cannot. Third, we argue that the residual effect of ancient introgression between archaic humans and ancestors of extant humans may contribute to this departure.</p><sec sec-type="appendix" id="s6-8-1"><title>8.1 Covariates of <inline-formula><mml:math id="inf352"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula></title><p>We expect the effects of background selection to be strongest in regions with low rates of recombination and high densities of functional sites, because neutral variation in such regions will be linked to more deleterious mutations. In line with these expectations, recombination rates increase with greater predicted <inline-formula><mml:math id="inf353"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39a</xref>); in particular, they increase sharply between the 99<sup>th</sup> and 100<sup>th</sup> percentile of predicted <inline-formula><mml:math id="inf354"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> to &gt;10 cM/Mb, which is tenfold the autosomal average. In addition, as expected, the average level of conservation around neutral sites decreases as <inline-formula><mml:math id="inf355"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> increases (<xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39b and c</xref>).</p><fig id="app1fig39" position="float"><label>Appendix 1—figure 39.</label><caption><title>Average recombination rate (<bold>a</bold>) and functional density per bp (<bold>b</bold>) and per cM (<bold>c</bold>) as a function of predicted <inline-formula><mml:math id="inf356"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>The predicted <inline-formula><mml:math id="inf357"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, binning of putatively neutral sites and LOESS fitting are as described in <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38a</xref>. We calculate the average recombination rate in each bin based on the African-American admixture map from <xref ref-type="bibr" rid="bib49">Hinch et al., 2011</xref> (Section 2.3). We measure functional density by calculating the mean phastCons score (based on the 99-vertebrate alignment) in a radius of 50 kb (<bold>b</bold>) or 0.05 cM (<bold>c</bold>) around each putatitively neutral site, and averaging these means over sites in each bin.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig39-v2.tif"/></fig><p>Next, we consider base composition and other factors that are known to affect rates of mutation and biased gene conversion (BGC). GC content has a J-shaped dependence on predicted <inline-formula><mml:math id="inf358"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40a</xref>). The greater peak in GC content, in regions under weak background selection (<inline-formula><mml:math id="inf359"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> near 1), is plausibly largely driven by the long-term effects of BGC due to higher rates of recombination in these regions (<xref ref-type="bibr" rid="bib30">Duret and Galtier, 2009</xref>; <xref ref-type="bibr" rid="bib65">Li et al., 2019</xref>; <xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39a</xref>). Both peaks (for low and high <inline-formula><mml:math id="inf360"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>) are associated with an increase in the proportion of GC sites in CpG islands but this proportion is small throughout (&lt;1%) (<xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40b</xref>), suggesting that it has little effect on GC content and on mutation rates. In contrast, methylated CpG content increases sharply as predicted <inline-formula><mml:math id="inf361"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> approaches 1 (<xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40c</xref>), suggesting a corresponding increase in the rate of C&gt;T transitions. The proportion of sites in C&gt;G hypermutable regions also increases with predicted <inline-formula><mml:math id="inf362"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40d</xref>).</p><fig id="app1fig40" position="float"><label>Appendix 1—figure 40.</label><caption><title>GC content (<bold>a</bold>), CpG sites in CpG islands (<bold>b</bold>), proportion of methylated CpGs in neutral sites (<bold>c</bold>), and proportion of neutral sites in C&gt;G hypermutable regions (<bold>d</bold>) as a function of predicted <inline-formula><mml:math id="inf363"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>Proportions and other quantities are measured for putatively neutral sites, whose type (i.e. GC and CpG) is defined based on the inferred state in the human-chimpanzee ancestor (Section 2.7). Data sources are detailed in Section 2.8. The predicted <inline-formula><mml:math id="inf364"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, binning of putatively neutral sites and LOESS fitting are as described for <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38a</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig40-v2.tif"/></fig><p>Lastly, predicted <inline-formula><mml:math id="inf365"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> is associated with chromosomal position, with regions under weak background selection (<inline-formula><mml:math id="inf366"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> near 1) clustered near telomeres (<xref ref-type="fig" rid="app1fig41">Appendix 1—figure 41</xref>). In turn, regions near telomeres are early replicating, and replication timing is known to affect mutational patterns (<xref ref-type="bibr" rid="bib103">Stamatoyannopoulos et al., 2009</xref>).</p><fig id="app1fig41" position="float"><label>Appendix 1—figure 41.</label><caption><title>The relationship between chromosomal position and predicted <inline-formula><mml:math id="inf367"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>(<bold>a</bold>) The distance to telomeres is measured on the same chromosome (see Section 2.8). (<bold>b</bold>) The distribution of putatively neutral sites by relative chromosomal position, defined as the ratio of a site’s distance to the centromere and the distance between centromere and telomeres on that chromosomal arm (see Section 2.8); relative distances corresponding to the shorter chromosome arm are shown on the left (in [–1, 0]) and to the longer arm on the right (in [0, 1]). The binning of putatively neutral sites by predicted <inline-formula><mml:math id="inf368"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> is as described for <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38a</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig41-v2.tif"/></fig></sec><sec sec-type="appendix" id="s6-8-2"><title>8.2 Mutational spectrum and biased gene conversion</title><p>As we noted, the covariates of <inline-formula><mml:math id="inf369"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> are associated with mutational processes and with biased gene conversion that affect diversity and divergence levels (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42</xref>). Specifically, we see the footprints of the following processes:</p><list list-type="bullet"><list-item><p>Increased rates of <bold>A&gt;C/T&gt;G</bold> and <bold>A&gt;G/T&gt;C</bold> substitutions near <inline-formula><mml:math id="inf370"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42a</xref>), due to higher rates of biased gene conversion that are associated with the higher rates of recombination (<xref ref-type="bibr" rid="bib30">Duret and Galtier, 2009</xref>; <xref ref-type="bibr" rid="bib65">Li et al., 2019</xref>; <xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39a</xref>). Biased gene conversion also reduces the rates of <bold>C&gt;A/G&gt;T</bold> and <bold>C&gt;T/G&gt;A</bold> substitutions, but this is not clearly visible (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42b</xref>), presumably because of other processes affecting these substitutions (see below).</p></list-item><list-item><p>Increased rate of <bold>C&gt;T mutations</bold> near <inline-formula><mml:math id="inf371"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42b</xref>), associated with the higher methylated CpG content (<xref ref-type="bibr" rid="bib5">Barrett et al., 2013</xref>; <xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40c</xref>).</p></list-item><list-item><p>Reduced rates of <bold>C&gt;A/G&gt;T</bold> and <bold>A&gt;T/T&gt;A</bold> mutations near <inline-formula><mml:math id="inf372"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42a and b</xref>), associated with improved repair of these types of mutations near origins of replication, which tend to be near telomeres (<xref ref-type="bibr" rid="bib103">Stamatoyannopoulos et al., 2009</xref>; <xref ref-type="fig" rid="app1fig41">Appendix 1—figure 41</xref>).</p></list-item><list-item><p>Increased rates of <bold>C&gt;G/G&gt;C</bold> mutations near <inline-formula><mml:math id="inf373"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42b</xref>), due to the enrichment of C&gt;G hypermutable regions (<xref ref-type="bibr" rid="bib55">Jónsson et al., 2017</xref>; <xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40d</xref>).</p></list-item></list><fig id="app1fig42" position="float"><label>Appendix 1—figure 42.</label><caption><title>Rates of different types of substitutions as a function of predicted <inline-formula><mml:math id="inf374"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>We bin putatively neutral sites by predicted <inline-formula><mml:math id="inf375"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> as described in <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38a</xref>. We calculate the rate of X&gt;Y substitutions in a bin by dividing the estimate of the number of X&gt;Y substitutions at its sites in an 8-primate phylogeny by the estimated number of its sites with state X in the ancestor of that phylogeny (see Section 3.3). We obtain the relative rate by dividing the rate in a bin by the average rate across bins. We show the rates of substitutions with ancestral state AT in (<bold>a</bold>) and GC in (<bold>b</bold>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig42-v2.tif"/></fig><p>Consequently, the rates of substitutions between any two bases covary with predicted <inline-formula><mml:math id="inf376"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42</xref>). However, the different types of substitutions have markedly different contributions to levels of diversity and divergence (<xref ref-type="fig" rid="app1fig43">Appendix 1—figure 43</xref>), and different dependencies on predicted <inline-formula><mml:math id="inf377"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42</xref>).</p><p>To investigate whether these processes can explain the departure from our predictions near <inline-formula><mml:math id="inf378"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, we break up the observed diversity levels by types of substitutions (<xref ref-type="fig" rid="app1fig44">Appendix 1—figure 44</xref>). We reason that if all types behave similarly near <inline-formula><mml:math id="inf379"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, the differential processes affecting them cannot explain the departure from predictions (at least not fully). Note that, up to a multiplicative constant, our observations are ratios of diversity levels and substitution rates (on the 8-primate phylogeny), implying that the signatures of processes that affect diversity and divergence similarly should cancel out. Conversely, for a process to affect our observations it must have noticeably different effects on diversity and divergence. We find that the observations associated with different types of substitutions largely align with each other and with the observations that include all types jointly (<xref ref-type="fig" rid="app1fig38">Appendix 1—figures 38</xref> and <xref ref-type="fig" rid="app1fig44">44</xref>). Specifically, they align with predictions throughout nearly the entire range of predicted <inline-formula><mml:math id="inf380"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, and are markedly higher than predictions near <inline-formula><mml:math id="inf381"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><fig id="app1fig43" position="float"><label>Appendix 1—figure 43.</label><caption><title>Contribution of different types of substitutions to diversity levels (<bold>a</bold>) and number of substitutions per site in the 8-primate phylogeny (<bold>b</bold>) as a function of predicted <inline-formula><mml:math id="inf382"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>We bin putatively neutral sites by predicted <inline-formula><mml:math id="inf383"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> as described in <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38a</xref>. (<bold>a</bold>) We define the ancestral state, that is, AT or GC, as the inferred state in the human-chimpanzee ancestor (see Section 2.7). We define the contribution of each type of substitution to the diversity level in a bin as the ratio of the number of pairwise differences of that type and the total number of pairwise comparisons in a bin; this way, the sum over types equals the observed diversity level in a bin (<inline-formula><mml:math id="inf384"><mml:mover accent="true"><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:math></inline-formula>). (<bold>b</bold>) We calculate the number of substitutions of each type in a bin as described in <xref ref-type="fig" rid="app1fig42">Appendix 1—figure 42</xref>, but here we normalize it by the number of sites, such that the sum over types equals the observed number of substitutions per site in the 8-primate phylogeny.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig43-v2.tif"/></fig><fig id="app1fig44" position="float"><label>Appendix 1—figure 44.</label><caption><title>Observed vs. predicted neutral diversity levels for different types of substitutions.</title><p>Diversity levels are presented as in <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38</xref>, but here, we calculate diversity levels and substitution rates for sites with a given ancestral state, that is, AT (<bold>a and c</bold>) or CG (<bold>b and d</bold>), and for each of the three types of substitutions from the ancestral state, as in <xref ref-type="fig" rid="app1fig43">Appendix 1—figure 43</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig44-v2.tif"/></fig><p>Nonetheless, the observed levels associated with C&gt;G mutations near <inline-formula><mml:math id="inf385"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> are markedly higher than for other types of substitutions (<xref ref-type="fig" rid="app1fig44">Appendix 1—figure 44b and d</xref>). This effect contributes negligibly to the total observed levels near <inline-formula><mml:math id="inf386"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, however, because C&gt;G substitutions have a minor contribution to total diversity and divergence levels (<xref ref-type="fig" rid="app1fig43">Appendix 1—figure 43</xref>). The higher levels of C&gt;G substitutions are associated with the enrichment of C&gt;G hypermutable regions near <inline-formula><mml:math id="inf387"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig40">Appendix 1—figure 40d</xref>). Notably, when we remove these regions, the observed levels associated with C&gt;G substitutions are no longer higher than for other types of substitutions (<xref ref-type="fig" rid="app1fig45">Appendix 1—figure 45</xref>). Removing C&gt;G hypermutable regions also affects the magnitude of the departure from predictions for other types of substitutions (compare <xref ref-type="fig" rid="app1fig45">Appendix 1—figure 45c and d</xref> with <xref ref-type="fig" rid="app1fig44">Appendix 1—figure 44c and d</xref>), because C&gt;G is not the only type of mutation whose rate is higher in these regions. These hypermutable regions plausibly affect diversity more than divergence (and thus our observations) given that their effects are stronger in recent human evolution than in the more distant past and in the lineages of closely related species (Ipsita Agarwal and Molly Przeworski, personal communication). This may reflect the fact that these regions were identified in extant humans and/or a dependence of their effects on life history (<xref ref-type="bibr" rid="bib55">Jónsson et al., 2017</xref>; <xref ref-type="bibr" rid="bib36">Gao et al., 2019</xref>). Setting the causes aside, even when we remove these regions from the set of putatively neutral sites used in our inference, observed levels of all types are still markedly higher than the revised predictions near <inline-formula><mml:math id="inf388"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig46">Appendix 1—figure 46</xref>).</p><fig id="app1fig45" position="float"><label>Appendix 1—figure 45.</label><caption><title>Observed vs. predicted neutral diversity levels for different types of substitutions after removing C&gt;G hypermutable regions (from both the inference and observations).</title><p>Other than removing ~12% of putatively neutral sites in these regions, the details are as in <xref ref-type="fig" rid="app1fig44">Appendix 1—figure 44</xref>. We note that while a greater proportion of sites is removed from bins near <inline-formula><mml:math id="inf389"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (~20% for the 100th percentile), this in itself has a minor effect on the departures from predictions in these bins.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig45-v2.tif"/></fig><fig id="app1fig46" position="float"><label>Appendix 1—figure 46.</label><caption><title>Comparison of the best-fitting CADD-based models with and without C&gt;G hypermutable regions.</title><p>All panels are as described for the corresponding ones in <xref ref-type="fig" rid="app1fig14">Appendix 1—figure 14</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig46-v2.tif"/></fig><p>In summary, while we cannot rule out that there are other mutational processes that contribute to the departure from predictions near <inline-formula><mml:math id="inf390"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, our analysis suggests that known mutational processes and biased gene conversion fall short of explaining these departures.</p></sec><sec sec-type="appendix" id="s6-8-3"><title>8.3 A footprint of archaic introgression?</title><p>Next, we consider whether the excess diversity observed near <inline-formula><mml:math id="inf391"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> could reflect a residual signal of archaic introgression. The presence of archaic alleles at a locus increases diversity because their coalescence with modern human alleles traces back to the ancestors of modern humans and the archaic hominin from which they originated. Archaic introgression could help to explain the excess diversity in regions with <inline-formula><mml:math id="inf392"><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, if archaic alleles were more common in these regions. As we argue below, there are good reasons to believe this to be the case.</p><p>Aside from evidence for positive selection on introgressed alleles in a few cases (<xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib111">Vernot and Akey, 2014</xref>; <xref ref-type="bibr" rid="bib87">Racimo et al., 2015</xref>), the pattern of Neanderthal and Denisovan introgression in contemporary human populations appears to be dominated by purifying selection to remove archaic ancestry from the human genome, as evidenced by the depletion of archaic introgression in and around genes (<xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib111">Vernot and Akey, 2014</xref>; <xref ref-type="bibr" rid="bib44">Harris and Nielsen, 2016</xref>; <xref ref-type="bibr" rid="bib56">Juric et al., 2016</xref>; <xref ref-type="bibr" rid="bib93">Sankararaman et al., 2016</xref>). The causes for this purifying selection are still being deliberated (<xref ref-type="bibr" rid="bib44">Harris and Nielsen, 2016</xref>; <xref ref-type="bibr" rid="bib56">Juric et al., 2016</xref>; <xref ref-type="bibr" rid="bib95">Schumer et al., 2018</xref>). One hypothesis is that selection acts against introgressed alleles that are incompatible with the genetic background in modern humans, for example, alleles that are part of Dobzhansky-Muller incompatibilities between archaic hominins and modern humans (<xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib95">Schumer et al., 2018</xref>). Another hypothesis is that selection acts against alleles that were deleterious in both archaic hominins and modern humans, which were more common in archaic hominins because their long-term effective population sizes were smaller than in modern humans (<xref ref-type="bibr" rid="bib44">Harris and Nielsen, 2016</xref>; <xref ref-type="bibr" rid="bib56">Juric et al., 2016</xref>; <xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>).</p><p>Regardless of its cause, we expect purifying selection to remove archaic alleles, including neutral variants, more rapidly in genomic regions under stronger background selection (<xref ref-type="bibr" rid="bib44">Harris and Nielsen, 2016</xref>; <xref ref-type="bibr" rid="bib56">Juric et al., 2016</xref>; <xref ref-type="bibr" rid="bib95">Schumer et al., 2018</xref>). This is because these regions harbor more selected sites (<xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39b</xref>) in which archaic alleles could be selected against, and because they have lower rates of recombination (<xref ref-type="fig" rid="app1fig39">Appendix 1—figure 39a</xref>) causing selection against archaic alleles to remove larger archaic segments. Conversely, we expect the highest, residual proportion of archaic neutral variants in regions with <inline-formula><mml:math id="inf393"><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>—precisely where we observe a 10–15% excess of diversity above our predictions (<xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38</xref>).</p><p>In order to test this expectation, we use fine-scale maps of archaic introgression inferred for European (CEU) and East-Asian (CHB/CHS) individuals from the 1000 Genomes Project (<xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>). These maps assign a probability of Neanderthal ancestry to contiguous 500 bp segments tiling individual genomes based on the high-coverage Altai Neanderthal genome (<xref ref-type="bibr" rid="bib85">Prüfer et al., 2014</xref>). We use them to estimate the average proportion of archaic ancestry per putatively neutral site in bins of predicted <inline-formula><mml:math id="inf394"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. As expected, we find that the estimated proportion of archaic alleles increases with predicted <inline-formula><mml:math id="inf395"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig47">Appendix 1—figure 47</xref>). The power to identify introgressed segments using this and other methods decreases substantially in regions with <inline-formula><mml:math id="inf396"><mml:mi>B</mml:mi><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, because higher recombination rate in these regions results in much shorter archaic segments (<xref ref-type="bibr" rid="bib101">Skov et al., 2018</xref>; <xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>). We therefore expect that the actual proportion of archaic ancestry increases more sharply near <inline-formula><mml:math id="inf397"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> than our analysis suggests, and may therefore better trace the sharp increase in diversity relative to predictions near <inline-formula><mml:math id="inf398"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><fig id="app1fig47" position="float"><label>Appendix 1—figure 47.</label><caption><title>Estimated proportion of Neanderthal (NE) ancestry as a function of predicted <inline-formula><mml:math id="inf399"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in Europeans (CEU) (<bold>a</bold>) and East-Asians (CHB/CHS) (<bold>b</bold>).</title><p>See text for the estimation procedure. The bins and LOESS curves were calculated as in <xref ref-type="fig" rid="app1fig38">Appendix 1—figure 38</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig47-v2.tif"/></fig><p>Current inferences about archaic introgression are divided into those that incorporate sequenced Neanderthal and Denisovan genomes (<xref ref-type="bibr" rid="bib42">Green et al., 2010</xref>; <xref ref-type="bibr" rid="bib89">Reich et al., 2010</xref>; <xref ref-type="bibr" rid="bib92">Sankararaman et al., 2014</xref>; <xref ref-type="bibr" rid="bib111">Vernot and Akey, 2014</xref>; <xref ref-type="bibr" rid="bib104">Steinrücken et al., 2018</xref>), such as the maps we used in <xref ref-type="fig" rid="app1fig47">Appendix 1—figure 47</xref>, and those that are based only on patterns of variation in contemporary humans (<xref ref-type="bibr" rid="bib81">Plagnol and Wall, 2006</xref>; <xref ref-type="bibr" rid="bib115">Wall et al., 2009</xref>; <xref ref-type="bibr" rid="bib101">Skov et al., 2018</xref>; <xref ref-type="bibr" rid="bib31">Durvasula and Sankararaman, 2020</xref>). When we repeat the analysis in <xref ref-type="fig" rid="app1fig47">Appendix 1—figure 47</xref> using ancestry-maps based on the latter approach in both Africans and non-Africans (<xref ref-type="bibr" rid="bib101">Skov et al., 2018</xref>; <xref ref-type="bibr" rid="bib31">Durvasula and Sankararaman, 2020</xref>), we find that levels of archaic ancestry either increase and level off at intermediate values of <inline-formula><mml:math id="inf400"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, or peak at intermediate values and decrease as <inline-formula><mml:math id="inf401"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> approaches 1. We believe that this departure from our expectation reflects a decrease in the power of these methods near <inline-formula><mml:math id="inf402"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> = 1 (due to higher rates of recombination), which is greater than the decrease for methods based on sequenced archaic genomes. An additional caveat is that the evidence for the contribution of archaic introgression to the African gene pool is based solely on patterns in contemporary genetic variation (<xref ref-type="bibr" rid="bib81">Plagnol and Wall, 2006</xref>; <xref ref-type="bibr" rid="bib114">Wall and Hammer, 2006</xref>; <xref ref-type="bibr" rid="bib115">Wall et al., 2009</xref>; <xref ref-type="bibr" rid="bib31">Durvasula and Sankararaman, 2020</xref>) and remain more speculative in lieu of more direct evidence. Thus, while it seems plausible that the greater retention of neutral archaic variants in regions with the highest ~2% of predicted <inline-formula><mml:math id="inf403"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> values contributes substantially to the departure from our predictions in both African and non-African populations, at present, the evidence for such a contribution remains equivocal.</p></sec></sec><sec sec-type="appendix" id="s6-9"><title>9. Additional figures</title><fig id="app1fig48" position="float"><label>Appendix 1—figure 48.</label><caption><title>Overfitting has negligible effects on our results (see also Section 6.1).</title><p>As an illustration, we compare the results corresponding to our best-fitting CADD-based model, using all the data jointly (orange), out-of-sample predictions in non-overlapping, contiguous, 2 Mb windows (light blue) and out-of-sample predictions for each autosome (pink). (<bold>a</bold>) Predicted and observed diversity levels along chromosome 1 in the YRI sample; (details as in <xref ref-type="fig" rid="fig2">Figure 2A</xref> in Main Text). (<bold>b</bold>) The proportion of variance in YRI diversity levels explained by background selection at different spatial scales (details as in <xref ref-type="fig" rid="fig2">Figure 2B</xref> in Main Text). (<bold>c</bold>) Predicted and observed neutral diversity levels around human-specific nonsynonymous (NS) substitutions (details as in <xref ref-type="fig" rid="fig3">Figure 3</xref> in Main Text). (<bold>d</bold>) Observed vs. predicted neutral diversity levels across the autosomes (details as in <xref ref-type="fig" rid="fig5">Figure 5</xref> in Main Text). As an example, the explained variance in diversity levels over 1 Mb windows is 59.9% without excluding data, 59.8% when we exclude 2 Mb windows, and 59.5% when we leave out one chromosome at a time. Our statistical analysis in Section 6.2 suggests that these minute difference are not significant. Moreover, the reduction in explained variance in the case in which we exclude one autosome at a time is plausibly due to the reduced amount of data rather than overfitting (e.g. chromosomes 1 and 2 correspond to 8% and 9.3% of our putatively neutral sites, respectively).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig48-v2.tif"/></fig><fig id="app1fig49" position="float"><label>Appendix 1—figure 49.</label><caption><title>A background selection model predicts neutral diversity levels around different genomic features.</title><p>Here we use our best-fitting CADD-based model and show diversity levels around: (<bold>a</bold>) human-specific synonymous substitutions; (<bold>b</bold>) human-specific substitutions in conserved regions; (<bold>c</bold>) exons; and (<bold>d</bold>) conserved exonic regions. The inference of human-specific substitutions is described in Section 2.7. Conserved regions are based on autosomal sites with the top 6% phastCons scores in the 99-vertebrate alignment (Section 4.1). The set of exons is described in Section 2.4. The genetic distance to the nearest element (e.g. exon) is measured to its closest edge. Other details are similar to <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig49-v2.tif"/></fig><fig id="app1fig50" position="float"><label>Appendix 1—figure 50.</label><caption><title>Predicted and observed neutral diversity levels along chromosome 1 based on data from representative continental populations.</title><p>Plots are generated as detailed in <xref ref-type="fig" rid="fig2">Figure 2A</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig50-v2.tif"/></fig><fig id="app1fig51" position="float"><label>Appendix 1—figure 51.</label><caption><title>A background selection model predicts neutral diversity levels around human-specific nonsynonymous (NS) substitutions in representative continental populations.</title><p>Plots are constructed as detailed in <xref ref-type="fig" rid="fig3">Figure 3</xref>, using polymorphism data from each population for both inferences and observations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig51-v2.tif"/></fig><fig id="app1fig52" position="float"><label>Appendix 1—figure 52.</label><caption><title>Observed vs. predicted neutral diversity levels across the autosomes in representative continental populations.</title><p>Plots are constructed as detailed in <xref ref-type="fig" rid="fig5">Figure 5</xref>, using polymorphism data from each population for both inferences and observations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-app1-fig52-v2.tif"/></fig></sec></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.76065.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Nordborg</surname><given-names>Magnus</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Gregor Mendel Institute</institution><country>Austria</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/2021.07.02.450762" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/2021.07.02.450762"/></front-stub><body><p>This paper uses state-of-the-art methods and the latest data to answer the question of whether variation in polymorphism levels along the human genome is mostly driven by linked purifying selection or selective sweeps. It makes a very strong case for the former. The paper is exceptionally well written and should be of interest to anyone wishing to understand patterns of polymorphism.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.76065.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Nordborg</surname><given-names>Magnus</given-names></name><role>Reviewing Editor</role><aff><institution>Gregor Mendel Institute</institution><country>Austria</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Nordborg</surname><given-names>Magnus</given-names></name><role>Reviewer</role><aff><institution>Gregor Mendel Institute</institution><country>Austria</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2021.07.02.450762">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2021.07.02.450762v2">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Broad-scale variation in human genetic diversity levels is predicted by purifying selection on coding and non-coding elements&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, including Magnus Nordborg as Reviewing Editor and Reviewer #1, and the evaluation has been overseen by Detlef Weigel as the Senior Editor.</p><p>The reviewers are unanimous in considering this an exceptionally clear and well-written paper. If published as-is, it would be better than most published papers. Respect!</p><p>Nonetheless, it can be improved, like all papers, and the reviews attached provide lots of suggestions along these lines. In addition, there are two things we would like to see done, and that we believe should not be too onerous.</p><p>Essential Revisions (both points are well-described by Reviewer 3 [indeed] and related to model fitting):</p><p>1) A simple neutral simulation to check the extent to which the composite likelihood model can over-fit to neutral variation.</p><p>2) An out-of-sample prediction, e.g. by leaving chromosomes out.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>This is an exceptionally well-written paper, and I have no major suggestions for improvement.</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>I am reasonably happy with the paper in its current form – I can see the incredible care the authors took to make this study robust. I would recommend some forward simulations using SLiM, but I will not require them (as I said, mostly since this itself would be a sizable task to do). However, I would like to see some simple neutral simulations with realistic recombination maps as a negative control as mentioned. This may seem like an unnecessary negative control – however, my main concern here is that I have heard others critique the Eylashiv et al. model, claiming it could be fitting neutral &quot;noise&quot; along the chromosome. I do not think this is the case, as I think the functional form doesn't have that many degrees of freedom, but I would still like to see this concern addressed.</p><p>My primary concern is that the out-sample prediction comparison analysis is conducted (beyond the current leave one window out approach) and included in the main text. I think this would strengthen the claim that BGS models alone closely match diversity data (and the present results are not just a consequence of overfitting). I don't think these out-sample models need to include the positive selection model since it has little effect.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.76065.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential Revisions (both points are well-described by Reviewer 3 [indeed] and related to model fitting):</p><p>1) A simple neutral simulation to check the extent to which the composite likelihood model can over-fit to neutral variation.</p></disp-quote><p>We do not expect over-fitting to be an issue, for the reasons detailed in our reply to Reviewer 3. Nonetheless, we ran simulations detailed below, and confirmed that assumption: our inference explains almost no variation in diversity levels and infers approximately zero effects of background selection when applied to data simulated under neutrality.</p><p>Specifically, we performed the inference on simulated data as follows:</p><p>– First, we used <italic>msprime</italic> (Baumdicker et al., 2022) to simulate a neutral diversity dataset that mimics the data we used in our inferences. Specifically, we simulated diversity on all autosomes in a sample size of 216 chromosomes (akin to the sample from 108 individuals in YRI we used), assuming <italic>N<sub>e</sub></italic> of 20,000, the genetic map from Hinch et al. (2011), and a mutation rate of 1.25·10<sup>-8</sup> per bp per generation.</p><p>– Second, we ran the inference corresponding to our best fitting CADD model, i.e., using the 6% of sites with the highest CADD scores as targets of selection.</p><p>We describe the results in the same terms that we used in the manuscript.</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-sa2-fig1-v2.tif"/></fig><p><xref ref-type="fig" rid="sa2fig1">Author response image 1</xref> is analogous to Figure 2. As expected, we infer practically no effects of background selection, and predicted diversity levels along chromosome 1 are flat (A: orange curve). The explained variance in diversity levels is indistinguishable from 0 over different spatial scales (B: orange circles), where for comparison, we also included the corresponding results based on the real data. For example, on the 1 Mb scale, the predicted map explains ~0.2% of the variance compared to ~60% for the map based on the same model with real data.</p><fig id="sa2fig2" position="float"><label>Author response image 2.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-sa2-fig2-v2.tif"/></fig><p><xref ref-type="fig" rid="sa2fig2">Author response image 2</xref> are analogous to Figures 3 and 5. As expected, diversity levels around nonsynonymous substitutions are nearly flat (on the left); the tiny dip at the center arises inferring a tiny rather than a strictly 0 rate of deleterious mutations at selected regions. The difference between this deleterious mutation rate and 0 is plausibly not statistically significant: ~2.7·10<sup>-10</sup> per bp per generation at selected regions compared to ~10<sup>-8</sup> for the same model with real data. The relationship between observed and predicted diversity levels (on the right) reflects noise within a narrow range of (0.985, 1.0) compared to calibrated predications in a range of (0.6, 1.0) for the real data (see Figure 5 in the paper).</p><disp-quote content-type="editor-comment"><p>2) An out-of-sample prediction, e.g. by leaving chromosomes out.</p></disp-quote><p>Following this suggestion, we replaced all the results of the main text for our best-fitting background selection models with maps based on out-of-sample prediction. Specifically, we tile autosomes with non-overlapping 2 Mb windows, where the predicted diversity level in a window is inferred while excluding the data from that window. The 2 Mb scale is substantially greater than the scale of LD blocks in human populations, so any correlation among tMRCAs at the edges of windows should have at most minute effects (see, e.g., Wall and Pritchard NRG 2003). We also include the main analyses based on leaving one chromosome out. The results of these analyses are shown together in the newly added Figure A48. With either approach, the results are indistinguishable from the results we obtained beforehand, as expected given the few parameters that we infer relative to the enormous amount of data (see reply to reviewer 3 on p. 10-11). As an example, the explained variance in diversity levels over 1 Mb windows using our best-fitting CADD-based model is 59.9% without excluding data; 59.8% when we exclude 2 Mb windows; and 59.5% when we leave out one chromosome at a time. Our statistical analysis in Appendix 6 Section 6.2 suggests that these tiny differences are not statistically significant. Moreover, the reduction in explained variance when one chromosome is excluded at a time is plausibly due to the reduction in the amount of data used rather than overfitting (e.g., chromosomes 1 and 2 correspond to 8% and 9.3% of our putatively neutral sites, respectively).</p><p>The results of our inference corresponding to leaving out one chromosome at a time are presented as <xref ref-type="fig" rid="sa2fig3">Author response image 3</xref>, using analogous figures to those in the main text:</p><fig id="sa2fig3" position="float"><label>Author response image 3.</label><caption><title>Fig. 1 equivalent: orange curve (<bold>A</bold>) and points (<bold>B</bold>) correspond to predictions in which data from the focal chromosome was excluded.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-sa2-fig3-v2.tif"/></fig><fig id="sa2fig4" position="float"><label>Author response image 4.</label><caption><title>Figures 3 and 5 equivalents.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-76065-sa2-fig4-v2.tif"/></fig><p>We hope that these additional analyses alleviate any concerns about overfitting.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>I am reasonably happy with the paper in its current form – I can see the incredible care the authors took to make this study robust. I would recommend some forward simulations using SLiM, but I will not require them (as I said, mostly since this itself would be a sizable task to do). However, I would like to see some simple neutral simulations with realistic recombination maps as a negative control as mentioned. This may seem like an unnecessary negative control – however, my main concern here is that I have heard others critique the Eylashiv et al. model, claiming it could be fitting neutral &quot;noise&quot; along the chromosome. I do not think this is the case, as I think the functional form doesn't have that many degrees of freedom, but I would still like to see this concern addressed.</p><p>My primary concern is that the out-sample prediction comparison analysis is conducted (beyond the current leave one window out approach) and included in the main text. I think this would strengthen the claim that BGS models alone closely match diversity data (and the present results are not just a consequence of overfitting). I don't think these out-sample models need to include the positive selection model since it has little effect.</p></disp-quote><p>The results of the requested neutral simulations and leave-one-out analyses are described above, in the response to essential revisions (pp. 2-3). In brief, and as anticipated by the reviewer, there is no overfitting of neutral diversity patterns or otherwise.</p><p>We note further that the concerns regarding both Elyashiv et al., (2016) and the current manuscript, which we were not aware of previously, are puzzling, for a number of reasons:</p><p>– First, we are fitting models with very few parameters, e.g., seven parameters in our best fitting CADD and phastCons based models, to highly variable diversity levels along the genome (see, e.g., Figure 2A) and relying on an enormous amount of data (diversity levels estimated from a sample of ~108 individuals at ~653M putatively neutral autosomal sites, spread over ~2600 LD blocks; Berisa and Pickrell, 2016). Based on first principles alone then, it seems hard to imagine how overfitting could account for substantial variance in diversity levels (e.g., ~60% of the variance in diversity levels in 2580 widows of 1 Mb along human autosomes).</p><p>– Regardless of the recombination landscape, given a uniform mutation rate, the expected heterozygosity under a neutral model should be constant throughout autosomes. Consequently, our method should infer no linked selection effects—which is indeed what we see in the neutral simulations above.</p><p>– Lastly, the models we fit are based on functional forms specific to the effects of linked selection, and they are fit based on the genetic distance to putative targets of selection. There is no reason to assume that these functional forms (alongside the spatial organization of selected regions) would capture neutral effects, and many reasons to suppose otherwise. If anything, it is remarkable that simple models of linked selection, based on rough annotations of selected regions and assuming mutation and selection parameters that are fixed within each annotation, should fit the data as well as they do.</p><p>Thus, based on first principle considerations, there is no reason to expect overfitting to be an issue here, and indeed simulations confirm that there is no overfitting. Hopefully, these additions will put the concern to rest.</p></body></sub-article></article>