<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">58931</article-id><article-id pub-id-type="doi">10.7554/eLife.58931</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Recurrent evolution of high virulence in isolated populations of a DNA virus</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-113261"><name><surname>Hill</surname><given-names>Tom</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4661-6391</contrib-id><xref ref-type="aff" rid="aff1"/><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-108407"><name><surname>Unckless</surname><given-names>Robert L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-8586-7137</contrib-id><email>unckless@ku.edu</email><xref ref-type="aff" rid="aff1"/><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><institution>The Department of Molecular Biosciences, University of Kansas</institution><addr-line><named-content content-type="city">Lawrence</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="senior_editor"><name><surname>Tautz</surname><given-names>Diethard</given-names></name><role>Senior Editor</role><aff><institution>Max-Planck Institute for Evolutionary Biology</institution><country>Germany</country></aff></contrib><contrib contrib-type="editor"><name><surname>Ebert</surname><given-names>Dieter</given-names></name><role>Reviewing Editor</role><aff><institution>University of Basel</institution><country>Switzerland</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>28</day><month>10</month><year>2020</year></pub-date><pub-date pub-type="collection"><year>2020</year></pub-date><volume>9</volume><elocation-id>e58931</elocation-id><history><date date-type="received" iso-8601-date="2020-05-14"><day>14</day><month>05</month><year>2020</year></date><date date-type="accepted" iso-8601-date="2020-10-28"><day>28</day><month>10</month><year>2020</year></date></history><permissions><copyright-statement>© 2020, Hill and Unckless</copyright-statement><copyright-year>2020</copyright-year><copyright-holder>Hill and Unckless</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-58931-v2.pdf"/><abstract><p>Hosts and viruses are constantly evolving in response to each other: as a host attempts to suppress a virus, the virus attempts to evade and suppress the host’s immune system. Here, we describe the recurrent evolution of a virulent strain of a DNA virus, which infects multiple Drosophila species. Specifically, we identified two distinct viral types that differ 100-fold in viral titer in infected individuals, with similar differences observed in multiple species. Our analysis suggests that one of the viral types recurrently evolved at least four times in the past ~30,000 years, three times in Arizona and once in another geographically distinct species. This recurrent evolution may be facilitated by an effective mutation rate which increases as each prior mutation increases viral titer and effective population size. The higher titer viral type suppresses the host-immune system and an increased virulence compared to the low viral titer type.</p></abstract><abstract abstract-type="executive-summary"><title>eLife digest</title><p>Animals constantly evolve to protect themselves against viruses, and in turn, viruses evolve to escape their host’s new defenses. As a result, genes involved in this arms’ race are some of the fastest evolving in nature. A better understanding of how host-virus evolution works could help in the search for treatments for many human and animal diseases.</p><p>Repetition is one of the gold standard requirements for biological experiments. Watching different groups of animals and viruses evolve under the same conditions makes it possible for researchers to work out whether certain changes are more likely than others. This is easy to do in the laboratory, where conditions can be controlled, but much more complicated to accomplish in the wild. Wild populations are rarely completely isolated, and often face different environmental conditions. One animal-virus pair for which this is not the case is made up of the fly <italic>Drosophila innubila</italic>, and its virus <italic>Drosophila innubila nudivirus</italic>. They live in the 'sky islands' of North America, patches of forests surrounded by hundreds of kilometers of desert. These islands are like natural test tubes, isolated ecosystems each with its own separate fly and virus populations and limited gene flow between populations.</p><p>To understand how this virus-host pair evolves, Hill and Unckless sequenced the genomes of flies and viruses from four different populations. While the fly genomes did not show evidence of strong differences between populations, the virus genomes did. There were two distinct types of virus, one of which was a lot more effective than the other at infecting flies, possibly because it was better at blocking the fly's immune defenses. Unexpectedly, this virus type had evolved more than once, emerging separately on at least four different occasions. Hill and Unckless suggest that the natural interactions between flies with similar genomes and the virus guide evolution down the same path time and time again.</p><p>This work on wild populations contributes to the understanding of the evolution of viruses and their hosts. One question left unanswered is why both types of virus (one more effective at infecting the flies and the other less so) persist in each population when one is better at blocking the fly's immune response? Future work using isolated populations like these could shed more light on the pressures that shape the evolution of viruses and their hosts, potentially helping in the study of human viruses, like HIV.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd><italic>Drosophila innubila</italic></kwd><kwd>DNA virus</kwd><kwd>co-evolution</kwd><kwd>immunity</kwd><kwd>genomics</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Other</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution>KU CMADP</institution></institution-wrap></funding-source><award-id>P20 GM103638</award-id><principal-award-recipient><name><surname>Hill</surname><given-names>Tom</given-names></name><name><surname>Unckless</surname><given-names>Robert L</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution>K-INBRE</institution></institution-wrap></funding-source><award-id>P20 GM103418</award-id><principal-award-recipient><name><surname>Hill</surname><given-names>Tom</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R00 GM114714</award-id><principal-award-recipient><name><surname>Unckless</surname><given-names>Robert L</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01 AI139154</award-id><principal-award-recipient><name><surname>Unckless</surname><given-names>Robert L</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>DEB-1737824</award-id><principal-award-recipient><name><surname>Hill</surname><given-names>Tom</given-names></name><name><surname>Unckless</surname><given-names>Robert L</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The same host–virus interactions can evolve multiple times in nature, due to the high effective mutation rate of viruses, and provide interesting systems of study.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Antagonistic coevolution between hosts and their parasites is nearly ubiquitous across the diversity of life (<xref ref-type="bibr" rid="bib11">Burt and Trivers, 2006</xref>). As a result, genes involved in immune defense are among the fastest evolving genes in host genomes (<xref ref-type="bibr" rid="bib66">Nielsen et al., 2005</xref>; <xref ref-type="bibr" rid="bib86">Sackton et al., 2007</xref>; <xref ref-type="bibr" rid="bib27">Enard et al., 2016</xref>; <xref ref-type="bibr" rid="bib91">Shultz and Sackton, 2019</xref>). Viruses are a particular fitness burden on hosts; for viruses to persist within populations, they must successfully invade the host organism, contend with the host-immune system, replicate and then transmit the newly produced particles to a new host (<xref ref-type="bibr" rid="bib40">Holmes, 2007</xref>; <xref ref-type="bibr" rid="bib32">Gifford, 2012</xref>). Once successfully established in a population, natural selection acts to modulate the rate the virus propagates relative to its virulence, optimizing the ratio of virulence to transmission (<xref ref-type="bibr" rid="bib104">Williams and Nesse, 1991</xref>; <xref ref-type="bibr" rid="bib60">May and Nowak, 1995</xref>; <xref ref-type="bibr" rid="bib53">Lipsitch et al., 1996</xref>). Due in part to their elevated mutation rate and large population sizes, viruses can accomplish this, and often co-opt or manipulate host-pathways in the process (<xref ref-type="bibr" rid="bib10">Burgyán and Havelda, 2011</xref>; <xref ref-type="bibr" rid="bib20">Davey et al., 2011</xref>; <xref ref-type="bibr" rid="bib70">Palmer et al., 2019</xref>).</p><p>Given the strong selective pressures viruses exert on their hosts (and hosts on viruses), it would be useful to assess how the two players behave in replicate evolutionary experiments. Experiments examining the evolution of pathogens and hosts are common in the lab (<xref ref-type="bibr" rid="bib76">Perron et al., 2006</xref>; <xref ref-type="bibr" rid="bib72">Paterson et al., 2010</xref>; <xref ref-type="bibr" rid="bib8">Bull et al., 2011</xref>; <xref ref-type="bibr" rid="bib58">Martins et al., 2014</xref>; <xref ref-type="bibr" rid="bib88">Scanlan et al., 2015</xref>) and even in patients (<xref ref-type="bibr" rid="bib74">Pennings, 2012</xref>; <xref ref-type="bibr" rid="bib75">Pennings et al., 2014</xref>; <xref ref-type="bibr" rid="bib28">Feder et al., 2019</xref>), but difficult in natural settings (<xref ref-type="bibr" rid="bib96">Souza et al., 2002</xref>; <xref ref-type="bibr" rid="bib34">Grubaugh et al., 2015</xref>), so we are rarely able to observe the coevolutionary dynamics between a virus and its host in replicate populations to examine natural coevolution and identify if evolution repeats itself. These studies are necessary as they frequently help characterize the pathways viruses have evolved to escape or suppress the host-immune system, and whether the same host pathways are common targets of the same virus (<xref ref-type="bibr" rid="bib85">Sabin et al., 2010</xref>).</p><p>Several studies of viruses in natural populations highlight that the same initial virus can follow the same adaptive path in multiple replicate samples. This results in parallel or convergent evolution of a virus better optimized to the host, possible due to the relatively high viral mutation rate and viral population size (<xref ref-type="bibr" rid="bib7">Bull et al., 1997</xref>; <xref ref-type="bibr" rid="bib12">Casino et al., 1999</xref>; <xref ref-type="bibr" rid="bib17">Crandall et al., 1999</xref>; <xref ref-type="bibr" rid="bib100">van Mierlo et al., 2012</xref>). However, in other systems, the same initial virus can evolve divergently in a host or location-dependent manner. Specifically, virulence and the diversity of mutations can differ dramatically based on differences in the host environment, the extremely strong selective pressures acting on the virus, and the unpredictability of some viruses due to their elevated mutation rate (<xref ref-type="bibr" rid="bib57">Martinez-Picado et al., 2002</xref>; <xref ref-type="bibr" rid="bib18">Cuevas et al., 2003</xref>; <xref ref-type="bibr" rid="bib81">Real et al., 2005</xref>; <xref ref-type="bibr" rid="bib34">Grubaugh et al., 2015</xref>). Characterizing how viruses adapt to their long-term hosts in naturally structured populations will help broaden and expand our understanding of how hosts and pathogens evolve in response to each other, and how repeatable evolution is in the face of minor environmental differences.</p><p>We took advantage of a natural host/DNA virus infection model involving populations separated by hundreds of kilometers to study how viruses evolve with their hosts in replicate populations with limited gene flow. <italic>Drosophila innubila</italic> is a mycophagous species that lives in montane forests in the ‘Sky islands’ in southwestern North America. They are commonly infected with a double-stranded DNA virus, the <italic>Drosophila</italic> innubila Nudivirus (DiNV) (<xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). A previous study examined the rates of evolution of DiNV (<xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>), finding the envelope and replication machinery to be rapidly evolving in DiNV, suggesting its importance in viral propagation (<xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). Additionally, a viral suppressor of Toll was identified in the genome of a closely related virus and DiNV, suggesting that components of the Toll pathway (perhaps antimicrobial peptides – AMPs) interact with the virus, which these viruses have then evolved to suppress (<xref ref-type="bibr" rid="bib70">Palmer et al., 2019</xref>). Consistent with this, genes in the Toll pathway and Toll-regulated AMPs are rapidly evolving in <italic>D. innubila</italic> (<xref ref-type="bibr" rid="bib36">Hill et al., 2019</xref>). A more thorough examination of the natural evolution of both the host and virus is necessary to understand how DiNV and <italic>D. innubila</italic> interact with each other beyond this suppressor protein.</p><p>We surveyed natural genetic variation in DiNV in four replicate populations to infer its co-evolutionary history with <italic>D. innubila.</italic> To this end, we (a) estimated how long DiNV has infected <italic>D. innubila</italic>, (b) inferred the evolutionary history of the host and virus among the populations, including signatures of selection and recombination (c) examined host and virus genetic variation associated with viral titer within individual hosts, and (d) characterized gene expression differences associated with that genetic variation. We identified two viral multilocus genotypes that differ by 11 focal SNPs and found that these viral types are maintained within the same host population and across multiple isolated host populations. These SNPs are tightly linked, likely brought together by a combination of recurrent mutation and recombination, and this linkage appears to be maintained by strong selection. One viral type is associated with 100-fold higher viral titer and increased virulence compared to the other. Further, we found evidence that the SNPs associated with the high titer type have evolved independently in at least three geographically-isolated viral populations infecting the sympatric <italic>D. innubila</italic> and <italic>D. azteca</italic>, and a geographically separate population of viruses infecting <italic>D. falleni</italic>. Together, these results suggest rapid evolutionary dynamics of host–virus interactions, due to recurrent evolution of a highly virulent haplotype that interacts with multiple host pathways.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>DiNV segregates for linked variants strongly associated with viral titer</title><p>To characterize the evolutionary dynamics of wild <italic>Drosophila</italic> innubila Nudivirus (DiNV) in its host (<italic>D. innubila</italic>), we sequenced wild-caught individuals from four populations with the expectation that some (~40% in previous samples) individuals would be infected (<xref ref-type="bibr" rid="bib39">Hill and Unckless, 2020</xref>). We considered strains to be infected with DiNV if they had at least 10x coverage for 95% of the genome. We confirmed the infection of a subsample of these using PCR for the presence of a DiNV specific gene (Supplementary Table 1). In total, we used sequencing information for 57, 92, 92, and 92 individuals from the Huachucas (HU), Santa Ritas (SR), Chiricahuas (CH), and Prescott (PR) populations with infection rates 26, 44, 71, and 79%, respectively (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - Table 1). We also sequenced 35 individual males collected in the Chiricahuas in 2001 (52% infected with DiNV) and 80 individual males collected in the Chiricahua’s in 2018 (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - Table 2, 40 infected with DiNV and 40 uninfected determined before sequencing using PCR).</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Candidate viral SNPs associated with viral titer with associated genes and the functional category of that gene.</title></caption><table frame="hsides" rules="groups"><thead><tr><th>SNP locus</th><th>Nearest gene <break/><break/></th><th>SNP functional annotation</th><th>Nearest gene <break/>functional annotation</th><th>p-value (FDR-corrected)</th></tr></thead><tbody><tr><td>G14249T</td><td><italic>19K/PIF-4</italic></td><td>Non-synonymous</td><td>Per OS Infectivity factor envelope protein, required for oral infection</td><td>3.49e-15 (4.89e-12)</td></tr><tr><td>C41210T</td><td><italic>LEF-4</italic></td><td>Non-synonymous</td><td>RNA polymerase subunit for RNA modification</td><td>7.99e-12 (1.12e-09)</td></tr><tr><td>G42389T</td><td><italic>gp83</italic></td><td>Upstream</td><td>Suspected virulence factor which suppresses Toll activity</td><td>6.73–14 (9.44e-11)</td></tr><tr><td>A59194G</td><td><italic>gp51</italic></td><td>Intergenic</td><td>Suspected virulence factor</td><td>2.96e-12 (4.15e-09)</td></tr><tr><td>C59275A</td><td><italic>PIF-6</italic></td><td>Upstream</td><td>Per OS Infectivity factor envelope protein, required for oral infection</td><td>1.65e-18 (2.31e-15)</td></tr><tr><td>T59276C</td><td><italic>PIF-6</italic></td><td>Upstream</td><td>Per OS Infectivity factor envelope protein, required for oral infection</td><td>1.65e-18 (2.31e-15)</td></tr><tr><td>G66615A</td><td><italic>gp19</italic></td><td>Non-synonymous</td><td>Suspected virulence factor</td><td>3.49e-15 (4.89e-12)</td></tr><tr><td>C78978T</td><td><italic>gp94</italic></td><td>Intergenic</td><td>Suspected virulence factor</td><td>5.93e-17 (8.32e-14)</td></tr><tr><td>G78991A</td><td><italic>gp94</italic></td><td>Intergenic</td><td>Suspected virulence factor</td><td>1.12e-16 (1.57e-13)</td></tr><tr><td>C126118A</td><td><italic>ODV-E56-2</italic></td><td>Upstream</td><td>Occlusion-derived virus envelope protein required for particle formation</td><td>1.43e-18 (2.01e-15)</td></tr><tr><td>A132593C</td><td><italic>Helicase-2</italic></td><td>Non-synonymous</td><td>Unwinds DNA and is critical for DNA replication</td><td>1.11e-09 (1.56e-06)</td></tr><tr><td>T140117C</td><td><italic>PIF-3</italic></td><td>Upstream</td><td>Per OS Infectivity factor envelope protein, required for oral infection</td><td>2.96e-15 (4.15e-12)</td></tr></tbody></table></table-wrap><p>We isolated and sequenced DNA from these samples, quality filtered reads, mapped to the genome, and called genetic variation in the viral genomes to assess the extent of adaptation in each viral population. Consistent with an arms-race between host and virus, envelope and novel virulence (GrBNV-like) genes have a significantly higher proportion of substitutions fixed by adaptive evolution, compared to other viral genes (McDonald-Kreitman-based statistic Direction of Selection [<xref ref-type="bibr" rid="bib98">Stoletzki and Eyre-Walker, 2011</xref>] and Selection Effect [<xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>; <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>], DoS &gt;0, GLM t-value &gt;1.31, p-value&lt;0.05).</p><p>Recurrent adaptive evolution in viral proteins known to interact with the host-immune system suggests an arms-race between <italic>D. innubila</italic> and DINV. To better understand the interactions between host and virus, we performed an association study between viral titer and natural genetic variation in both host and virus. We consider viral titer a proxy for virulence. For each virus-infected individual, we quantified viral titer (as the logarithm of the viral genome coverage normalized to host autosomal genome coverage) and performed an association study across both host and virus variable sites to identify variants significantly associated with viral titer using PLINK (<xref ref-type="bibr" rid="bib78">Purcell et al., 2007</xref>).</p><p>Of 5283 viral SNPs in the 155kbp DiNV genome, 1,403 SNPs are segregating in at least five infected host individuals (five was our minor allele threshold for inclusion in the association study). Of those 1,403 SNPs, 78 are significantly associated with viral titer after multiple testing correction (FDR &lt; 0.01, Significantly associated SNPs p&lt;0.001 over 1000 permutations, <xref ref-type="fig" rid="fig1">Figure 1A</xref>). Of these, 16 are less than 2000 bp upstream of the start site of a gene, 18 are coding nonsynonymous, 11 are coding synonymous and 33 are intergenic. The SNP with the absolute lowest <italic>p</italic>-value for association with viral titer was also the only significantly associated SNP that seemed to segregate within individuals. In fact, the frequency of the derived allele in this nonsynonymous polymorphism in the active site of <italic>Helicase-2</italic> shows a negative linear relationship with the log of viral titer (<xref ref-type="fig" rid="fig1s2">Figure 1—figure Supplement 2</xref>, GLM t-value = −20.516, p-value=5.55e-21). However, when we ranked samples by viral titer, we found the derived SNP frequency exhibited an almost bell-shaped curve when plotted against rank (Supplementary Figure 2).</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Viral genome-wide association study for DiNV titer in wild <italic>D. innubila</italic>.</title><p>(<bold>A</bold>) Manhattan plot for each DiNV SNP and the significance of its association with DiNV titer. SNP point, shape and color denotes if they are upstream (black upward arrow), downstream (black downward arrow), intergenic (gray cross), synonymous (blue circle), non-synonymous (red square) or nonsense (green asterisk) mutations. Named SNPs are either part of the significantly associated viral haplotype or, if a smaller size and italicized, are in or near other genes of interest (e.g. <italic>Helicase-2</italic>). The FDR-corrected <italic>p</italic>-value cutoff of 0.01 is shown as a dashed line (multiple testing correction for 1403 tests), while the permutation-based genome-wide significance threshold of p=0.01 is shown as a dotted line (based on 1000 permutations). (<bold>B</bold>) Viral titer for individual wild-caught flies infected with Low and High DiNV haplotypes (containing all 11 High type alleles). The middle bar represents median value, upper and lower bars represent 25<sup>th</sup> and 75<sup>th</sup> percentile and whiskers represent a 95% confidence interval. (<bold>C</bold>) Association between the number of High type SNP variants and the viral titer of a sample. (<bold>D</bold>) Across five populations, the frequency of the High type is correlated with the frequency of the virus infection. (<bold>E</bold>) Linkage disequilibrium heatmap between the eleven focal haplotype SNPs pairwise, and with the Helicase-2 SNP. Tiles are colored by the estimated linkage between SNPs, from red (highly linked, r<sup>2</sup> = 1) to white (unlinked, r<sup>2</sup> = 0).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Frequency of each significantly associated SNP within each individual, with individuals ranked by viral titer (left = lowest, right = highest), to show the strong linkage of SNPs and little evidence of co-infection among the high and low haplotype SNPs.</title><p>Also shown is the bell-shaped relationship between viral titer rank and the significantly associated SNP in <italic>Helicase-2</italic>, samples here are colored by the assigned haplotype, if the strain is High type (black), intermediate (gray) or Low type (white).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig1-figsupp1-v2.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Permutation test for association test for viral virulence.</title><p>(<bold>A</bold>) Distribution of mean differences between artificial High type and Low type titers generated by permuting the titer across strains. We binned the Titer of strains by the allele of most significant SNP from the permuted associations and its 10 most strongly linked other sites. Whichever group has the higher titer during in each permutation is taken as the High type. The mean difference between the true High type and Low type is shown as a dashed line on the plot. (<bold>B</bold>) qqPlot of observed p-values versus the permuted p-values expected under the null expectation of no association. Black circles represent the p-value comparisons including focal haplotype SNPs. Red squares represent p-value comparisons with number of focal SNPs as a covariate, excluding these SNPs from the association study. A one to one correlation is shown as a dashed line.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig1-figsupp2-v2.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Linkage disequilibrium between SNPs in DiNV as a heatmap generated in LDheatplot.</title><p>The labelled SNPs (significantly associated SNPs found in the association study) are strongly linked, shown by the high r<sup>2</sup> between these SNPs. Points are colored by the estimated linkage between SNPs, from red (highly linked, r<sup>2</sup> = 1) to white (unlinked, r<sup>2</sup> = 0). Linkage disequilibrium heatmap for just the significantly associated SNPs is also shown to show the high linkage between SNPs.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig1-figsupp3-v2.tif"/></fig></fig-group><p>We also identified a striking association between viral titer and eleven strongly linked polymorphisms found across the DiNV genome (<xref ref-type="fig" rid="fig1">Figure 1A &amp; D</xref>, highlighted SNPs, <xref ref-type="table" rid="table1">Table 1</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplements 1</xref>–<xref ref-type="fig" rid="fig1s3">3</xref>, Sig. SNPs). These SNPs were significantly associated in every population when performing the association study on individual populations, or as a single group with population as a covariate. We binned strains with one of the two complete sets of alleles and referred to them as the ‘High Type’ and ‘Low Type’ (<xref ref-type="fig" rid="fig1">Figure 1B</xref>, only contains individuals infected by viruses with a complete set of ‘High’ or ‘Low’ Type SNP variants). This multilocus genotype includes three non-synonymous SNPs, five SNPs in the UTRs of known virulence factor genes and three intergenic SNPs (<xref ref-type="table" rid="table1">Table 1</xref>). Viral titer is, on average, 100-fold higher in individuals infected with the full High type virus compared to the ancestral Low type (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). When comparing the ‘High’ and ‘Low’ viral type to <italic>Helicase-2</italic> allele frequency, we found that viral titer increases with <italic>Helicase-2</italic> SNP frequency in the Low type (and some intermediate types), while titer decreases with <italic>Helicase-2</italic> SNP frequency for most intermediate and High types, suggesting some form of negative interaction between the High type SNP variants and the <italic>Helicase-2</italic> SNP (Supplementary Figure 2). Consistent with this, we found a negative interaction between <italic>Helicase-2</italic> SNP frequency and the presence of High type on viral titer (GLM Log10(titer) ~<italic>Helicase-2</italic> SNP * Haplotype: t-value = −7.815, p-value=6.39e-14).</p><p>Since these 11 SNPs are almost perfectly linked (<xref ref-type="fig" rid="fig1">Figure 1E</xref>), if one SNP were a false positive result, all would be false positives. We sought to address this in two ways. First, these SNPs are significantly associated with viral titer in every population when performing the association test separately or as a group with population as a covariate. Therefore, the likelihood of the same SNP(s) being false positives in three independent tests is quite small (p&lt;1E-09 over 3 iterations of 1000 permutations). Second, we permuted the titer among all individuals within populations and performed the association test (100,000 total permutations). For each permutation, we took the SNP with the lowest <italic>p</italic>-value from each permutation test and the 10 other SNPs most strongly linked to it. We then compared the viral titer in those with this permuted ‘High’ haplotype to those with the ‘Low’ haplotype. In no case were the permuted differences between High and Low haplotypes as large as those observed from the real data (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2A</xref>, 100,000 permutations), and the distribution of true <italic>p</italic>-values diverges dramatically from the permuted null expectation (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2B</xref>, black). We found this divergence from the null expectation of <italic>p</italic>-values disappears when including viral haplotype as a covariate (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2B</xref>, red), suggesting that the High type drives most of the signal seen in the association study.</p><p>Though we found few strains with an intermediate number of SNPs (intermediate types), viral titer appears to increase as the number of High type SNPs increases (<xref ref-type="fig" rid="fig1">Figure 1C</xref>, GLM t-value = 34.971, p-value=5.912e-16), but the rate of increase slows as the number of High type SNPs increases suggesting diminishing returns (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). Some of these polymorphisms are associated with known virulence factors, or are related to the formation of the viral envelope co-opting the host vesicle trafficking system and are rapidly evolving in nudiviruses (e.g. <italic>19K, ODV-E56, PIF-3</italic>) (<xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>; <xref ref-type="bibr" rid="bib37">Hill and Unckless, 2017</xref>; <xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). Additionally, several are associated with genes exclusive to a few nudivirus genomes thought to be novel virulence factors, including <italic>gp83,</italic> a gene that downregulates Toll-induced antimicrobial peptides (AMPs) and upregulates those induced by IMD (<xref ref-type="bibr" rid="bib70">Palmer et al., 2019</xref>). Both the Toll and IMD pathway may interact with DNA viruses (<xref ref-type="bibr" rid="bib107">Zambon et al., 2005</xref>; <xref ref-type="bibr" rid="bib15">Costa et al., 2009</xref>; <xref ref-type="bibr" rid="bib63">Merkling and van Rij, 2013</xref>; <xref ref-type="bibr" rid="bib29">Ferreira et al., 2014</xref>; <xref ref-type="bibr" rid="bib49">Lamiable et al., 2016</xref>; <xref ref-type="bibr" rid="bib70">Palmer et al., 2019</xref>).</p><p>Among populations, we found a positive correlation between the frequency of the High type and overall DiNV infection frequency (<xref ref-type="fig" rid="fig1">Figure 1D</xref>, GLM logistic regression z-value = 6.104, p-value=0.00883), suggesting that the High type may have a higher transmission rate than the Low type, resulting in a larger number of new individuals infected per DiNV-infected individual. We also found both viral types in collections from 2001 and 2017, with the High type significantly more common in the 2017 collection (Fisher Exact Test p-value=0.0167, <xref ref-type="fig" rid="fig1">Figure 1D</xref>).</p><p>It is possible that mutations could segregate within individuals, but this is relatively rare overall, and we found no evidence of the 11 haplotype SNPs segregating within any individual fly (Supplementary Figure 2). In fact, no individual fly contains more than two significantly associated SNPs segregating within the sample, suggesting that hosts are either infected completely with Low type or High type virus particles.</p></sec><sec id="s2-2"><title>The high DiNV viral type more effectively suppresses the Drosophila immune system</title><p>Given the striking difference in viral titer between High and Low type viruses, we sought to further characterize the differences in infection dynamics between the types, focusing on the differences in expression between viral types. We sequenced mRNA from 80 wild <italic>D. innubila</italic> males collected in 2018 (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - Table 2, 40 infected with DiNV, 40 uninfected) and performed a differential expression analysis between infected and uninfected individuals. Few genes were differentially expressed (DE) between infection states in <italic>D. innubila</italic>, but these DE genes were enriched for several interesting categories. Specifically, we found IMD-induced antimicrobial peptides (AMPs) were upregulated upon DiNV infection, while one Toll-induced AMP, and several chorion and heat shock protein genes were downregulated (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A</xref>). We also compared these results to a laboratory experiment in <italic>D. melanogaster</italic> of differential expression after infection with a close relative of DiNV (<xref ref-type="bibr" rid="bib69">Palmer et al., 2018</xref>). These same genes tend to also be differentially expressed in <italic>D. melanogaster</italic>. Of the 12 genes which are differentially expressed in the same direction in both species, five are AMPs and five are chorion genes (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1B</xref>). This suggests that even with significant evolutionary divergence in both host and virus, the transcriptional response to infection is similar in both hosts. It was unexpected that chorion proteins are differentially expressed upon infection by DiNV and Kallithea virus (<xref ref-type="fig" rid="fig2">Figure 2</xref>, <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>), especially as DiNV is thought to infect the Drosophila gut (<xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>) and these samples are male. This suggests DiNV may infect more tissues than just the gut, and that chorion proteins are expressed in the germline of both <italic>D. innubila</italic> sexes, potentially fulfilling different roles in different sexes/species.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>DIfferential expression between two viral haplotypes.</title><p>(<bold>A</bold>) Differential expression of <italic>D. innubila</italic> and DiNV genes between <italic>D. innubila</italic> infected with either the Low type or High type DiNV multilocus genotypes. For host genes, the log-fold change of mRNA fragments per million fragments is compared, while for viral genes the log-fold change of viral mRNA fragments per million fragments per viral particle is compared. Genes are colored/labelled by categories of interest, specifically antimicrobial peptides (AMPs), proteins involved in the extracellular matrix and viral proteins. Specific genes of interest, such as <italic>Myd88</italic>, are also named. The FDR-corrected significance cut-off of 0.01 (10,320 tests) is shown as a dashed line. (<bold>B</bold>) Expression (in FPKM per viral particle) of <italic>gp83</italic> increases with the number of High type SNPs.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Volcano plot of changes in gene expression between <italic>D. innubila</italic> infected with DiNV and uninfected controls.</title><p>(<bold>A</bold>) Gene categories of interest, such as enriched categories, are highlighted in color. The FDR-correct significance cut-off of 0.01 (10,320 tests) is shown as a dashed line. (<bold>B</bold>) Comparison of gene expression changes upon infection for <italic>D. innubila</italic> and <italic>D. melanogaster</italic>. Significantly differentially expressed genes (p-value&lt;0.01, FDR-corrected) are colored, genes differentially expressed in both species are colored blue, genes differentially expressed in just <italic>D. melanogaster</italic> are colored yellow and genes differentially expressed in just <italic>D. innubila</italic> are colored red.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Expression changes (shown as transcript fragments per 1 million reads per 1kbp of exon) of antimicrobial peptides between strains infected with High type DiNV, Low type DiNV or not infected.</title><p>Black stars above low samples show significant differential expression between DiNV-infected strains and uninfected strains (multiple testing corrected p-value&lt;0.05). Red stars above high samples show significant differential expression between Low type infected strains and high type infected strains (multiple testing corrected p-value&lt;0.05).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig2-figsupp2-v2.tif"/></fig></fig-group><p>To determine how the High and Low type differ in their ability to infect <italic>D. innubila,</italic> we compared gene expression between <italic>D. innubila</italic> infected with the two types from this same collection (excluding intermediate types, which also showed a significant difference in viral titer between High type and Low type, p-value=0.0000143). We found 17 host genes and nine viral genes differentially expressed between types (after we controlled for virus copy number as FPKM/titer for viral genes <xref ref-type="fig" rid="fig2">Figure 2A</xref>, FDR-corrected p-value&lt;0.01). Specifically, three Toll-regulated immune peptides (<italic>IM33</italic>, Bomanins <italic>BomBC2,</italic> and <italic>BomT2</italic>) and one JAK-STAT regulated immune peptide (<italic>Listericin</italic>) have reduced expression in High type infected individuals compared to the Low type (<xref ref-type="fig" rid="fig2">Figure 2</xref>, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). Viral genes of interest (<italic>PIF-3, 19K, gp83</italic>) have higher expression per viral particle (FPKM/titer) in the High type compared to the Low type. <italic>gp83</italic> also increases in expression per viral particle (FPKM/titer) as the number of High type alleles increases (<xref ref-type="fig" rid="fig2">Figure 2B</xref>, t-value = 13.732, p-value=3.36e-15). Together these results suggest that the High type has increased expression of key virulence factors, which in turn, manipulate the expression of host genes involved in immune defense to result in the observed differences in viral titer. Specifically, higher <italic>gp83</italic> expression may cause the lower Toll-mediated AMP expression (<xref ref-type="bibr" rid="bib70">Palmer et al., 2019</xref>). We reasoned that if the High type is suppressing the host Toll pathway, <italic>Myd88</italic>, the Toll signaling protein upstream of AMPs would also have lower expression. We did find <italic>Myd88</italic> expression is lower in strains infected with the High type (<xref ref-type="fig" rid="fig2">Figure 2A</xref>, though no significantly so), which in turn might prevent the host from enacting a proper immune response to DiNV infection (<xref ref-type="fig" rid="fig2">Figure 2</xref>, <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>).</p></sec><sec id="s2-3"><title>Experimental infections recapitulate differences in viral type virulence</title><p>To assess if the virulence differs between virus types, we performed experimental infections of <italic>D. innubila</italic> males using viral filtrate of strains infected with one of the two types of DiNV. Before injections, we used qPCR to identify the viral titer of each sample and dilute all samples to the same viral concentration. In these experiments, as viral titer increases, the survival of infected flies decreases, regardless of viral type (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplements 1</xref> and <xref ref-type="fig" rid="fig3s2">2</xref>, ANOVA residual deviance = 3.536, p-value=2.454e-07, Cox Hazard Ratio z-value &gt;2.227, p-value&lt;0.02592). In both types viral titer also increased for the first 3 days of infection (GLM t-value = 9.817, p-value=3.6e-14). This established that viral titer is a reasonable proxy for virulence and that DiNV is virulent in <italic>D. innubila</italic>.</p><p>For a set of four High type and four Low type-infected individuals, we isolated viruses, diluted to roughly equal concentrations of viral particles and performed infections for replicates of 10 males with microneedles dipped in one of the filtrate samples. Survival is significantly lower for flies infected with High type viruses when compared to either flies pricked with sterile media (<xref ref-type="fig" rid="fig3">Figure 3A</xref>, Cox Hazard Ratio z-value = 3.671, p-value=0.000242) or those pricked with Low type virus (<xref ref-type="fig" rid="fig3">Figure 3A</xref>, Cox Hazard Ratio z-value = 4.611, p-value=4.02e-06). Flies pricked with Low type virus show a non-significant reduction in survival compared to control flies (Cox Hazard Ratio z-value = 1.353, p-value=0.176). We also measured viral titer over time using qPCR, and found titer increases through time in flies infected with either type (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, GLM Log<sub>10</sub>(titer) ~days + type + vial | strain, days t-value = 9.912, p-value=1.76e-14). Flies infected with High type virus have significantly higher viral titer compared to flies infected with Low type virus (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, GLM Log<sub>10</sub>(titer) ~days + type + vial | strain, type t-value = 3.934, p-value=0.000211). These results suggest that the higher viral titer observed in the High Type is also associated with higher virulence.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Effect of viral type in experimental infections.</title><p>(<bold>A</bold>) Survival curves of <italic>D. innubila</italic> infected with high and low viral types compared to control flies pricked with sterile media, for 15 days post-infection. Survival 5 days post-infection separated by strain is shown in <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>. (<bold>B</bold>) qPCR copy number of viral <italic>p47</italic> relative to <italic>tpi</italic> in <italic>D. innubila</italic> infected with DiNV filtrate of high and low types.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Effect of differences in viral type and titer in experimental infections.</title><p>(<bold>A</bold>) Survival curves of <italic>D. innubila</italic> infected with DiNV filtrate of different dilutions compared to control flies pricked with sterile media, for 15 days post infection. Each dilution was used to infect 30 flies in 3 replicates of 10. (<bold>B</bold>) qPCR copy number of viral <italic>p47</italic> relative to <italic>tpi</italic> in samples of <italic>D. innubila</italic> infected with DiNV filtrate of different dilutions, between 1 and 1000 viral particles per host genome copy. 2 to 3 replicates of 5 flies were quantified at each timepoint, with later time points limited to two time points due to lack of surviving flies.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Survival of flies following infection with different titers and types of DiNV.</title><p>(<bold>A</bold>) Survival of <italic>D. innubila</italic> reference strain 5 days post infection, using filtrate from different samples versus uninfected control, colored by High type virus or Low type virus. (<bold>B</bold>) Survival of <italic>D. innubila</italic> reference strain 5 days post infection using serial dilutions of IPR01 filtrate versus control. Each viral sample/dilution was used to infect 40 flies in 4 replicates of 10. (<bold>C</bold>) Viral titer estimated per viral genotype at 5 days post-infection, colored by High type virus or Low type virus. (<bold>D</bold>) Viral titer of DiNV infecting <italic>D. innubila</italic> reference strain 5 days post infection using serial dilutions of IPR01 filtrate versus control. Each sample/dilution was used to infect 40 flies in 4 replicates of 10, we then quantified the viral titer of surviving flies in groups of 5 flies, varying from 4 to 6 replicates depending on the number of surviving flies.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig3-figsupp2-v2.tif"/></fig></fig-group></sec><sec id="s2-4"><title>DiNV types are under strong selection in the host</title><p>We next tested whether genes likely to be involved in host/virus interaction show signs of recurrent natural selection. Using McDonald-Kreitman based statistics for estimating the proportion of substitutions fixed by selection (<xref ref-type="bibr" rid="bib61">McDonald and Kreitman, 1991</xref>; <xref ref-type="bibr" rid="bib98">Stoletzki and Eyre-Walker, 2011</xref>; <xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>), we tested whether genes that are associated with the High and Low types exhibited different signatures of natural selection compared to other viral genes. We calculated the Selection Effect, the proportion of substitutions fixed by adaptive evolution, weighted by the total number of substitutions in the genome (<xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>). We found that genes associated with SNPs in the initial association study for viral titer, which defined the High and Low types (listed in <xref ref-type="table" rid="table1">Table 1</xref> and including <italic>19K, PIF-3</italic> and <italic>LEF-4</italic>) have significantly higher rate of substitutions being fixed due to selection than background genes across all categories (<xref ref-type="fig" rid="fig4">Figure 4</xref>¸ type-associated genes versus all other, t-value = 2.718, p-value=0.00068). When separating the polymorphisms from the substitutions, we found that envelope genes have a significant excess of functional substitutions per site compared to other genes (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>, GLM t-value = 3.62, p-value=0.00107), while genes of unknown function have a significant excess of non-synonymous polymorphisms (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>, GLM t-value = 2.33, p-value=0.02241) and a deficit of non-synonymous substitutions (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>, GLM t-value = −3.894, p-value=9.93e-05), which will lower the average selection effect of background genes, suggesting that the relative excess adaptation seen in some genes is in part due to an excess of variation in genes on unknown function.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Genes implicated in host/virus interaction are rapidly evolving by positive selection in the Chiricahua population.</title><p>Difference in selection effect for viral and host gene categories of interest from nearby background genes (average shown as 0, the dashed line), as indicated by the proportion of substitutions fixed by adaptation, weighted by mutations in SnIPRE (<xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>). Genes that have associated SNPs from the association study are highlighted in red, while genes which are differentially expressed upon infection, or between viral types are labelled in blue. All association study hits are also differentially expressed and labelled in red. Genes of interest are named.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>McDonald-Kreitman based statistics for each gene in population of <italic>Drosophila</italic> innubila Nudivirus, with viral envelope and GrBNV potential virulence factors shown separately.</title><p>DoS = direction of selection, Selection Effect = SnIPRE estimated weighted DoS. Boxplots marked with a * are significantly higher than background/other viral genes (GLM p-value&lt;0.05). In both cases, values above 0 suggest some proportion of substitutions are fixed by adaptive evolution.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>The ratio of nonsynonymous to synonymous polymorphisms and substitutions for both DiNV and <italic>D. innubila</italic>, used to generate <xref ref-type="fig" rid="fig4">Figure 4</xref>.</title><p>(<bold>A</bold>) The ratio of nonsynonymous polymorphism (Pn) to synonymous polymorphism (Ps) for DiNV functional categories, the ratio of nonsynonymous substitutions (Dn) to synonymous substitutions (Ds) for DiNV functional categories and the resulting estimate of Selection effect (estimated using SnIPRE) for DiNV functional categories. (<bold>B</bold>) The ratio of nonsynonymous polymorphism (Pn) to synonymous polymorphism (Ps), the ratio of nonsynonymous substitutions (Dn) to synonymous substitutions (Ds) and the estimated Selection effect, all for the host <italic>D. innubila</italic>, separated by functional categories of interest highlighted in the differential expression analysis or the association study. We have included a point for each gene in all categories apart from the background, due to the sheer number of points in the background category. When 0 polymorphisms or substitutions are found in a gene, that gene is not shown in the Pn/Ps or Dn/Ds plots.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig4-figsupp2-v2.tif"/></fig><fig id="fig4s3" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 3.</label><caption><title>Host genome-wide association study for DiNV titer in wild <italic>D.innubila</italic>.</title><p>(<bold>A</bold>) Manhattan plot of significance of SNP on viral titer after factoring in interaction with the viral haplotype. The significance cut offs are labelled (p-value&lt;0.05 after multiple testing correction dotted, p-value&lt;0.05 after permutations dashed). (<bold>B</bold>) Manhattan plot of SNP x viral haplotype interaction for viral titer association study in <italic>D. innubila</italic>, calculated using <italic>PLINK</italic>. The significance cut offs are labelled (p-value&lt;0.05 after multiple testing correction dashed, p-value&lt;0.05 after permutations dotted). (<bold>C</bold>) Manhattan plot of SNP * sex interaction for viral titer association study in <italic>D. innubila</italic>, calculated using <italic>PLINK</italic>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig4-figsupp3-v2.tif"/></fig></fig-group><p>We also performed an association study using the host polymorphism and found 13 significantly associated SNPs, after controlling for the viral haplotype (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>, p&lt;0.01), but found no significant enrichments (p-value&gt;0.05) or genes of interest (e.g. those involved in Toll signaling or antiviral pathways). We therefore looked for enrichments in genes associated with the top 100 significantly associated SNPs versus all other genes and found piRNA genes enriched (GO enrichment = 3.44, p-value=0.035, FDR-corrected). Given that siRNA genes are not highly ubiquitously expressed in <italic>D. innubila</italic> (<xref ref-type="bibr" rid="bib36">Hill et al., 2019</xref>), that DiNV reduces host fecundity (<xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>), and that a close relative of DiNV infects the host ovaries (<xref ref-type="bibr" rid="bib69">Palmer et al., 2018</xref>), DiNV could interact with chorion proteins during oogenesis, and piRNAs could be suppressing DiNV (<xref ref-type="bibr" rid="bib50">Lewis et al., 2018</xref>). Consistent with the arms race model, host genes we suspect are interacting with DiNV (such as the association study hits, AMPs, chorion genes, piRNA genes, and extracellular genes) show elevated levels of substitutions fixed by selection compared to background genes in <italic>D. innubila</italic> (<xref ref-type="fig" rid="fig4">Figure 4</xref> and <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, GLM p-value&lt;0.05) (<xref ref-type="bibr" rid="bib39">Hill and Unckless, 2020</xref>). Finally, differentially expressed chorion genes, extracellular genes and AMPs have significantly more adaptive substitutions than non-differentially expressed genes in the same categories (<xref ref-type="fig" rid="fig4">Figure 4</xref>, blue dots, differentially expressed versus all other T-test: <italic>D. innubila</italic> t-value = 4.755, p-value=0.000671). When separating the polymorphisms from the substitutions, we found an excess of functional substitutions per site in AMPs (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2B</xref>, GLM t-value = 4.776, p-value=1.81e-06), driving their excess of adaptive substitutions compared to other genes. Overall, these results suggest strong selection is acting on both the host to suppress viral activity and the virus to escape this suppression.</p></sec><sec id="s2-5"><title>Recombination in DiNV may facilitate the evolution of the high type</title><p>We were interested in examining the genetic variation of DiNV to determine the relationship of the host and the two putative types of DiNV. To determine the appropriate approaches for measuring these patterns, we first need to determine the effective rate of recombination in the virus. Though recombination is necessary for proper nudivirus replication (<xref ref-type="bibr" rid="bib46">Kelly, 1982</xref>; <xref ref-type="bibr" rid="bib44">Kamita et al., 2003</xref>; <xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>), often these recombination events will be between nearly identical viral particles resulting in no detectable signature of recombination across the genome (<xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>). For recombination to leave its signature in genetic variation, two divergent viruses must coinfect the same cell – we refer to this as effective recombination. It is unclear how common such recombination is in nudiviruses.</p><p>We used three methods to examine rates of effective recombination across the DiNV genome: First we used GARD to identify the number of recombination breakpoints across samples (<xref ref-type="bibr" rid="bib48">Kosakovsky Pond et al., 2006</xref>). Second, we screened for recombination events by finding all four combinations of alleles between two SNPs. Third we calculated the linkage disequilibrium pairwise between all SNPs (<xref ref-type="bibr" rid="bib90">Shin et al., 2006</xref>). We found recombination is relatively common in DiNV (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>), with 307 potential recombination events genome-wide in recent history (based on combinations of alleles across our samples), and that the genome has relatively low linkage disequilibrium (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>, <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplements 1</xref> and <xref ref-type="fig" rid="fig5s2">2</xref>). In DiNV, the eleven SNPs significantly associated with viral titer (the High and Low types), are spread across the ~155 kilobase pair genome, yet are nearly perfectly linked to each other but not to other SNPs (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>, <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplements 1</xref> and <xref ref-type="fig" rid="fig5s2">2</xref>).</p><p>We wanted to examine if it was possible that the High type was generated by recombination of SNPs onto the same background, under the assumption that if the High type was formed via recombination we would expect to find recombination events each side of each significantly associated SNP. Using GARD and the four-allele test, six of the significantly associated SNPs show evidence of a recombination event on one side of the SNP, while three have evidence of a recombination event on each side of the significantly associated SNP (<xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>, Supplementary Data). This suggests that recombination could have aided in the formation of the multilocus genotype, by recombination between two intermediate types to form the High type. However, we found no evidence of recombination between individuals in different populations (e.g. no recombinant haplotypes that are ½ CH and ½ PR), suggesting little movement of one or more significantly associated SNPs between populations. We also found no evidence of recombination between the complete High and complete Low types. This suggests that the initial SNPs for the High type have evolved on separate backgrounds in all three populations, though recombination between intermediate strains may have allowed for the formation of the complete High type, or for the generation of other intermediate types. For this to happen, each SNP may have recurrently evolved in each population, even if not sequentially.</p></sec><sec id="s2-6"><title>The high viral type of DiNV evolved repeatedly in three <italic>D. innubila</italic> populations</title><p>We next sought to understand the evolutionary origin of the two putative types. Given that both types are found in all populations surveyed (<xref ref-type="fig" rid="fig1">Figure 1D</xref>) with no evidence of recombination between populations, we hypothesized that this could occur one of three ways: First, the derived haplotype was present ancestrally and has been maintained since before geographic isolation occurred. Second, the derived haplotype evolved following geographic isolation and has spread via migration between locations. Third, the derived haplotype has recurrently evolved in each location.</p><p>To distinguish between these possibilities and determine the timeframe of divergence, we used the site frequency spectrum of silent DiNV polymorphism to estimate effective population size backwards in time for all populations (<xref ref-type="bibr" rid="bib54">Liu and Fu, 2015</xref>). We found that the three populations (CH, HU and SR) expanded from a single viral particle (N<sub>e</sub> = 1) to millions of particles during the last glacial maximum (30–100 thousand years ago) when <italic>D. innubila</italic> settled its current range (<xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4</xref>; <xref ref-type="bibr" rid="bib39">Hill and Unckless, 2020</xref>). This supports a single invasion event during a host-range change for each location. PR appears to expand between 1 and 10 thousand years ago, suggesting a much more recent bottleneck during the range expansion in PR (<xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4</xref>; <xref ref-type="bibr" rid="bib39">Hill and Unckless, 2020</xref>).</p><p>We aligned genomic regions containing SNPs to two related nudiviruses, Kallithea virus and Oryctes rhinoceros Nudivirus (OrNV) (<xref ref-type="bibr" rid="bib102">Wang et al., 2008</xref>; <xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>; <xref ref-type="bibr" rid="bib69">Palmer et al., 2018</xref>). The High type alleles are not present in either Kallithea or OrNV, and are not found in short read information for wild <italic>D. melanogaster</italic> infected with Kallithea virus (<xref ref-type="bibr" rid="bib103">Webster et al., 2015</xref>), suggesting they are derived in DiNV.</p><p>We generated consensus DiNV sequences for each infected <italic>D. innubila</italic> individual and created a whole-genome phylogeny to infer geographic diffusion of samples using BEAST2 (<xref ref-type="bibr" rid="bib5">Bouckaert et al., 2014</xref>). We then performed ancestral reconstruction of the presence of the High type across the phylogeny using APE (<xref ref-type="bibr" rid="bib71">Paradis et al., 2004</xref>). Our samples grouped as three populations (with HU and SR forming one population) and, consistent with our expectation, the Low type is the ancestral state (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). Interestingly we found PR is nested within HU/SR, based on five SNPs which segregate in the Low type HU/SR but are fixed in all PR samples (<xref ref-type="fig" rid="fig5">Figure 5A</xref>), consistent with the expansion of DiNV north over time. Surprisingly, the High type appears to have evolved repeatedly and convergently within each population, forming separate groups within each population (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). The High type also clusters within each population in a phylogeny or principal component analysis of all viral SNPs (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1C</xref>), and when repeating these analyses while excluding the eleven focal SNPs.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>The evolution and maintenance of two viral types.</title><p>(<bold>A</bold>) Phylogeographic reconstruction of the spread of DiNV through <italic>D. innubila,</italic> rooted on the Kallithea virus reference sequence, including a reconstruction of the High type evolution (with strains containing all 11 High type variants shown in purple, strains with an intermediate number of high type variants are shown in pink, and strains with no high type variants are shown in black). Branches are colored when the SNPs found in the background for each High haplotype are present in the population, showing that the background differs per population. Black branches show states where branch tips do not contain all the shared High/Low population specific background SNPs. (<bold>B</bold>) Order of mutations in the viral haplotype appearing in each population. Apart from three mutations the order is consistent between locations. SNP order has been bootstrapped across multiple sample phylogenies, SNPs with bootstrap support of &gt;95% are labelled with a *. (<bold>C</bold>) The number of generations needed for ‘High titer’ mutations to evolve in simulated populations, given that each mutation increases the mutation rate. The number of generations between each mutation appearing decreases as titer increases.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>SNP association in the viral haplotype and principle component analysis of viral samples.</title><p>(<bold>A</bold>) Frequency of samples with different numbers of SNPs in the viral haplotype, there are very few intermediate types. (<bold>B</bold>) Frequency of each SNP in samples infected with the virus, showing there is little evidence of co-infections. (<bold>C</bold>) Principal component analysis of DiNV strains using variation of strains. Strains are colored by the viral type, showing its recurrent evolution. Point shape denotes species in which DiNV was found (<italic>D. azteca, D. falleni or D. innubila</italic>). Strains cluster by collection location. (<bold>D</bold>) Linkage between SNPs in the viral haplotype (r<sup>2</sup>) and other SNPs in the haplotype, to other SNPs in the viral genome.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig5-figsupp1-v2.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Linkage (r<sup>2</sup>) between different types of SNPs in each population of DiNV, and across all samples.</title><p>Other = SNPs which are not significantly associated with DiNV titer and do not form the viral haplotype. Sig = SNPs which are significantly associated with DiNV titer and do not form the viral haplotype. Background = SNPs which are in the same background that the high viral type evolved on in each population. The Other–Other relationship line strongly overlaps with the relationship line seen for all SNPs, due to this we have not included all SNPs together in this plot.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig5-figsupp2-v2.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>Associations of significantly associated SNPs and their neighboring SNPs, to attempt to determine if gene conversion is the cause of the recurrent evolution of the viral haplotype.</title><p>The significantly associated SNPs are shown at position 0, with the 5 SNPs upstream (−5 to −1) and downstream (1 to 5) plotted around them for each strain. Derived allele strains are shown in black, while ancestral allele strains for each position are shown in gray. For the helicase-2 allele (132593), windows are plotted as derived if the allele frequency is greater than 0.5.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig5-figsupp3-v2.tif"/></fig><fig id="fig5s4" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 4.</label><caption><title>Effective population size backwards for each population of DiNV going backwards in time, estimated using StairwayPlot.</title><p>Dotted lines indicate the error windows for N<sub>e</sub> at a given time point. Lines are colored by population.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig5-figsupp4-v2.tif"/></fig></fig-group><p>We next surveyed each background SNP (e.g. SNPs not associated with the High or Low type) to determine if the general background supports one of the three outlined ways in which the High type evolved and spread in each location. We grouped SNPs by their presence in just the High type or Low type (supporting a single origin and spread by migration) or if they were unique to a single population but shared between the High and Low types. In total, 341 SNPs (24% of SNPs surveyed) are unique to a single population yet are still shared between both High and Low types (<xref ref-type="fig" rid="fig5">Figure 5</xref> and <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>), compared to 23 SNPs (including the 11 High type SNPs) shared between locations but exclusive to High type samples. This is consistent with the lack of recombination between individuals in different populations, suggesting the High type associated SNPs are derived recurrently in each population on different backgrounds. Consistent with this, using TreeTime we found that the eleven significantly associated SNPs could have recurrently evolved across our samples, after accounting for recombination and mutation rate (<xref ref-type="bibr" rid="bib48">Kosakovsky Pond et al., 2006</xref>; <xref ref-type="bibr" rid="bib87">Sagulenko et al., 2018</xref>).</p><p>We identified 341 SNPs (161 for CH, 127 for HU and SR, and 53 for PR), are present in all High type samples of a single population but a variable proportion of Low types for that population (between 19 and 94%) and are unique to that population. This pattern fits with the High type recurrently evolving on a single background (a different background in each location), supporting recurrent evolution of the High type, and against two ancestrally maintained viral types. The population-specific background SNPs are spread throughout the DiNV genome, with little evidence of recombination with the High type SNPs, making it unlikely that these SNPs recombined onto different backgrounds, and were instead present when the High type first evolved in each population (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplements 1</xref> and <xref ref-type="fig" rid="fig5s2">2</xref>). We found little evidence of gene conversion producing this background signature (<xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>). While we identified recombination events between background SNPs in the Low type, we found no evidence of this occurring due to their fixed states in the High type, likely because the High type has swept to higher frequencies within individuals before recombination can occur.</p></sec><sec id="s2-7"><title>The elevated DiNV mutation rate associated with increased titer allows the high type to recurrently evolve in each population</title><p>Though there is strong linkage between the High type SNPs, they are not perfectly associated with each other (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplements 1</xref> and <xref ref-type="fig" rid="fig5s2">2</xref>). Using this slight disassociation and APE (<xref ref-type="bibr" rid="bib71">Paradis et al., 2004</xref>), we performed ancestral reconstruction of SNP origins in each population (assuming recurrent evolution) and found that, excluding three variable SNPs, the evolution of these SNPs was a similar order in each population (<xref ref-type="fig" rid="fig5">Figure 5B</xref>).</p><p>The recurrent evolution of these mutations in a specific order is feasible as the mutation rate is high and the waiting time between mutations should decrease as titer increases. Logically, if the wait time for the appearance of a beneficial mutation is 1/(2N<sub>e</sub>*μ*s), increasing N<sub>e</sub> (with titer) should decrease the wait time for the appearance of the High type mutations (<xref ref-type="bibr" rid="bib33">Gillespie, 2004</xref>). To determine if this recurrent evolution is plausible in our estimated timeframe (~10,000 years), we simulated viral populations using a discrete susceptible-infectious model implemented in deSolve (<xref ref-type="bibr" rid="bib95">Soetaert et al., 2010</xref>) using estimated baculovirus mutation rates, recombination rates, ranges of viral titer taken our samples and estimated population sizes for each viral population (parameters described in the methods). In this model we used an effective mutation rate scaled to viral titer, considering the mutation rate per particle, so total mutations per generation increase with viral titer. For simplicity, we limited these simulations to the first five mutations as these are mutations of largest effect and appear to be necessary for the remaining six mutations to appear. We also included epistasis which reduces the increase in titer for each subsequent mutation (<inline-formula><mml:math id="inf1"><mml:msup><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:msqrt><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mo>.</mml:mo> <mml:mi/><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:msqrt></mml:mrow></mml:msup></mml:math></inline-formula>), as a similar reduction is seen in our samples (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). The simulations suggest that waiting time for the first mutation that increases titer is highly variable between replicates but usually occurs within 1000 generations (~200 years at most, assuming the virus is transmitted from adult hosts to larval hosts via feces, with a 5 host generations per year as a conservative minimum estimate of viral generations per year, in &gt;99.93% of replicates, <xref ref-type="fig" rid="fig5">Figure 5C</xref>). The average wait time for each subsequent mutation decreases monotonically (GLM t-value = -2.389, <italic>p</italic>-value = 0.03686). In most cases, the next mutation appears in the background of the previous high titer mutation (<xref ref-type="fig" rid="fig5">Figure 5C</xref>) due to the elevated effective mutation rate and increased basic reproduction number (R<sub>0</sub>). In 34 of 1000 simulations, when a mutation does appear on a different background, recombination facilitates the generation of the full complement of mutations. Given that effective mutation rate is much higher than the effective recombination rate following the appearance of the first mutation (as recombination does not scale with titer in nudiviruses and baculoviruses <xref ref-type="bibr" rid="bib44">Kamita et al., 2003</xref>), this likely accounts for the mutation rates dominance in our simulations, though recombination likely does play a role in recombination between intermediate strains. The accumulation of mutations occurs at close to a geometric (approximately exponential) rate. Additionally, the standard deviation of time wait times also decreases with each new mutation (GLM t-value = -2.441, <italic>p</italic>-value = 0.04241), increasing the certainty that the entire multilocus genotype will appear in a population rapidly once the initial mutations appear. This chain reaction of adaptation could easily facilitate the repeated evolution of the virulent High type independently in three populations, with all eleven mutations fixing in a population within 6000 generations (~1200 years maximum) in all replicates (3372 generations on average, ~675 years maximum), a plausible amount of time given our estimated timeframe.</p></sec><sec id="s2-8"><title>Both viral types are found in two other <italic>Drosophila</italic> species and have also evolved in a geographically distinct population</title><p>Since we found two types of DiNV are present in all <italic>D. innubila</italic> populations, and that other species are infected with DiNV (<xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>), we hypothesized that another species could be a reservoir for the less virulent Low type. We chose to study <italic>D. azteca</italic> from the Chiricahuas since it is frequently infected with DiNV (~33% infection), overlaps with <italic>D. innubila,</italic> and is genetically divergent (40–60 million years) which could mean a very different genetic interaction between host and virus (<xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>). We also examined DiNV-infected <italic>D. falleni</italic> (a close relative of <italic>D. innubila</italic> with nonoverlapping geographic range, collected in Athens, Georgia) as an outgroup. We sequenced 36 <italic>D. azteca</italic> and 56 <italic>D. falleni</italic>. Both viral types are present in both additional species, but the High type is rare in <italic>D. azteca</italic> (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). The High type has a significantly higher titer than the Low type in both cases (<xref ref-type="fig" rid="fig6">Figure 6A,D</xref>. <italic>azteca</italic> GLM t-value = 6.71, p-value=0.0056, <italic>D. falleni</italic> GLM t-value = 8.12, p-value=0.000371). Viral titer is not significantly different across species for either High or Low type (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, GLM t-value = −1.351, p-value=0.179). We also found the <italic>D. azteca</italic> samples cluster with CH <italic>D. innubila</italic> samples and contain the CH background SNPs (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1C</xref>), suggesting no differentiation in the virus infecting different species. <italic>D. falleni</italic> DiNV, on the other hand, clusters completely separately from the other samples, likely due to its geographic separation. However, the <italic>D. falleni</italic> samples still have a derived cluster of High type virus, suggesting a fourth independent evolution of the High type in Georgia. Despite the lack of divergence between viruses infecting the two species in Arizona, a lower proportion of the <italic>D. azteca</italic> population is infected with DiNV, and the High Type is less common than the Low type DiNV (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). Perhaps, even though the relative differences in titer are preserved between the two species, the Low Type is favored in <italic>D. azteca</italic> because this reduced virulence leads to a greater basic reproductive rate for the virus in <italic>D. azteca</italic>. Thus, the two types of the virus may be maintained in both host species because though they have become specialized to maximize fitness in one host, transmission between host species could lead to their continued presence in both hosts.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Titer and frequency of different DiNV types infecting different species.</title><p>(<bold>A</bold>) Viral titer for CH samples of <italic>D. azteca, D. falleni</italic> and <italic>D. innubila</italic> infected with High and Low type DiNV. The <italic>D.innubila</italic> data presented here is a reconstruction of <xref ref-type="fig" rid="fig1">Figure 1B</xref>. (<bold>B</bold>) Proportion of <italic>D. azteca</italic> 2017, <italic>D. falleni</italic> and <italic>D. innubila</italic> 2017 CH population infected with High and Low type DiNV.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-58931-fig6-v2.tif"/></fig><p>We repeated our association study in <italic>D. azteca, D. falleni,</italic> and each <italic>D. innubila</italic> population separately. In all cases we found the 11 High type SNPs are associated with higher viral titer (GLM t-value &gt;4.28, p-value&gt;0.0001 in all cases). This was not the case for the <italic>Helicase-2</italic> SNP (despite its presence in <italic>D. falleni</italic>) nor any other SNPs shared between populations. After controlling for the High type, we found no other significant DiNV SNPs in <italic>D. azteca</italic> associated with viral titer. For DiNV infecting <italic>D. falleni</italic>, we found 478 significantly associated SNPs (FDR-corrected p-value&lt;0.01), though none of them with as large an effect as the High type associated SNPs.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Viruses are constantly evolving not just to propagate within a host, but also to optimize their infection across hosts. This optimization involves the relationship between their ability to infect an individual and to transmit to others, tempered by the pathogenic effects on infected hosts caused by viral activity (<xref ref-type="bibr" rid="bib60">May and Nowak, 1995</xref>; <xref ref-type="bibr" rid="bib53">Lipsitch et al., 1996</xref>; <xref ref-type="bibr" rid="bib1">Alizon and van Baalen, 2008</xref>). Since the host is also evolving in response to the virus, an evolutionary arms-race often ensues (<xref ref-type="bibr" rid="bib21">Dawkins and Krebs, 1979</xref>; <xref ref-type="bibr" rid="bib43">Kaltz and Shykoff, 1998</xref>; <xref ref-type="bibr" rid="bib19">Daugherty and Malik, 2012</xref>). DNA viruses have large genomes and often recombination, placing them as a somewhat transitionary pathogen between RNA viruses, bacteria and eukaryotic pathogens and parasites. Here, to work towards expanding our understanding of the co-evolution of viruses and their hosts, we examine the population dynamics of <italic>Drosophila</italic> innubila Nudivirus (DiNV), a DNA virus infecting <italic>D. innubila</italic> (<xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>). Within our set of viral samples, we found two DiNV types which differ by 11 SNPs (<xref ref-type="fig" rid="fig1">Figure 1</xref>, named High and Low types). One haplotype (the High type) is associated with higher viral titer, likely due to an increased manipulation of the host-immune system and increased expression of viral factors (<xref ref-type="fig" rid="fig2">Figure 2</xref>).</p><p>The derived High type has likely recurrently evolved in each population since the last glacial maximum (~10,000 years ago). The two types appear to not co-infect individuals, and mutations appear in a similar order as if navigating an epistatic fitness landscape (<xref ref-type="bibr" rid="bib23">Dobzhansky, 1937</xref>; <xref ref-type="bibr" rid="bib47">Kondrashov et al., 2002</xref>; <xref ref-type="bibr" rid="bib30">Gavrilets, 2004</xref>). Finally, despite the higher titer and transmission rate of the High type compared to the Low type, we found the two types in <italic>all</italic> sampled populations. Thus, a fundamental question is: why are there two haplotypes? Possible explanations are (a) that the two haplotypes are neutral and coexist due to genetic drift, (b) we caught High type amid a selective sweep, or (c) the two haplotypes are adaptively maintained due to a tradeoff.</p><p>Given the immense differences in titer and survival between High and Low types and the rest of the preponderance of evidence presented above, we find it implausible that the two haplotypes are associated with equal viral fitness and therefore evolving under a model of genetic drift (<xref ref-type="bibr" rid="bib33">Gillespie, 2004</xref>).</p><p>It is possible we caught the High type DiNV in the middle of a selective sweep in each population. Differing basic reproduction numbers (R<sub>0</sub>) would result in changes in the ratio of types over time (as seen between 2001 and 2017, <xref ref-type="fig" rid="fig1">Figure 1D</xref>). As we only have two time points, we could be witnessing a selective sweep of the High type spreading to fixation (<xref ref-type="bibr" rid="bib65">Nielsen, 2005</xref>), with recombination causing the observed differences in the background. As we found the High type appears to have evolved recurrently, it would be unlikely that we have caught intermediate sweeps in all four populations sampled (<xref ref-type="fig" rid="fig1">Figure 1D</xref>). Using the frequency of the High type between 2001 and 2017 CH samples, we can calculate the selection coefficient for the High type if increasing at an exponential rate (which is likely if mid-selective sweep given the intermediate frequency at both time points). <italic>D. innubila</italic> is active during the monsoon season if Arizona (late July to early September, 10 weeks with a generation time of 2–3 weeks) (<xref ref-type="bibr" rid="bib73">Patterson and Stone, 1949</xref>). If we assume five generations per year (the minimum number of viral generations per year, limited by the host generation time) and an increase among infected individuals from 52% on 2001 to 71% in 2017, the selection coefficient = 0.0055, which suggests the High type would take ~350 years maximum to fix in a population once it has arisen (from a frequency 1/N<sub>e</sub> to one in the viral population). Given the coalescence time of the two types is close to the expansion time of the two viruses (2–30 thousand years), this does not fit with our results, suggesting the two types are more likely to be adaptively maintained and not amid a selective sweep. Further, if the High type was sweeping, it would be remarkable for us to catch these sweeps occurring in all four populations sampled, given the lack of gene flow between populations, our estimated time to fixation of 350 years, and estimated initial infection ~30 thousand years ago.</p><p>Finally, two viral types could be maintained because of a tradeoff between the two virus types. We can imagine several scenarios for such a tradeoff. First, within a single population of genetically identical hosts, two viral types could coexist if they had equal basic reproduction numbers (<italic>R<sub>0</sub></italic>) (<xref ref-type="bibr" rid="bib67">Nowak and May, 1994</xref>; <xref ref-type="bibr" rid="bib60">May and Nowak, 1995</xref>). The basic reproduction number in a simple epidemiology model is the ratio of the transmission rate of the pathogen to its virulence. So, while we’ve observed that virulence is higher for the High type virus compared to the Low type, it may be also true that transmission is also higher in the High type (inferred from <xref ref-type="fig" rid="fig1">Figure 1D</xref>) and those differences are perfectly balanced creating equal <italic>R</italic><sub>0</sub>. We find this unlikely but a useful starting point for considering the adaptive maintenance of the two types. A more likely second scenario is one where in a single population of a given species, different host genotypes favor different viral type and that the interaction between optimal <italic>R<sub>0</sub></italic> among the viral genotypes and susceptibility of the host genotypes creates a stable equilibrium for both host genotypes and viral genotypes (<xref ref-type="bibr" rid="bib4">Anderson and May, 1981</xref>). These host genotypes could also be different host species. In fact, in the Chiricahuas, <italic>D. innubila</italic> was more likely to be infected with the High type virus and <italic>D. azteca</italic> was more likely to be infected with the Low type virus (<xref ref-type="fig" rid="fig6">Figure 6</xref>) even though the two species are sympatric <italic>and</italic> the differences in viral titer between the High and Low types were qualitatively similar. Therefore, <italic>R<sub>0</sub></italic> might be higher for the High type in <italic>D. innubila</italic> and higher for the Low type in <italic>D. azteca</italic> with some transmission between host species. These community dynamics could maintain both viruses in both species. The second broad category of tradeoffs involves environmental heterogeneity, which could also create the conditions for the maintenance of both viral types. Furthermore, differing environmental conditions (such as differences in population densities) could affect the frequencies in each location (<xref ref-type="bibr" rid="bib4">Anderson and May, 1981</xref>). Finally, different viral types could also preferentially infect different tissues, and have different titers due to limitations of the given infected tissues. Previous work in nudiviruses identified nudivurses that infect different tissues including the gut, fat body and ovaries (<xref ref-type="bibr" rid="bib41">Jackson et al., 2005</xref>; <xref ref-type="bibr" rid="bib9">Burand et al., 2012</xref>; <xref ref-type="bibr" rid="bib69">Palmer et al., 2018</xref>). Two DiNV strains infecting different tissues may also help explain the lack of recombination between High and Low types, as viral particles cannot exchange information if they segregate to different tissues. However, this would ignore the fact that we found no evidence of coinfection and viruses infecting two tissues could more easily coinfect.</p><p>It is possible that the High type has not recurrently evolved in each population and instead the two viral types are maintained by selection in the face of ongoing gene conversion. However, this would not explain the shared background variants which are fixed in the High type and population exclusive, or the evidence of recurrent evolution in the assembled phylogeny (<xref ref-type="fig" rid="fig5">Figure 5</xref>). It is entirely possible however that within each population, selection is maintaining the High type due to the fitness deficits of intermediate types, while gene conversion erodes the association of the eleven haplotype SNPs, which is possible given the recombination events found around the significantly associated SNPs (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>).</p><p>Studies of different viruses have found that they can undergo speciation-like events through the accumulation of genetic incompatibilities (<xref ref-type="bibr" rid="bib59">Matsubara and Otsuji, 1978</xref>; <xref ref-type="bibr" rid="bib84">Rokyta and Wichman, 2009</xref>; <xref ref-type="bibr" rid="bib64">Meyer et al., 2016</xref>). In these cases, the two viral types cannot recombine and produce viable viral progeny. This may be occurring in DiNV if infection by one viral type limits the ability of the other type to infect the same cell – akin to prezygotic reproductive isolation (<xref ref-type="bibr" rid="bib16">Coyne and Orr, 2004</xref>). Alternatively, coinfection could happen but generate inviable recombinant particles (<xref ref-type="bibr" rid="bib64">Meyer et al., 2016</xref>) – akin to postzygotic reproductive isolation. For example, the fourth mutation might only be favored when in the presence of the derived allele of the third mutation because the third mutation causes a conformational change in protein A that allows the conformational change in protein B caused by the fourth mutation to produce infective viral particles (<xref ref-type="bibr" rid="bib68">Orr, 1995</xref>). The fact that <italic>19K</italic> is known to form a complex with other PIF proteins and other membrane proteins suggests it could play the central role in the formation of this incompatibility system (<xref ref-type="bibr" rid="bib101">Wang et al., 2007</xref>; <xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>). If incompatibilities are present between types, they are likely in rapidly evolving vital nudivirus genes (<xref ref-type="bibr" rid="bib37">Hill and Unckless, 2017</xref>; <xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). As well as being rapidly evolving across baculoviruses and nudivurses, we found these genes also have high rates of adaptive evolution (<xref ref-type="fig" rid="fig4">Figure 4</xref>) and are associated with SNPs that distinguish the High and Low DiNV strains (<xref ref-type="fig" rid="fig1">Figure 1</xref>), further supporting the idea that the few genes are key to nudivirus replication are targets of host-suppression and locked in arms-races across the entire nudivirus/baculovirus phylogeny (<xref ref-type="bibr" rid="bib37">Hill and Unckless, 2017</xref>).</p><p>DNA viruses such as DiNV have complicated replication cycles and large genomes. This makes them a sort of evolutionary intermediate between RNA viruses (small genomes, high mutation rates) and eukaryotes (large genomes, low mutation rates) and tangential to bacteria and archaea (intermediate genomes, low recombination rates) (<xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>). However, adaptation appears to occur through changes in a few key proteins (<xref ref-type="bibr" rid="bib37">Hill and Unckless, 2017</xref>; <xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). Here we found the evolution of two competing viral types that differ in variants near these few key genes, which cause the strains to differ in virulence and titer. Overall our results suggest that the high mutation rates and extremely high levels of selection can result in the repeated and convergent evolution of novel host-virus interactions. Additionally, we found that these host-virus interactions for large DNA viruses can be much more complicated than previous models suggest (<xref ref-type="bibr" rid="bib24">Dolan et al., 2018</xref>; <xref ref-type="bibr" rid="bib28">Feder et al., 2019</xref>).</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Fly collection, DNA isolation, and sequencing</title><p>In this study, we used previously collected and sequenced <italic>D. innubila</italic> (<xref ref-type="bibr" rid="bib39">Hill and Unckless, 2020</xref>). Briefly we collected these flies across the four mountainous locations in Arizona between the 22nd of August and the 11th of September 2017. Specifically, we collected at the Southwest research station in the Chiricahua mountains (~5400 feet elevation, 31.871 latitude −109.237 longitude), Prescott National Forest (~7900 feet elevation, 34.540 latitude −112.469 longitude), Madera Canyon in the Santa Rita mountains (~4900 feet elevation, 31.729 latitude −110.881 longitude) and Miller Peak in the Huachuca mountains (~5900 feet elevation, 31.632 latitude −110.340 longitude). Baits consisted of store-bought white button mushrooms (<italic>Agaricus bisporus</italic>) placed in large piles about 30 cm in diameter, at least five baits per location. We used a sweep net to collect flies over the baits in either the early morning or late afternoon between one and three days after the bait was set. Flies were sorted by sex and species at the University of Arizona and were flash frozen at −80°C before being shipped on dry ice to the University of Kansas in Lawrence, KS. During these collections we also obtained <italic>D. azteca</italic> which we also sorted by species and sex and flash frozen. <italic>D. falleni</italic> were collected using a similar method in the Smoky Mountains (~6600 feet elevation) in Georgia in 2017 by Kelly Dyer, these flies were then sorted at the University of Georgia in Athens GA and shipped on dry ice to the University of Kansas in Lawrence, KS.</p><p>For collected CH <italic>D. innubila, D. falleni</italic> and <italic>D. azteca</italic>, we attempted to assess the frequency of DiNV infection using PCR, looking for amplification of the viral gene <italic>p47</italic>. Using primers from <xref ref-type="bibr" rid="bib99">Unckless, 2011</xref>, P47F: 5′–<named-content content-type="sequence">TGAAACCAGAATGACATATATAACGC</named-content> and P47R: 5′–<named-content content-type="sequence">TCGGTTTCTCAATTAACTTGATAGC</named-content>. We used the following conditions: 95°C 30 s, 55°C 30 s, 72°C 60 s per cycle for 35 cycles. We chose the cutoff for viral infection based these PCR infection results. We compared the PCR positives with the fold coverage in the genome and found that fly samples with 10x coverage of the viral genome for 90% of the viral genome also had PCR positive results, so we chose this as our cutoff.</p><p>We sorted 343 <italic>D. innubila</italic> flies, 60 DiNV positive <italic>D. falleni</italic> and 40 DiNV positive <italic>D. azteca</italic> which we then homogenized and used to extract DNA using the Qiagen Gentra Puregene Tissue kit (USA Qiagen Inc, Germantown, MD, USA). We prepared a genomic DNA library of these 343 DNA samples using a modified version of the Nextera DNA library prep kit (~350 bp insert size, Illumina Inc, San Diego, CA, USA) meant to conserve reagents. We sequenced the <italic>D. innubila</italic> libraries on two lanes of an Illumina HiSeq 4000 run (150 bp paired-end) (Data to be deposited in the SRA). We sequenced the <italic>D. falleni</italic> and <italic>D. azteca</italic> libraries on a separate run of a lane of an Illumina HiSeq 4000 (150bp paired-end).</p><p>For 80 male <italic>Drosophila innubila</italic> collected in 2018 (indicated in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - Table 2), we split the sample homogenate in half, isolated DNA from half as described above and isolating RNA using the Direct-zol RNA Microprep protocol (R2061, ZymoResearch, Irvine, CA, USA). We then polyA-selected on these samples to isolate mRNA and prepared a cDNA library for each of these 80 RNA samples using a modified version of the Nextera TruSeq library prep kit meant to conserve reagents and sequenced these samples on a NovaSeq NS6K SP 100SE (100 bp single end). We also sequenced DNA for these samples, with DNA isolated and prepared as above, also sequenced on a NovaSeq NS6K SP 100SE (100 bp single end).</p></sec><sec id="s4-2"><title>Sample filtering, mapping, and alignment</title><p>Following sequencing, we removed primer and adapter sequences using cutadapt (<xref ref-type="bibr" rid="bib56">Martin, 2011</xref>) and Scythe (<xref ref-type="bibr" rid="bib6">Buffalo, 2018</xref>) and trimmed all data using Sickle (-t sanger -q 20 l 50) (<xref ref-type="bibr" rid="bib42">Joshi and Fass, 2011</xref>). We masked the <italic>D. innubila</italic> reference genome (<xref ref-type="bibr" rid="bib36">Hill et al., 2019</xref>), using <italic>D. innubila</italic> TE sequences and RepeatMasker (<xref ref-type="bibr" rid="bib92">Smit and Hubley, 2008</xref>; <xref ref-type="bibr" rid="bib93">Smit and Hubley, 2013</xref>). We then mapped short reads to the masked genome and the <italic>Drosophila</italic> innubila Nudivirus genome (DiNV) (<xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>) using BWA MEM (<xref ref-type="bibr" rid="bib52">Li and Durbin, 2009</xref>) and sorted using SAMtools (<xref ref-type="bibr" rid="bib51">Li et al., 2009</xref>). Following this we added read groups, marked and removed sequencing and optical duplicates, and realigned around indels in each mapped BAM file using GATK and <xref ref-type="bibr" rid="bib77">Picard, 2020</xref> (<ext-link ext-link-type="uri" xlink:href="http://broadinstitute.github.io/picard">http://broadinstitute.github.io/picard</ext-link>; <xref ref-type="bibr" rid="bib62">McKenna et al., 2010</xref>; <xref ref-type="bibr" rid="bib22">DePristo et al., 2011</xref>). We considered lines to be infected with DiNV if at least 95% of the viral genome is covered to at least 10-fold coverage. This most significantly overlaps with our CH <italic>D. innubila</italic> PCR results (χ<sup>2</sup> = 71.791, p-value=2.392e-17). Additionally, index-switching at a rate of 0.0001 from the highest titer sample to an uninfected sample would still leave the uninfected with ~5.2-fold coverage of the viral genome, and so still considered as uninfected.</p><p>We then filtered for low coverage and mis-identified species by removing individuals with low coverage of the <italic>D. innubila</italic> genome (less than 5-fold coverage for 80% of the non-repetitive genome), and individuals we suspected of being misidentified as <italic>D. innubila</italic> following collection. This left us with 318 <italic>D. innubila</italic> wild flies with at least 5-fold coverage across at least 80% of the euchromatic genome, of which 254 are infected with DiNV (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - <xref ref-type="table" rid="table1">Table 1</xref>). We also checked for read pairs which were split mapped between the DiNV genome and the <italic>D. innubila</italic> genome using SAMtools.</p><p>For <italic>D. falleni</italic> we used a previously generated <italic>D. innubila</italic> genome with <italic>D. falleni</italic> variants inserted (<xref ref-type="bibr" rid="bib36">Hill et al., 2019</xref>). We masked the genome with Repeatmasker (<xref ref-type="bibr" rid="bib93">Smit and Hubley, 2013</xref>) and mapped short reads to the masked genome, the repeat sequences and the DiNV genome using BWA MEM and SAMtools (<xref ref-type="bibr" rid="bib52">Li and Durbin, 2009</xref>; <xref ref-type="bibr" rid="bib51">Li et al., 2009</xref>). Then, as with <italic>D. innubila</italic> we filtered for low coverage and mis-identified species by removing individuals with low coverage (less than fivefold coverage for 80% of the non-repetitive genome) leaving us with 56 <italic>D. falleni</italic> samples infected with DiNV.</p><p>For <italic>D. azteca</italic>, we downloaded the genome from NCBI (Accession: GCA_005876895.1) which we then called repeats from with RepeatModeler (<xref ref-type="bibr" rid="bib92">Smit and Hubley, 2008</xref>). We masked the genome with Repeatmasker (<xref ref-type="bibr" rid="bib93">Smit and Hubley, 2013</xref>) and mapped short reads to the masked genome, the repeat sequences and the DiNV genome using BWA MEM and SAMtools (<xref ref-type="bibr" rid="bib52">Li and Durbin, 2009</xref>; <xref ref-type="bibr" rid="bib51">Li et al., 2009</xref>). As with <italic>D. innubila</italic> we then filtered for low coverage and mis-identified species by removing individuals with low coverage of the <italic>D. azteca</italic> genome (less than 5-fold coverage for 80% of the non-repetitive genome), which left us with 37 <italic>D. azteca</italic> samples infected with DiNV. We then called DiNV variation using LoFreq (<xref ref-type="bibr" rid="bib105">Wilm et al., 2012</xref>).</p></sec><sec id="s4-3"><title>Calling nucleotide polymorphisms across the population samples</title><p>For the 318 sequenced samples with reasonable coverage, for host polymorphism, we used the previously generated multiple strain VCF file, generated using a standard GATK HaplotypeCaller/BCFTools pipeline. We used LoFreq (<xref ref-type="bibr" rid="bib105">Wilm et al., 2012</xref>) to call polymorphic viral SNPs within each of the 254 DiNV-infected samples, following filtering using BCFtools to remove sites below a quality of 950 and a frequency less than 5%. We then merged each VCF to create a multiple strain VCF file, containing 5,283 SNPs in the DiNV genome. The LoFreq VCF (<xref ref-type="bibr" rid="bib105">Wilm et al., 2012</xref>) output contains estimates of the frequency of each SNP in DiNV in each sample, to confirm these frequencies, in SAMtools (<xref ref-type="bibr" rid="bib51">Li et al., 2009</xref>) we generated mPileups for each sample and for SNPs of interest (related to viral titer), we counted the number of each nucleotide to confirm the estimated frequencies of these nucleotides at each position in each sample. To confirm that there are no coinfections of types, we also subsampled samples and randomly merged Low and High type samples and again generated mPileup files, for SNPs of interest we again counted the number of each nucleotide at each position and confirmed these matched our expected counts in the merged files. We then compared these artificial coinfections to actual samples to confirm the presence or absence of coinfections, finding no samples consistent with coinfections. We then used SNPeff to identify the annotation of each SNP and label synonymous and non-synonymous (<xref ref-type="bibr" rid="bib14">Cingolani et al., 2012</xref>). We extracted the synonymous site frequency spectrum to estimate the effective population size backwards in time using StairwayPlot (<xref ref-type="bibr" rid="bib54">Liu and Fu, 2015</xref>).</p><p>Using the viral VCF and the genetics r package we calculated r<sup>2</sup> measure of linkage disequilibrium. We visualized the linkage between the eleven focal SNPs, and between the eleven focal SNPs and 1000 random SNPs using LDheatmap (<xref ref-type="bibr" rid="bib90">Shin et al., 2006</xref>). We also sorted r<sup>2</sup> scores by the types of SNPs the measure is between, if it is between the focal type SNPs, the SNPs found on each populations High type background, and other SNPs.</p><p>For 100,000 permutations, we randomized the viral titer associated with each strain. For each permutation, we binned individuals into high and low artificial haplotypes based on the allele for the SNP at 126118 (the most significantly associated SNP) and the 10 SNP most strongly linked to this SNP. Finally, we found the difference between the two artificial bins and compared this to the difference between the viral titer of the two haplotypes. We then counted the proportion of the 100,000 permutations with a difference lower than the true difference.</p></sec><sec id="s4-4"><title>Identifying signatures of adaptive evolution using McDonald-Kreitman-based tests</title><p>We filtered the total viral VCF with annotations by SNPeff and retained only non-synonymous (replacement) or synonymous (silent) SNPs. We also mapped reads from Kallithea virus (dS ~ 0.133) and Orcytes rhinoceros Nudivirus (OrNV, dS ~ 0.279) to the DiNV genome to identify the ancestral state at each site and polarize SNPs to specific branches (<xref ref-type="bibr" rid="bib38">Hill and Unckless, 2018</xref>). Specifically, we sought to determine sites which are derived polymorphic in DiNV and which are substitutions fixed on the DiNV branch of the phylogeny. After removing singletons, we used the raw counts of fixed and polymorphic silent and replacement sites per gene to estimate McDonald-Kreitman-based statistics, specifically direction of selection (DoS) (<xref ref-type="bibr" rid="bib61">McDonald and Kreitman, 1991</xref>; <xref ref-type="bibr" rid="bib94">Smith and Eyre-Walker, 2002</xref>; <xref ref-type="bibr" rid="bib98">Stoletzki and Eyre-Walker, 2011</xref>). We also used these values in SnIPRE (<xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>), which reframes McDonald-Kreitman based statistics as a linear model, taking into account the total number of non-synonymous and synonymous mutations occurring in user defined categories to predict the expected number of these substitutions and calculate a selection effect relative to the observed and expected number of mutations (<xref ref-type="bibr" rid="bib26">Eilertson et al., 2012</xref>). We calculated the SnIPRE selection effect for each gene using the total number of mutations on the chromosome of the focal gene.</p><p>For the host, we repeated this process using the SNPeff annotated VCF in the SnIPRE pipeline to identify signatures of adaptation in the host genome.</p><p>We then calculated the difference in each statistic between each gene and the median of all other viral genes, to identify how much that gene deferred from the genomic background. We repeated this for the host genome, limiting the comparison to nearby genes (within 100kbp on the same chromosome).</p></sec><sec id="s4-5"><title>Identifying differentially expressed genes between DiNV-infected and uninfected <italic>Drosophila</italic> innubila</title><p>For 100 male <italic>Drosophila innubila</italic> collected in 2018 (indicated in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> - Table 2), we homogenized each fly separately in 100 µL of PBS. We then split the sample homogenate in half, isolated DNA from half as described above and isolating RNA using the Direct-zol RNA Microprep protocol (R2061). Using the isolated DNA, we tested each sample for DiNV using PCR for <italic>P47</italic> as described previously, using 40 DiNV-infected samples and 40 uninfected samples. We then prepared a cDNA library for each of these 80 RNA samples using a modified version of the Nextera TruSeq library prep kit meant to conserve reagents and sequenced these samples on a NovaSeq NS6K SP 100SE (100 bp single end). We also sequenced DNA for these samples, with DNA isolated and prepared as above, also sequenced on a NovaSeq NS6K SP 100SE (100 bp single end). The DNA sequenced here was mapped as described above, with variation called as described above for other DNA samples.</p><p>Following trimming and filtering the data as described in the methods, we mapped all mRNA sequencing data to a database of rRNA (<xref ref-type="bibr" rid="bib79">Quast et al., 2013</xref>) to remove rRNA contaminants. Then we mapped the short read data to the masked <italic>D. innubila</italic> genome and DiNV genome using GSNAP (-N 1 -o sam) (<xref ref-type="bibr" rid="bib106">Wu and Nacu, 2010</xref>). We estimated counts of reads uniquely mapped to <italic>D. innubila</italic> or DiNV genes using HTSEQ (<xref ref-type="bibr" rid="bib3">Anders et al., 2015</xref>) for each sample. Using EdgeR (<xref ref-type="bibr" rid="bib82">Robinson et al., 2010</xref>) we calculated the counts per million (CPM) of each gene in each sample and counted the number of samples with CPM &gt; 1 for each gene. We find that over 70.3% of genes have a CPM &gt; 1 in at least 70 samples. For the remaining genes, we find these genes are expressed in all samples of a subset of the strains (e.g. DiNV uninfected, DiNV-infected, DiNV-high infected, DiNV-low infected). This supports the validity of the annotation of <italic>D. innubila</italic>, given most genes are expressed in some manner, and suggests our RNA sequencing samples show expression results consistent with the original annotation of the <italic>D. innubila</italic> genome.</p><p>We attempted to improve the annotation of the <italic>D. innubila</italic> genome to find genes expressed only under infection. We extracted reads that mapped to unannotated portions of the genome and combined these for uninfected samples, samples infected with High type DiNV and samples infected with Low type DiNV as three separate samples. We then generated a de novo assembly for each of these three groups using Trinity and Velvet (<xref ref-type="bibr" rid="bib89">Schulz et al., 2012</xref>; <xref ref-type="bibr" rid="bib35">Haas et al., 2013</xref>). We then remapped these assemblies to the genome to identify other transcripts and found the consensus of these two for each sample. Using the Cufflinks pipeline (<xref ref-type="bibr" rid="bib31">Ghosh and Chan, 2016</xref>), we mapped reads to the <italic>D. innubila</italic> genome and counted the number of reads mapping to each of these putative novel transcript regions, identifying 15,676 regions of at least 100 bp, with at least one read mapping in at least one sample. Of these, 717 putative genic regions have at least 1 CPM in all 80 samples, or in all samples of one group (DiNV uninfected, DiNV-infected, DiNV-low infected, DiNV-high infected). We next attempted to identify if any of these genes are differentially expressed between types, specifically between uninfected strains and DiNV-infected strains, and between Low-type infected and High-type infected strains. Using a matrix of CPM for each putative transcript region in each sample, we calculated the extent of differential expression between each type using EdgeR (<xref ref-type="bibr" rid="bib82">Robinson et al., 2010</xref>), after removing regions that are under expressed, normalizing data and estimating the dispersion of expression. We find that 26 putative genes are differentially expressed between infected and uninfected types, and 69 putative genes are differentially expressed between High and Low types. We took these regions and identified any homology to <italic>D. virilis</italic> transcripts using blastn (<xref ref-type="bibr" rid="bib2">Altschul et al., 1990</xref>). We find annotations for 37 putative genes are either expressed in all samples, or differentially expressed between samples. Of the 14 putative genes expressed in all samples, nine have the closest blast hit to an rRNA gene, and five have hits to unknown genes. For 23 differentially expressed putative genes with blast hits, three genes are like antimicrobial peptides (<italic>IM1, IM14, IM3</italic>), these genes are significantly downregulated upon infection, like other Toll-regulated AMPs, and have significantly lower expression in strains infected with High type DiNV compared to Low types. The remaining 20 genes all have similarity to genes associated with cell cycle regulation, actin regulation and tumor suppression genes.</p></sec><sec id="s4-6"><title>Identifying genes associated with viral titer in <italic>Drosophila</italic> innubila</title><p>As the logarithm of viral titer was normally distributed (Shapiro-Wilk test W = 0.05413, p-value=0.342), we used PLINK (<xref ref-type="bibr" rid="bib78">Purcell et al., 2007</xref>) to associate nucleotide polymorphism to logarithm of viral titer in infected samples. We first generated a relationship matrix for the viral samples using PLINK. This kinship matrix allows us control for pairs of individuals who, on average, have similar alleles and which, on average, have dissimilar values. We then fit a linear model in PLINK including population, sex, <italic>Wolbachia</italic> presence, the date of collection and the distance matrix for relationship of each sample (inferred using PLINK, shown as relationship[strain]).</p><p>Before performing the association study, we removed host SNPs found in fewer than five samples and merged SNPs in perfect linkage within 10kbp of each other, filtering down from 5283 viral polymorphisms to 1403 viral polymorphisms. We then identified associations between the logarithm of viral titer and the frequency of the viral polymorphism in each individual sample, resulting in the following model:<disp-formula id="equ1"><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mtable columnalign="left left" columnspacing="1em" rowspacing="4pt"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>v</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mtext> </mml:mtext><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:mtext> </mml:mtext><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mo>+</mml:mo><mml:mi>d</mml:mi><mml:mi>c</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mo>∗</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mtext> </mml:mtext><mml:mo>∗</mml:mo><mml:mtext> </mml:mtext><mml:mi>p</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mo>∗</mml:mo><mml:mi>w</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mtext> </mml:mtext><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Where <italic>hs</italic> = host sex, p=population of collection, <italic>w</italic> = <italic>Wolbachia</italic> presence, <italic>dc</italic> = date collected, relationship[strain]=distance matrix. Following model fitting, performed a posthoc analysis to determine if any clumped SNPs were significant and if they should be assessed separately, we also identified covariates which seemed to show little or no effect on viral titer (p-value&gt;0.1) using an ANOVA in R (<xref ref-type="bibr" rid="bib80">R Development Core Team, 2013</xref>), and removed these, refitting the model. This was done step-wise, resulting in the following model:<disp-formula id="equ2"><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>v</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mtext> </mml:mtext><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:mtext> </mml:mtext><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mtext> </mml:mtext><mml:mo>∗</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mtext> </mml:mtext><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Following this we also performed an association study using PLINK (<xref ref-type="bibr" rid="bib78">Purcell et al., 2007</xref>) in the host, using previously called host variation, and considering viral haplotype as an additional covariate. Before we performed this analysis, we clumped SNPs that are strongly linked (r<sup>2</sup> &gt;0.99) within 10kbp of each other.<disp-formula id="equ3"><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>v</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mtext> </mml:mtext><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:mtext> </mml:mtext><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mtext> </mml:mtext><mml:mo>∗</mml:mo><mml:mi>h</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>h</mml:mi><mml:mo>+</mml:mo><mml:mtext> </mml:mtext><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Whereas previous, and with <italic>vh</italic> = viral haplotype. Using GOrilla (<xref ref-type="bibr" rid="bib25">Eden et al., 2009</xref>), we found no gene categories enriched in the significant SNPs (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>).</p><p>We repeated this analysis for DiNV variants in <italic>D. azteca</italic> and <italic>D. falleni</italic> separately. We performed the association study twice, first using the original model, then including viral haplotype as an additional covariate.</p><p>Because we are performing multiple correlated tests, we next determined the genome-wide significance threshold for the association between a SNP and the viral titer by permutation. We permuted the viral titer information randomly across samples and repeated the genome-wide association studies as described above. We recorded the minimum <italic>p</italic>-value across the genome from 1000 permutations to generate a null distribution. To identify if the SNPs involved in the High type are more likely to appear together than expected, we again used a permutation test to identify the frequency that these combinations occur across 1000 permutations, where a P of 0.01 = 7.987812e-10.</p></sec><sec id="s4-7"><title>Estimating viral titer using qPCR</title><p>Following the identification of the viral haplotype associated with viral titer we sought to determine the effect of viral haplotypes in actual infections. For 20 samples with fly homogenate, we determine the viral titer and haplotype following filtration with a 0.22 µM filter.</p><p>We performed qPCR for the viral gene <italic>p47</italic> (Forward 5-<named-content content-type="sequence">TCGTGCCGCTAAGCATATAG</named-content>-3, Reverse 5-<named-content content-type="sequence">AAAGCTACATCTGTGCGAGG</named-content>-3) on 1 µL of fly filtrate per sample and compared the estimated Cq values across three replicates to estimated viral copy number to confirm viral concentration (protocol: 2 min at 95°C, 40 cycles of 95°C for 30 s and 59°C for 20 s, followed by 2 min at 72°C). Following this we diluted samples to similar Cq values, relative to the sample with the highest Cq value. We confirmed this by repeating qPCR with <italic>p47</italic> primers of 1 µL of each sample.</p><p>For each filtrate sample we performed infections on 30 <italic>D. innubila</italic> males 4–5 days following emergence using pricks with sterile needles dipped in viral filtrate. We recorded survival of each fly each day and removed dead flies. Finally, we took samples 1, 3, and 5 days post infection and measured viral copies of <italic>p47</italic> relative to <italic>tpi</italic> at each time point.</p></sec><sec id="s4-8"><title>Phylogeography of DiNV infection and the evolution of the different viral haplotypes</title><p>For each DiNV-infected <italic>D. innubila</italic> sample, we reconstructed the consensus DiNV genome infecting them using GATK AlternateReferenceMaker and the VCF generated for each strain (<xref ref-type="bibr" rid="bib62">McKenna et al., 2010</xref>; <xref ref-type="bibr" rid="bib22">DePristo et al., 2011</xref>). We used BEAST2 with the SkyLine package (<xref ref-type="bibr" rid="bib97">Stadler et al., 2013</xref>) to build the phylogeny of DiNV genomes using 100 million iterations with a burn in of 1 million, sampling every 1000 trees (<xref ref-type="bibr" rid="bib5">Bouckaert et al., 2014</xref>). After generating a set of phylogenies in BEAST2, we checked that convergence had occurred across the samples using Tracer and generated a final consensus phylogeny using TreeAnnotator (<xref ref-type="bibr" rid="bib5">Bouckaert et al., 2014</xref>). For each High type SNP, we reconstructed the evolution of the SNP as character states across the BEAST2 phylogenies to infer the SNP appearance order. We used the all different rates (ARD) discrete model in APE (<xref ref-type="bibr" rid="bib71">Paradis et al., 2004</xref>) on 1000 randomly sampled converged BEAST2 phylogenies to infer SNP states at each branch and calculated the bootstrap support for the order of appearance for each mutation in each population. To independently identify recurrent mutations across the DiNV phylogeny, we also used TreeTime (<xref ref-type="bibr" rid="bib87">Sagulenko et al., 2018</xref>). Finally, we created a matrix of SNPs present in at least two viral samples and used this matrix in a principal component analysis in R (<xref ref-type="bibr" rid="bib80">R Development Core Team, 2013</xref>), labeling each sample by their viral type in the PCA (based on the presence or absence of the significant <italic>19K</italic> SNP).</p></sec><sec id="s4-9"><title>Simulating the evolution of the high and low viral haplotypes</title><p>We sought to simulate the infection of DiNV in <italic>D. innubila</italic> when considering the evolution of a high titer viral haplotype, specifically if two viral types can be maintained against each other at stable frequencies, and if the high viral haplotype with five shared mutations could evolve recurrently in the given time period given realistic parameters. We used the R package DeSolve (<xref ref-type="bibr" rid="bib95">Soetaert et al., 2010</xref>) to simulate infection dynamics in an SI model. We did not include the resistance class, under the assumption that flies won’t live long enough to clear the infection. Therefore, the proportion of population infected per generation is as follows:<disp-formula id="equ4"><mml:math id="m4"><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:mi>S</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>b</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>*</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>*</mml:mi> <mml:mi/><mml:mi>β</mml:mi></mml:mrow></mml:mfenced><mml:mo>-</mml:mo><mml:mo>(</mml:mo><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>*</mml:mi> <mml:mi/><mml:mi>γ</mml:mi><mml:mo>)</mml:mo> <mml:mi/></mml:math></disp-formula></p><p>Where <inline-formula><mml:math id="inf2"><mml:mi>β</mml:mi></mml:math></inline-formula> = infection parameter (e.g. the increased likelihood an infected individual spreading its infection before dying) and <inline-formula><mml:math id="inf3"><mml:mi>γ</mml:mi></mml:math></inline-formula> = virulence parameter (e.g. the increased likelihood an infected individual has of dying before it can spread its infection). Infected = The proportion of the population infected with DiNV.</p><p>Susceptible = the proportion of the uninfected population.</p><p>We attempted to assess if the ‘high titer’ viral haplotype could possibly evolve recurrently in each population in the time scale seen in our findings. We again used the SI model, this time discrete with population sizes set to 1 million individuals, based on <italic>StairwayPlot</italic> estimates (<sc>Liu and Fu</sc> 2015), starting with 1 infected individual (1/N<sub>e</sub>). We reasoned that if the wait time between beneficial mutations is 1/(2N<sub>e</sub>*u*s), and increasing viral titer also increases N<sub>e</sub>, then the increased titer also decreases the wait time between the appearance of mutations. For each infected individual in the population, we tracked the viral titer and also recorded the presence of absence of five mutations, with each mutation increasing the titer of infection, but each further mutation having successively smaller increases in viral titer (<inline-formula><mml:math id="inf4"><mml:msup><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:msqrt><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mo>.</mml:mo> <mml:mi/><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:msqrt></mml:mrow></mml:msup></mml:math></inline-formula>), representing the epistatic interaction of high titer associated mutations seen in the viral haplotype. Viral titer did not increase if a mutation occurs before the preceding mutation was present occurred, to matching the order of appearance of mutations seen in DiNV. In all simulations, we assumed 5 viral generations per year, based on the assumption of 5 host generations per year being the minimum number of viral generations (likely much higher), and so useful to conservatively estimate the amount of time required for the recurrent evolution of the High type.</p><p>In the simulations, we multiplied the infection parameter, mutation parameter and virulence parameter by viral titer, under the assumption that viral titer increases both infection and death rate, and the mutation rate is per viral particle (<xref ref-type="bibr" rid="bib55">Maeda et al., 1993</xref>; <xref ref-type="bibr" rid="bib44">Kamita et al., 2003</xref>). We considered a per site mutation rate of 10<sup>−6</sup> mutations per generation, based on estimated baculovirus mutation rate (<xref ref-type="bibr" rid="bib83">Rohrmann, 2013</xref>; <xref ref-type="bibr" rid="bib13">Chateigner et al., 2015</xref>), with a specific mutation rate for each of the high titer variants of 3.3e-7 (1e-6 mutations per base per generation * 1/3 for the correct mutation) * viral titer, where each mutation occurs independently (so all five could arise in one generation at a rate of (3.3e-7)<sup>5</sup> = 4.12e-33). We then simulated populations in replicate 1000 times for 100,000 generations with a starting infection frequency of 10% for the ‘low titer’ haplotype, recording the frequency of the virus in a population, the frequency of the haplotype and the time that each ‘high titer’ mutation reaches high enough frequency to escape stochastic behavior and behave deterministically under selection (<xref ref-type="bibr" rid="bib33">Gillespie, 2004</xref>). We also factored in recombination between each variant site a rate of 2% per site-window per generation (<xref ref-type="bibr" rid="bib44">Kamita et al., 2003</xref>). Specifically, we randomly paired viral genomes and in 2% of pairs we randomly recombined the variant combinations. We did not increase recombination rate as titer increased as this was not observed in nature (<xref ref-type="bibr" rid="bib44">Kamita et al., 2003</xref>).</p><p>To estimate the possible selection coefficient for DiNV in the CH population, we assumed an exponential distribution and five host generations per year (80 host generations between 2001 and 2017). We then solved the following equation:<disp-formula id="equ5"><mml:math id="m5"><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2017</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2001</mml:mn></mml:mrow></mml:msub><mml:mi>*</mml:mi><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msup></mml:math></disp-formula></p><p>Where P<sub>2017</sub> = the frequency of the High type among viral samples in 2017 (71%), P<sub>2001</sub> = the frequency of the High type among viral samples in 2001 (52%), s = the selection coefficient and t = the number of generations (80). We then used this estimated selection coefficient in the same equation to find the number of generations (t) to go from 1/2N<sub>e</sub>s (0.0000714, assuming an N<sub>e</sub> of 1000000) to fixation (0.99):<disp-formula id="equ6"><mml:math id="m6"><mml:mn>0.99</mml:mn><mml:mo>=</mml:mo><mml:mn>0.0000714</mml:mn><mml:mi>*</mml:mi><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mn>0.0055</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msup></mml:math></disp-formula></p></sec><sec id="s4-10"><title>Experimental infections of <italic>Drosophila</italic> innubila with DiNV</title><p>We chose <italic>D. innubila</italic> samples infected with DiNV and with sequenced genomes, four infected with the High type DiNV and four infected with the Low type. For these samples we estimated their viral copy number per host genome as described previously. We used qPCR on <italic>p47</italic> and <italic>tpi</italic> to find the differences in Cq values to calculate the concentration of each sample relative to the lowest concentration sample and diluted 50 µL of filtrate for each sample to match the concentration of each sample to the samples with the lowest titer (IPR07). For a separate 50 µL of the IPR01 sample, we performed 1 in 10 serial dilutions to give 45 µL of filtrate at full concentration, 1 in 10 concentration, 1 in 100 concentration and 1 in 1000 concentration. Using these sets of samples (matched titer and serial dilutions) we next performed experimental infections.</p><p>We transferred 50 <italic>D. innubila</italic> (of roughly equal sex ratio) to new food and let them lay eggs for 1 week, following this we collected male offspring aged 2–5 days for experimental infections.</p><p>Across four separate days in the mornings (between 9am and 11am), we infected the collected male flies with each sample. For flies in batches of 10, we performed pricks with microneedles dipped in the prepared viral filtrate. For each day we also had two control replicates of 10 flies pricked with microneedles dipped in sterile media. Following infections, we checked on each vial of 10 flies one-hour post infection and removed dead flies (likely killed by the needle instead of the virus). We also checked each vial each morning for 15 days, removing dead flies (freezing to determine the viral titer), and flipping flies to new food every 3–4 days. Checking at 10am each day, we recorded the day that each fly died, what filtrate they had been infected with, and what replicate/infection day set they belonged to. We next looked for differences in survival over time compared to sterile wound controls using a Fit proportional hazards regression model in R (<xref ref-type="bibr" rid="bib80">R Development Core Team, 2013</xref>; <xref ref-type="bibr" rid="bib45">Kassambara et al., 2017</xref>), considering titer, viral isolate (nested in haplotype) and replicate/vial as co-variates (day of death ~ [titer or haplotype | strain] + infection date).</p><p>For a second set of experimental infections (performed as described above, stabbed with diluted filtrate from different strains), we also removed three living flies 1 hr, 1 day, and 5 days post infection. Using qPCR, we found the difference in <italic>p47</italic> log-Cq and <italic>tpi</italic> log-Cq to estimate the viral copy number for each sample over time.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>This work was completed with helpful discussion from Justin Blumenstiel, Joanne Chapman, John Kelly, Stuart MacDonald, Andrew Mongue and Carolyn Wessinger. We would especially like to thank Maria Orive, Kelly Dyer and Paul Ginsberg for helpful feedback in the writing of the manuscript and framing of the discussion. Collections were completed with assistance from Todd Schlenke, Paul Ginsberg, Kelly Dyer, Brandon Cooper, John Jaenike and the Southwest Research Station. We thank Brittny Smith and Jenny Hackett at the KU CMADP Genome Sequencing Core (NIH Grant P20 GM103638) and K-INBRE Bioinformatics Core for assistance in genome isolation, library preparation, sequencing and computational resources. This work was supported by a K-INBRE postdoctoral grant to TH (NIH Grant P20 GM103418). This work was also funded by NIH Grants R00 GM114714 and R01 AI139154 to RLU. <italic>D. falleni</italic> collection was funded by NSF grant DEB-1737824. Supplementary Data 1-9 is available in the following data dryad folder: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5061/dryad.2fqz612mh">https://doi.org/10.5061/dryad.2fqz612mh</ext-link>. (doi:10.5061/dryad.2fqz612mh). Additional data regarding <italic>D. innubila</italic> population genomics is available in the following FigShare folder: <ext-link ext-link-type="uri" xlink:href="https://figshare.com/projects/innubila_population_genomics/87662">https://figshare.com/projects/innubila_population_genomics/87662</ext-link>.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Funding acquisition, Investigation, Visualization, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Supervision, Validation, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Next-generation sequencing information of Drosophila infected with DiNV used in this survey.</title><p>Table 1: Summary of <italic>Drosophila innubila</italic> and <italic>D. azteca</italic> fly samples collected and sequenced for this study, table includes summary of coverage for X chromosome,, Muller B, other autosomes, virus and <italic>Wolbachia. </italic>Also contains SRA accessions for each strain. Table 2: Summary of <italic>Drosophila</italic> innubila fly RNA and DNA collected and sequenced for this study, including if infected with DiNV.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-58931-supp1-v2.xlsx"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-58931-transrepform-v2.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>Sequencing data have been deposited on the NCBI SRA under the study accession: SRP187240 Genomes used in this study are available at the following accessions: <italic>Drosophila innubila</italic> - GCF_004354385.1 <italic>Drosophila</italic> innubila Nudivirus - GCF_004132165.1 <italic>Drosophila</italic> azteca - GCA_005876895.1.</p><p>The following datasets were generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Hill</surname><given-names>T</given-names></name><name><surname>Unckless</surname><given-names>RL</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>Drosophila Sky Island sequencing</data-title><source>NCBI Sequence Read Archive</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra/SRX5449484[accn]">SRP187240</pub-id></element-citation></p><p><element-citation id="dataset2" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Hill</surname><given-names>T</given-names></name><name><surname>Unckless</surname><given-names>RL</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>Drosophila Sky Island data analysis</data-title><source>Dryad Digital Repository</source><pub-id assigning-authority="Dryad" pub-id-type="doi">10.5061/dryad.2fqz612mh</pub-id></element-citation></p><p><element-citation id="dataset3" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Hill</surname><given-names>T</given-names></name><name><surname>Unckless</surname><given-names>RL</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>Innubila population genomics</data-title><source>figshare</source><pub-id assigning-authority="figshare" pub-id-type="accession" xlink:href="https://figshare.com/projects/innubila_population_genomics/87662">innubila_population_genomics/87662</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alizon</surname> <given-names>S</given-names></name><name><surname>van Baalen</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Transmission-virulence trade-offs in vector-borne diseases</article-title><source>Theoretical Population Biology</source><volume>74</volume><fpage>6</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1016/j.tpb.2008.04.003</pub-id><pub-id pub-id-type="pmid">18508101</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Altschul</surname> <given-names>SF</given-names></name><name><surname>Gish</surname> <given-names>W</given-names></name><name><surname>Miller</surname> <given-names>W</given-names></name><name><surname>Myers</surname> <given-names>EW</given-names></name><name><surname>Lipman</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="1990">1990</year><article-title>Basic local alignment search tool</article-title><source>Journal of Molecular Biology</source><volume>215</volume><fpage>403</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1016/S0022-2836(05)80360-2</pub-id><pub-id pub-id-type="pmid">2231712</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Anders</surname> <given-names>S</given-names></name><name><surname>Pyl</surname> <given-names>PT</given-names></name><name><surname>Huber</surname> <given-names>W</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>HTSeq--a Python framework to work with high-throughput sequencing data</article-title><source>Bioinformatics</source><volume>31</volume><fpage>166</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btu638</pub-id><pub-id pub-id-type="pmid">25260700</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>RM</given-names></name><name><surname>May</surname> <given-names>RM</given-names></name></person-group><year iso-8601-date="1981">1981</year><article-title>The population dynamics of microparasites and their invertebrate hosts</article-title><source>Philosophical Transactions of the Royal Society of London</source><volume>291</volume><fpage>451</fpage><lpage>524</lpage><pub-id pub-id-type="doi">10.1098/rstb.1981.0005</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bouckaert</surname> <given-names>R</given-names></name><name><surname>Heled</surname> <given-names>J</given-names></name><name><surname>Kühnert</surname> <given-names>D</given-names></name><name><surname>Vaughan</surname> <given-names>T</given-names></name><name><surname>Wu</surname> <given-names>CH</given-names></name><name><surname>Xie</surname> <given-names>D</given-names></name><name><surname>Suchard</surname> <given-names>MA</given-names></name><name><surname>Rambaut</surname> <given-names>A</given-names></name><name><surname>Drummond</surname> <given-names>AJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>BEAST 2: a software platform for bayesian evolutionary analysis</article-title><source>PLOS Computational Biology</source><volume>10</volume><elocation-id>e1003537</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1003537</pub-id><pub-id pub-id-type="pmid">24722319</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Buffalo</surname> <given-names>V</given-names></name></person-group><year iso-8601-date="2018">2018</year><source>Scythe</source></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bull</surname> <given-names>JJ</given-names></name><name><surname>Badgett</surname> <given-names>MR</given-names></name><name><surname>Wichman</surname> <given-names>HA</given-names></name><name><surname>Huelsenbeck</surname> <given-names>JP</given-names></name><name><surname>Hillis</surname> <given-names>DM</given-names></name><name><surname>Gulati</surname> <given-names>A</given-names></name><name><surname>Ho</surname> <given-names>C</given-names></name><name><surname>Molineux</surname> <given-names>IJ</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Exceptional convergent evolution in a virus</article-title><source>Genetics</source><volume>147</volume><fpage>1497</fpage><lpage>1507</lpage></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bull</surname> <given-names>JJ</given-names></name><name><surname>Heineman</surname> <given-names>RH</given-names></name><name><surname>Wilke</surname> <given-names>CO</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The phenotype-fitness map in experimental evolution of phages</article-title><source>PLOS ONE</source><volume>6</volume><elocation-id>e27796</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0027796</pub-id><pub-id pub-id-type="pmid">22132144</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burand</surname> <given-names>JP</given-names></name><name><surname>Kim</surname> <given-names>W</given-names></name><name><surname>Afonso</surname> <given-names>CL</given-names></name><name><surname>Tulman</surname> <given-names>ER</given-names></name><name><surname>Kutish</surname> <given-names>GF</given-names></name><name><surname>Lu</surname> <given-names>Z</given-names></name><name><surname>Rock</surname> <given-names>DL</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Analysis of the genome of the sexually transmitted insect virus Helicoverpa zea Nudivirus 2</article-title><source>Viruses</source><volume>4</volume><fpage>28</fpage><lpage>61</lpage><pub-id pub-id-type="doi">10.3390/v4010028</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burgyán</surname> <given-names>J</given-names></name><name><surname>Havelda</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Viral suppressors of RNA silencing</article-title><source>Trends in Plant Science</source><volume>16</volume><fpage>265</fpage><lpage>272</lpage><pub-id pub-id-type="doi">10.1016/j.tplants.2011.02.010</pub-id><pub-id pub-id-type="pmid">21439890</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Burt</surname> <given-names>A</given-names></name><name><surname>Trivers</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2006">2006</year><source>Genes in Conflict</source><publisher-name>Harvard University Press</publisher-name></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Casino</surname> <given-names>C</given-names></name><name><surname>McAllister</surname> <given-names>J</given-names></name><name><surname>Davidson</surname> <given-names>F</given-names></name><name><surname>Power</surname> <given-names>J</given-names></name><name><surname>Lawlor</surname> <given-names>E</given-names></name><name><surname>Yap</surname> <given-names>PL</given-names></name><name><surname>Simmonds</surname> <given-names>P</given-names></name><name><surname>Smith</surname> <given-names>DB</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Variation of hepatitis C virus following serial transmission: multiple mechanisms of diversification of the hypervariable region and evidence for convergent genome evolution</article-title><source>Journal of General Virology</source><volume>80</volume><fpage>717</fpage><lpage>725</lpage><pub-id pub-id-type="doi">10.1099/0022-1317-80-3-717</pub-id><pub-id pub-id-type="pmid">10092012</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chateigner</surname> <given-names>A</given-names></name><name><surname>Bézier</surname> <given-names>A</given-names></name><name><surname>Labrousse</surname> <given-names>C</given-names></name><name><surname>Jiolle</surname> <given-names>D</given-names></name><name><surname>Barbe</surname> <given-names>V</given-names></name><name><surname>Herniou</surname> <given-names>EA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Ultra deep sequencing of a baculovirus population reveals widespread genomic variations</article-title><source>Viruses</source><volume>7</volume><fpage>3625</fpage><lpage>3646</lpage><pub-id pub-id-type="doi">10.3390/v7072788</pub-id><pub-id pub-id-type="pmid">26198241</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cingolani</surname> <given-names>P</given-names></name><name><surname>Platts</surname> <given-names>A</given-names></name><name><surname>Wang</surname> <given-names>leL</given-names></name><name><surname>Coon</surname> <given-names>M</given-names></name><name><surname>Nguyen</surname> <given-names>T</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Land</surname> <given-names>SJ</given-names></name><name><surname>Lu</surname> <given-names>X</given-names></name><name><surname>Ruden</surname> <given-names>DM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A program for annotating and predicting the effects of single Nucleotide Polymorphisms, SnpEff: snps in the genome of <italic>Drosophila melanogaster</italic> strain w1118; iso-2; iso-3</article-title><source>Fly</source><volume>6</volume><fpage>80</fpage><lpage>92</lpage><pub-id pub-id-type="doi">10.4161/fly.19695</pub-id><pub-id pub-id-type="pmid">22728672</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Costa</surname> <given-names>A</given-names></name><name><surname>Jan</surname> <given-names>E</given-names></name><name><surname>Sarnow</surname> <given-names>P</given-names></name><name><surname>Schneider</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The imd pathway is involved in antiviral immune responses in <italic>Drosophila</italic></article-title><source>PLOS ONE</source><volume>4</volume><elocation-id>e7436</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0007436</pub-id><pub-id pub-id-type="pmid">19829691</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Coyne</surname> <given-names>J</given-names></name><name><surname>Orr</surname> <given-names>HA</given-names></name></person-group><year iso-8601-date="2004">2004</year><source>Speciation</source></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Crandall</surname> <given-names>KA</given-names></name><name><surname>Kelsey</surname> <given-names>CR</given-names></name><name><surname>Imamichi</surname> <given-names>H</given-names></name><name><surname>Lane</surname> <given-names>HC</given-names></name><name><surname>Salzman</surname> <given-names>NP</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Parallel evolution of drug resistance in HIV: failure of nonsynonymous/synonymous substitution rate ratio to detect selection</article-title><source>Molecular Biology and Evolution</source><volume>16</volume><fpage>372</fpage><lpage>382</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a026118</pub-id><pub-id pub-id-type="pmid">10331263</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cuevas</surname> <given-names>JM</given-names></name><name><surname>Moya</surname> <given-names>A</given-names></name><name><surname>Elena</surname> <given-names>SF</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Evolution of RNA virus in spatially structured heterogeneous environments</article-title><source>Journal of Evolutionary Biology</source><volume>16</volume><fpage>456</fpage><lpage>466</lpage><pub-id pub-id-type="doi">10.1046/j.1420-9101.2003.00547.x</pub-id><pub-id pub-id-type="pmid">14635845</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daugherty</surname> <given-names>MD</given-names></name><name><surname>Malik</surname> <given-names>HS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Rules of engagement: molecular insights from host-virus arms races</article-title><source>Annual Review of Genetics</source><volume>46</volume><fpage>677</fpage><lpage>700</lpage><pub-id pub-id-type="doi">10.1146/annurev-genet-110711-155522</pub-id><pub-id pub-id-type="pmid">23145935</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davey</surname> <given-names>NE</given-names></name><name><surname>Travé</surname> <given-names>G</given-names></name><name><surname>Gibson</surname> <given-names>TJ</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>How viruses hijack cell regulation</article-title><source>Trends in Biochemical Sciences</source><volume>36</volume><fpage>159</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1016/j.tibs.2010.10.002</pub-id><pub-id pub-id-type="pmid">21146412</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dawkins</surname> <given-names>R</given-names></name><name><surname>Krebs</surname> <given-names>JR</given-names></name></person-group><year iso-8601-date="1979">1979</year><article-title>Arms races between and within species</article-title><source>Proceedings of the Royal Society of London B: Biological Sciences</source><volume>205</volume><fpage>489</fpage><lpage>511</lpage><pub-id pub-id-type="doi">10.1098/rspb.1979.0081</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>DePristo</surname> <given-names>MA</given-names></name><name><surname>Banks</surname> <given-names>E</given-names></name><name><surname>Poplin</surname> <given-names>R</given-names></name><name><surname>Garimella</surname> <given-names>KV</given-names></name><name><surname>Maguire</surname> <given-names>JR</given-names></name><name><surname>Hartl</surname> <given-names>C</given-names></name><name><surname>Philippakis</surname> <given-names>AA</given-names></name><name><surname>del Angel</surname> <given-names>G</given-names></name><name><surname>Rivas</surname> <given-names>MA</given-names></name><name><surname>Hanna</surname> <given-names>M</given-names></name><name><surname>McKenna</surname> <given-names>A</given-names></name><name><surname>Fennell</surname> <given-names>TJ</given-names></name><name><surname>Kernytsky</surname> <given-names>AM</given-names></name><name><surname>Sivachenko</surname> <given-names>AY</given-names></name><name><surname>Cibulskis</surname> <given-names>K</given-names></name><name><surname>Gabriel</surname> <given-names>SB</given-names></name><name><surname>Altshuler</surname> <given-names>D</given-names></name><name><surname>Daly</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A framework for variation discovery and genotyping using next-generation DNA sequencing data</article-title><source>Nature Genetics</source><volume>43</volume><fpage>491</fpage><lpage>498</lpage><pub-id pub-id-type="doi">10.1038/ng.806</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Dobzhansky</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="1937">1937</year><source>Genetics and the origin of species</source></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dolan</surname> <given-names>PT</given-names></name><name><surname>Whitfield</surname> <given-names>ZJ</given-names></name><name><surname>Andino</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mechanisms and concepts in RNA virus population dynamics and evolution</article-title><source>Annual Review of Virology</source><volume>5</volume><fpage>69</fpage><lpage>92</lpage><pub-id pub-id-type="doi">10.1146/annurev-virology-101416-041718</pub-id><pub-id pub-id-type="pmid">30048219</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eden</surname> <given-names>E</given-names></name><name><surname>Navon</surname> <given-names>R</given-names></name><name><surname>Steinfeld</surname> <given-names>I</given-names></name><name><surname>Lipson</surname> <given-names>D</given-names></name><name><surname>Yakhini</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>GOrilla: a tool for discovery and visualization of enriched GO terms in ranked gene lists</article-title><source>BMC Bioinformatics</source><volume>10</volume><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1186/1471-2105-10-48</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eilertson</surname> <given-names>KE</given-names></name><name><surname>Booth</surname> <given-names>JG</given-names></name><name><surname>Bustamante</surname> <given-names>CD</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>SnIPRE: selection inference using a poisson random effects model</article-title><source>PLOS Computational Biology</source><volume>8</volume><elocation-id>e1002806</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1002806</pub-id><pub-id pub-id-type="pmid">23236270</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Enard</surname> <given-names>D</given-names></name><name><surname>Cai</surname> <given-names>L</given-names></name><name><surname>Gwennap</surname> <given-names>C</given-names></name><name><surname>Petrov</surname> <given-names>DA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Viruses are a dominant driver of protein adaptation in mammals</article-title><source>eLife</source><volume>5</volume><elocation-id>e12469</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.12469</pub-id><pub-id pub-id-type="pmid">27187613</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feder</surname> <given-names>AF</given-names></name><name><surname>Pennings</surname> <given-names>PS</given-names></name><name><surname>Hermisson</surname> <given-names>J</given-names></name><name><surname>Petrov</surname> <given-names>DA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Evolutionary dynamics in structured populations under strong population genetic forces</article-title><source>G3: Genes, Genomes, Genetics</source><volume>9</volume><fpage>3395</fpage><lpage>3407</lpage><pub-id pub-id-type="doi">10.1534/g3.119.400605</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ferreira</surname> <given-names>ÁG</given-names></name><name><surname>Naylor</surname> <given-names>H</given-names></name><name><surname>Esteves</surname> <given-names>SS</given-names></name><name><surname>Pais</surname> <given-names>IS</given-names></name><name><surname>Martins</surname> <given-names>NE</given-names></name><name><surname>Teixeira</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The Toll-Dorsal pathway is required for resistance to viral oral infection in <italic>Drosophila</italic></article-title><source>PLOS Pathogens</source><volume>10</volume><elocation-id>e1004507</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1004507</pub-id><pub-id pub-id-type="pmid">25473839</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Gavrilets</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2004">2004</year><source>Fitness Landscapes and the Origin of Species</source><publisher-name>Princeton University Press</publisher-name></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>S</given-names></name><name><surname>Chan</surname> <given-names>CK</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Analysis of RNA-Seq data using TopHat and cufflinks</article-title><source>Methods Mol Biol</source><volume>1374</volume><fpage>339</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1007/978-1-4939-3167-5_18</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gifford</surname> <given-names>RJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Viral evolution in deep time: lentiviruses and mammals</article-title><source>Trends in Genetics</source><volume>28</volume><fpage>89</fpage><lpage>100</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2011.11.003</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Gillespie</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2004">2004</year><source>Population Genetics: A Concise Guide</source></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grubaugh</surname> <given-names>ND</given-names></name><name><surname>Smith</surname> <given-names>DR</given-names></name><name><surname>Brackney</surname> <given-names>DE</given-names></name><name><surname>Bosco-Lauth</surname> <given-names>AM</given-names></name><name><surname>Fauver</surname> <given-names>JR</given-names></name><name><surname>Campbell</surname> <given-names>CL</given-names></name><name><surname>Felix</surname> <given-names>TA</given-names></name><name><surname>Romo</surname> <given-names>H</given-names></name><name><surname>Duggal</surname> <given-names>NK</given-names></name><name><surname>Dietrich</surname> <given-names>EA</given-names></name><name><surname>Eike</surname> <given-names>T</given-names></name><name><surname>Beane</surname> <given-names>JE</given-names></name><name><surname>Bowen</surname> <given-names>RA</given-names></name><name><surname>Black</surname> <given-names>WC</given-names></name><name><surname>Brault</surname> <given-names>AC</given-names></name><name><surname>Ebel</surname> <given-names>GD</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Experimental evolution of an RNA virus in wild birds: evidence for host-dependent impacts on population structure and competitive fitness</article-title><source>PLOS Pathogens</source><volume>11</volume><elocation-id>e1004874</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1004874</pub-id><pub-id pub-id-type="pmid">25993022</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haas</surname> <given-names>BJ</given-names></name><name><surname>Papanicolaou</surname> <given-names>A</given-names></name><name><surname>Yassour</surname> <given-names>M</given-names></name><name><surname>Grabherr</surname> <given-names>M</given-names></name><name><surname>Blood</surname> <given-names>PD</given-names></name><name><surname>Bowden</surname> <given-names>J</given-names></name><name><surname>Couger</surname> <given-names>MB</given-names></name><name><surname>Eccles</surname> <given-names>D</given-names></name><name><surname>Li</surname> <given-names>B</given-names></name><name><surname>Lieber</surname> <given-names>M</given-names></name><name><surname>MacManes</surname> <given-names>MD</given-names></name><name><surname>Ott</surname> <given-names>M</given-names></name><name><surname>Orvis</surname> <given-names>J</given-names></name><name><surname>Pochet</surname> <given-names>N</given-names></name><name><surname>Strozzi</surname> <given-names>F</given-names></name><name><surname>Weeks</surname> <given-names>N</given-names></name><name><surname>Westerman</surname> <given-names>R</given-names></name><name><surname>William</surname> <given-names>T</given-names></name><name><surname>Dewey</surname> <given-names>CN</given-names></name><name><surname>Henschel</surname> <given-names>R</given-names></name><name><surname>LeDuc</surname> <given-names>RD</given-names></name><name><surname>Friedman</surname> <given-names>N</given-names></name><name><surname>Regev</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>De novo transcript sequence reconstruction from RNA-seq using the trinity platform for reference generation and analysis</article-title><source>Nature Protocols</source><volume>8</volume><fpage>1494</fpage><lpage>1512</lpage><pub-id pub-id-type="doi">10.1038/nprot.2013.084</pub-id><pub-id pub-id-type="pmid">23845962</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>T</given-names></name><name><surname>Koseva</surname> <given-names>BS</given-names></name><name><surname>Unckless</surname> <given-names>RL</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The genome of <italic>Drosophila innubila</italic> Reveals Lineage-Specific Patterns of Selection in Immune Genes</article-title><source>Molecular Biology and Evolution</source><volume>36</volume><fpage>1405</fpage><lpage>1417</lpage><pub-id pub-id-type="doi">10.1093/molbev/msz059</pub-id><pub-id pub-id-type="pmid">30865231</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>T</given-names></name><name><surname>Unckless</surname> <given-names>RL</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Baculovirus molecular evolution via gene turnover and recurrent positive selection of key genes</article-title><source>Journal of Virology</source><volume>91</volume><elocation-id>e01319-17</elocation-id><pub-id pub-id-type="doi">10.1128/JVI.01319-17</pub-id><pub-id pub-id-type="pmid">28814516</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>T</given-names></name><name><surname>Unckless</surname> <given-names>RL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The dynamic evolution of <italic>Drosophila</italic> innubila Nudivirus</article-title><source>Infection, Genetics and Evolution</source><volume>57</volume><fpage>151</fpage><lpage>157</lpage><pub-id pub-id-type="doi">10.1016/j.meegid.2017.11.013</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>T</given-names></name><name><surname>Unckless</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Selection and demography shape genomic variation in a ‘Sky Island’ species</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.05.14.096008</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Holmes</surname> <given-names>EC</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Viral evolution in the genomic age</article-title><source>PLOS Biology</source><volume>5</volume><elocation-id>e278</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0050278</pub-id><pub-id pub-id-type="pmid">17914905</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jackson</surname> <given-names>TA</given-names></name><name><surname>Crawford</surname> <given-names>AM</given-names></name><name><surname>Glare</surname> <given-names>TR</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Oryctes virus--time for a new look at a useful biocontrol agent</article-title><source>Journal of Invertebrate Pathology</source><volume>89</volume><fpage>91</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.1016/j.jip.2005.03.009</pub-id><pub-id pub-id-type="pmid">16039310</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Joshi</surname> <given-names>N</given-names></name><name><surname>Fass</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2011">2011</year><source>Sickle: A sliding window, adaptive, quality-based trimming tool for fastQ files</source></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaltz</surname> <given-names>O</given-names></name><name><surname>Shykoff</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Local adaptation in host–parasite systems</article-title><source>Heredity</source><volume>81</volume><fpage>361</fpage><lpage>370</lpage><pub-id pub-id-type="doi">10.1046/j.1365-2540.1998.00435.x</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kamita</surname> <given-names>SG</given-names></name><name><surname>Maeda</surname> <given-names>S</given-names></name><name><surname>Hammock</surname> <given-names>BD</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>High-frequency homologous recombination between baculoviruses involves DNA replication</article-title><source>Journal of Virology</source><volume>77</volume><fpage>13053</fpage><lpage>13061</lpage><pub-id pub-id-type="doi">10.1128/JVI.77.24.13053-13061.2003</pub-id><pub-id pub-id-type="pmid">14645562</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Kassambara</surname> <given-names>A</given-names></name><name><surname>Kosinski</surname> <given-names>M</given-names></name><name><surname>Biecek</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2017">2017</year><source>Survminer: Drawing Survival Curves Using'ggplot2'</source></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kelly</surname> <given-names>DC</given-names></name></person-group><year iso-8601-date="1982">1982</year><article-title>Baculovirus Replication</article-title><source>Journal of General Virology</source><volume>63</volume><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1099/0022-1317-63-1-1</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kondrashov</surname> <given-names>AS</given-names></name><name><surname>Sunyaev</surname> <given-names>S</given-names></name><name><surname>Kondrashov</surname> <given-names>FA</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Dobzhansky-Muller incompatibilities in protein evolution</article-title><source>PNAS</source><volume>99</volume><fpage>14878</fpage><lpage>14883</lpage><pub-id pub-id-type="doi">10.1073/pnas.232565499</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kosakovsky Pond</surname> <given-names>SL</given-names></name><name><surname>Posada</surname> <given-names>D</given-names></name><name><surname>Gravenor</surname> <given-names>MB</given-names></name><name><surname>Woelk</surname> <given-names>CH</given-names></name><name><surname>Frost</surname> <given-names>SD</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>GARD: a genetic algorithm for recombination detection</article-title><source>Bioinformatics</source><volume>22</volume><fpage>3096</fpage><lpage>3098</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btl474</pub-id><pub-id pub-id-type="pmid">17110367</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lamiable</surname> <given-names>O</given-names></name><name><surname>Kellenberger</surname> <given-names>C</given-names></name><name><surname>Kemp</surname> <given-names>C</given-names></name><name><surname>Troxler</surname> <given-names>L</given-names></name><name><surname>Pelte</surname> <given-names>N</given-names></name><name><surname>Boutros</surname> <given-names>M</given-names></name><name><surname>Marques</surname> <given-names>JT</given-names></name><name><surname>Daeffler</surname> <given-names>L</given-names></name><name><surname>Hoffmann</surname> <given-names>JA</given-names></name><name><surname>Roussel</surname> <given-names>A</given-names></name><name><surname>Imler</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Cytokine diedel and a viral homologue suppress the IMD pathway in <italic>Drosophila</italic></article-title><source>PNAS</source><volume>113</volume><fpage>698</fpage><lpage>703</lpage><pub-id pub-id-type="doi">10.1073/pnas.1516122113</pub-id><pub-id pub-id-type="pmid">26739560</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>SH</given-names></name><name><surname>Quarles</surname> <given-names>KA</given-names></name><name><surname>Yang</surname> <given-names>Y</given-names></name><name><surname>Tanguy</surname> <given-names>M</given-names></name><name><surname>Frézal</surname> <given-names>L</given-names></name><name><surname>Smith</surname> <given-names>SA</given-names></name><name><surname>Sharma</surname> <given-names>PP</given-names></name><name><surname>Cordaux</surname> <given-names>R</given-names></name><name><surname>Gilbert</surname> <given-names>C</given-names></name><name><surname>Giraud</surname> <given-names>I</given-names></name><name><surname>Collins</surname> <given-names>DH</given-names></name><name><surname>Zamore</surname> <given-names>PD</given-names></name><name><surname>Miska</surname> <given-names>EA</given-names></name><name><surname>Sarkies</surname> <given-names>P</given-names></name><name><surname>Jiggins</surname> <given-names>FM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Pan-arthropod analysis reveals somatic piRNAs as an ancestral defence against transposable elements</article-title><source>Nature Ecology &amp; Evolution</source><volume>2</volume><fpage>174</fpage><lpage>181</lpage><pub-id pub-id-type="doi">10.1038/s41559-017-0403-4</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Handsaker</surname> <given-names>B</given-names></name><name><surname>Wysoker</surname> <given-names>A</given-names></name><name><surname>Fennell</surname> <given-names>T</given-names></name><name><surname>Ruan</surname> <given-names>J</given-names></name><name><surname>Homer</surname> <given-names>N</given-names></name><name><surname>Marth</surname> <given-names>G</given-names></name><name><surname>Abecasis</surname> <given-names>G</given-names></name><name><surname>Durbin</surname> <given-names>R</given-names></name><collab>1000 Genome Project Data Processing Subgroup</collab></person-group><year iso-8601-date="2009">2009</year><article-title>The sequence alignment/Map format and SAMtools</article-title><source>Bioinformatics</source><volume>25</volume><fpage>2078</fpage><lpage>2079</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Durbin</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Fast and accurate short read alignment with Burrows-Wheeler transform</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1754</fpage><lpage>1760</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id><pub-id pub-id-type="pmid">19451168</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lipsitch</surname> <given-names>M</given-names></name><name><surname>Siller</surname> <given-names>S</given-names></name><name><surname>Nowak</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>The evolution of virulence in pathogens with vertical and horizontal transmission</article-title><source>Evolution</source><volume>50</volume><fpage>1729</fpage><lpage>1741</lpage><pub-id pub-id-type="doi">10.1111/j.1558-5646.1996.tb03560.x</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Fu</surname> <given-names>YX</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Exploring population size changes using SNP frequency spectra</article-title><source>Nature Genetics</source><volume>47</volume><fpage>555</fpage><lpage>559</lpage><pub-id pub-id-type="doi">10.1038/ng.3254</pub-id><pub-id pub-id-type="pmid">25848749</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maeda</surname> <given-names>S</given-names></name><name><surname>Kamita</surname> <given-names>SG</given-names></name><name><surname>Kondo</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Host range expansion of <italic>Autographa californica</italic> nuclear polyhedrosis virus (NPV) following recombination of a 0.6-kilobase-pair DNA fragment originating from <italic>Bombyx mori</italic> NPV</article-title><source>Journal of Virology</source><volume>67</volume><fpage>6234</fpage><lpage>6238</lpage><pub-id pub-id-type="doi">10.1128/JVI.67.10.6234-6238.1993</pub-id><pub-id pub-id-type="pmid">8396678</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Cutadapt removes adapter sequences from high-throughput sequencing reads</article-title><source>Technical Notes</source><volume>1</volume><elocation-id>12</elocation-id><pub-id pub-id-type="doi">10.14806/ej.17.1.200</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martinez-Picado</surname> <given-names>J</given-names></name><name><surname>Frost</surname> <given-names>SD</given-names></name><name><surname>Izquierdo</surname> <given-names>N</given-names></name><name><surname>Morales-Lopetegi</surname> <given-names>K</given-names></name><name><surname>Marfil</surname> <given-names>S</given-names></name><name><surname>Puig</surname> <given-names>T</given-names></name><name><surname>Cabrera</surname> <given-names>C</given-names></name><name><surname>Clotet</surname> <given-names>B</given-names></name><name><surname>Ruiz</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Viral evolution during structured treatment interruptions in chronically human immunodeficiency virus-infected individuals</article-title><source>Journal of Virology</source><volume>76</volume><fpage>12344</fpage><lpage>12348</lpage><pub-id pub-id-type="doi">10.1128/JVI.76.23.12344-12348.2002</pub-id><pub-id pub-id-type="pmid">12414975</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martins</surname> <given-names>NE</given-names></name><name><surname>Faria</surname> <given-names>VG</given-names></name><name><surname>Nolte</surname> <given-names>V</given-names></name><name><surname>Schlötterer</surname> <given-names>C</given-names></name><name><surname>Teixeira</surname> <given-names>L</given-names></name><name><surname>Sucena</surname> <given-names>É</given-names></name><name><surname>Magalhães</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Host adaptation to viruses relies on few genes with different cross-resistance properties</article-title><source>PNAS</source><volume>111</volume><fpage>5938</fpage><lpage>5943</lpage><pub-id pub-id-type="doi">10.1073/pnas.1400378111</pub-id><pub-id pub-id-type="pmid">24711428</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matsubara</surname> <given-names>K</given-names></name><name><surname>Otsuji</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="1978">1978</year><article-title>Preparation of plasmids from lambdoid phages and studies on their incompatibilities</article-title><source>Plasmid</source><volume>1</volume><fpage>284</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1016/0147-619X(78)90046-X</pub-id><pub-id pub-id-type="pmid">372966</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>May</surname> <given-names>RM</given-names></name><name><surname>Nowak</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Coinfection and the evolution of parasite virulence</article-title><source>Proceedings. Biological Sciences</source><volume>261</volume><fpage>209</fpage><lpage>215</lpage><pub-id pub-id-type="doi">10.1098/rspb.1995.0138</pub-id><pub-id pub-id-type="pmid">7568274</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McDonald</surname> <given-names>JH</given-names></name><name><surname>Kreitman</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Adaptive protein evolution at the adh locus in <italic>Drosophila</italic></article-title><source>Nature</source><volume>351</volume><fpage>652</fpage><lpage>654</lpage><pub-id pub-id-type="doi">10.1038/351652a0</pub-id><pub-id pub-id-type="pmid">1904993</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McKenna</surname> <given-names>A</given-names></name><name><surname>Hanna</surname> <given-names>M</given-names></name><name><surname>Banks</surname> <given-names>E</given-names></name><name><surname>Sivachenko</surname> <given-names>A</given-names></name><name><surname>Cibulskis</surname> <given-names>K</given-names></name><name><surname>Kernytsky</surname> <given-names>A</given-names></name><name><surname>Garimella</surname> <given-names>K</given-names></name><name><surname>Altshuler</surname> <given-names>D</given-names></name><name><surname>Gabriel</surname> <given-names>S</given-names></name><name><surname>Daly</surname> <given-names>M</given-names></name><name><surname>DePristo</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The genome analysis toolkit: a MapReduce framework for analyzing next-generation DNA sequencing data</article-title><source>Genome Research</source><volume>20</volume><fpage>1297</fpage><lpage>1303</lpage><pub-id pub-id-type="doi">10.1101/gr.107524.110</pub-id><pub-id pub-id-type="pmid">20644199</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merkling</surname> <given-names>SH</given-names></name><name><surname>van Rij</surname> <given-names>RP</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Beyond RNAi: antiviral defense strategies in <italic>Drosophila</italic> and mosquito</article-title><source>Journal of Insect Physiology</source><volume>59</volume><fpage>159</fpage><lpage>170</lpage><pub-id pub-id-type="doi">10.1016/j.jinsphys.2012.07.004</pub-id><pub-id pub-id-type="pmid">22824741</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Meyer</surname> <given-names>JR</given-names></name><name><surname>Dobias</surname> <given-names>DT</given-names></name><name><surname>Medina</surname> <given-names>SJ</given-names></name><name><surname>Servilio</surname> <given-names>L</given-names></name><name><surname>Gupta</surname> <given-names>A</given-names></name><name><surname>Lenski</surname> <given-names>RE</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Ecological speciation of bacteriophage lambda in allopatry and sympatry</article-title><source>Science</source><volume>354</volume><fpage>1301</fpage><lpage>1304</lpage><pub-id pub-id-type="doi">10.1126/science.aai8446</pub-id><pub-id pub-id-type="pmid">27884940</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielsen</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Molecular signatures of natural selection</article-title><source>Annual Review of Genetics</source><volume>39</volume><fpage>197</fpage><lpage>218</lpage><pub-id pub-id-type="doi">10.1146/annurev.genet.39.073003.112420</pub-id><pub-id pub-id-type="pmid">16285858</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielsen</surname> <given-names>R</given-names></name><name><surname>Bustamante</surname> <given-names>C</given-names></name><name><surname>Clark</surname> <given-names>AG</given-names></name><name><surname>Glanowski</surname> <given-names>S</given-names></name><name><surname>Sackton</surname> <given-names>TB</given-names></name><name><surname>Hubisz</surname> <given-names>MJ</given-names></name><name><surname>Fledel-Alon</surname> <given-names>A</given-names></name><name><surname>Tanenbaum</surname> <given-names>DM</given-names></name><name><surname>Civello</surname> <given-names>D</given-names></name><name><surname>White</surname> <given-names>TJ</given-names></name><name><surname>J Sninsky</surname> <given-names>J</given-names></name><name><surname>Adams</surname> <given-names>MD</given-names></name><name><surname>Cargill</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>A scan for positively selected genes in the genomes of humans and chimpanzees</article-title><source>PLOS Biology</source><volume>3</volume><elocation-id>e170</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0030170</pub-id><pub-id pub-id-type="pmid">15869325</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nowak</surname> <given-names>MA</given-names></name><name><surname>May</surname> <given-names>RM</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Superinfection and the evolution of parasite virulence</article-title><source>Proceedings R. Soc</source><volume>255</volume><fpage>81</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.1098/rspb.1994.0012</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Orr</surname> <given-names>HA</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Population genetics of speciation: the evolution of hybrid incompatibilities</article-title><source>Genetics</source><volume>139</volume><fpage>1805</fpage><lpage>1813</lpage><pub-id pub-id-type="pmid">7789779</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palmer</surname> <given-names>WH</given-names></name><name><surname>Medd</surname> <given-names>NC</given-names></name><name><surname>Beard</surname> <given-names>PM</given-names></name><name><surname>Obbard</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Isolation of a natural DNA virus of <italic>Drosophila melanogaster</italic>, and characterisation of host resistance and immune responses</article-title><source>PLOS Pathogens</source><volume>14</volume><elocation-id>e1007050</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1007050</pub-id><pub-id pub-id-type="pmid">29864164</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palmer</surname> <given-names>WH</given-names></name><name><surname>Joosten</surname> <given-names>J</given-names></name><name><surname>Overheul</surname> <given-names>GJ</given-names></name><name><surname>Jansen</surname> <given-names>PW</given-names></name><name><surname>Vermeulen</surname> <given-names>M</given-names></name><name><surname>Obbard</surname> <given-names>DJ</given-names></name><name><surname>Van Rij</surname> <given-names>RP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Induction and suppression of NF- κB signalling by a DNA virus of <italic>Drosophila</italic></article-title><source>Journal of Virology</source><volume>93</volume><elocation-id>e01443</elocation-id><pub-id pub-id-type="doi">10.1128/JVI.01443-18</pub-id><pub-id pub-id-type="pmid">30404807</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paradis</surname> <given-names>E</given-names></name><name><surname>Claude</surname> <given-names>J</given-names></name><name><surname>Strimmer</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>APE: analyses of phylogenetics and evolution in R language</article-title><source>Bioinformatics</source><volume>20</volume><fpage>289</fpage><lpage>290</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btg412</pub-id><pub-id pub-id-type="pmid">14734327</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paterson</surname> <given-names>S</given-names></name><name><surname>Vogwill</surname> <given-names>T</given-names></name><name><surname>Buckling</surname> <given-names>A</given-names></name><name><surname>Benmayor</surname> <given-names>R</given-names></name><name><surname>Spiers</surname> <given-names>AJ</given-names></name><name><surname>Thomson</surname> <given-names>NR</given-names></name><name><surname>Quail</surname> <given-names>M</given-names></name><name><surname>Smith</surname> <given-names>F</given-names></name><name><surname>Walker</surname> <given-names>D</given-names></name><name><surname>Libberton</surname> <given-names>B</given-names></name><name><surname>Fenton</surname> <given-names>A</given-names></name><name><surname>Hall</surname> <given-names>N</given-names></name><name><surname>Brockhurst</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Antagonistic coevolution accelerates molecular evolution</article-title><source>Nature</source><volume>464</volume><fpage>275</fpage><lpage>278</lpage><pub-id pub-id-type="doi">10.1038/nature08798</pub-id><pub-id pub-id-type="pmid">20182425</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Patterson</surname> <given-names>JT</given-names></name><name><surname>Stone</surname> <given-names>WS</given-names></name></person-group><year iso-8601-date="1949">1949</year><source>Studies in the genetics of Drosophila</source><publisher-name>University of Texas Publications</publisher-name></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pennings</surname> <given-names>PS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Standing genetic variation and the evolution of drug resistance in HIV</article-title><source>PLOS Computational Biology</source><volume>8</volume><elocation-id>e1002527</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1002527</pub-id><pub-id pub-id-type="pmid">22685388</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pennings</surname> <given-names>PS</given-names></name><name><surname>Kryazhimskiy</surname> <given-names>S</given-names></name><name><surname>Wakeley</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Loss and recovery of genetic diversity in adapting populations of HIV</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004000</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004000</pub-id><pub-id pub-id-type="pmid">24465214</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Perron</surname> <given-names>GG</given-names></name><name><surname>Zasloff</surname> <given-names>M</given-names></name><name><surname>Bell</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Experimental evolution of resistance to an antimicrobial peptide</article-title><source>Proceedings of the Royal Society B: Biological Sciences</source><volume>273</volume><fpage>251</fpage><lpage>256</lpage><pub-id pub-id-type="doi">10.1098/rspb.2005.3301</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="software"><person-group person-group-type="author"><collab>Picard</collab></person-group><year iso-8601-date="2020">2020</year><source>Picard</source><ext-link ext-link-type="uri" xlink:href="http://broadinstitute.github.io/picard">http://broadinstitute.github.io/picard</ext-link></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Purcell</surname> <given-names>S</given-names></name><name><surname>Neale</surname> <given-names>B</given-names></name><name><surname>Todd-Brown</surname> <given-names>K</given-names></name><name><surname>Thomas</surname> <given-names>L</given-names></name><name><surname>Ferreira</surname> <given-names>MA</given-names></name><name><surname>Bender</surname> <given-names>D</given-names></name><name><surname>Maller</surname> <given-names>J</given-names></name><name><surname>Sklar</surname> <given-names>P</given-names></name><name><surname>de Bakker</surname> <given-names>PI</given-names></name><name><surname>Daly</surname> <given-names>MJ</given-names></name><name><surname>Sham</surname> <given-names>PC</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>PLINK: a tool set for whole-genome association and population-based linkage analyses</article-title><source>The American Journal of Human Genetics</source><volume>81</volume><fpage>559</fpage><lpage>575</lpage><pub-id pub-id-type="doi">10.1086/519795</pub-id><pub-id pub-id-type="pmid">17701901</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quast</surname> <given-names>C</given-names></name><name><surname>Pruesse</surname> <given-names>E</given-names></name><name><surname>Yilmaz</surname> <given-names>P</given-names></name><name><surname>Gerken</surname> <given-names>J</given-names></name><name><surname>Schweer</surname> <given-names>T</given-names></name><name><surname>Yarza</surname> <given-names>P</given-names></name><name><surname>Peplies</surname> <given-names>J</given-names></name><name><surname>Glöckner</surname> <given-names>FO</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The SILVA ribosomal RNA gene database project: improved data processing and web-based tools</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>D590</fpage><lpage>D596</lpage><pub-id pub-id-type="doi">10.1093/nar/gks1219</pub-id><pub-id pub-id-type="pmid">23193283</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="software"><person-group person-group-type="author"><collab>R Development Core Team</collab></person-group><year iso-8601-date="2013">2013</year><data-title>R: A Language and Environment for Statistical Computing</data-title><publisher-loc>Vienna, Austria</publisher-loc><publisher-name>R Foundation for Statistical Computing</publisher-name><ext-link ext-link-type="uri" xlink:href="http://www.r-project.org">http://www.r-project.org</ext-link></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Real</surname> <given-names>LA</given-names></name><name><surname>Henderson</surname> <given-names>JC</given-names></name><name><surname>Biek</surname> <given-names>R</given-names></name><name><surname>Snaman</surname> <given-names>J</given-names></name><name><surname>Jack</surname> <given-names>TL</given-names></name><name><surname>Childs</surname> <given-names>JE</given-names></name><name><surname>Stahl</surname> <given-names>E</given-names></name><name><surname>Waller</surname> <given-names>L</given-names></name><name><surname>Tinline</surname> <given-names>R</given-names></name><name><surname>Nadin-Davis</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Unifying the spatial population dynamics and molecular evolution of epidemic Rabies virus</article-title><source>PNAS</source><volume>102</volume><fpage>12107</fpage><lpage>12111</lpage><pub-id pub-id-type="doi">10.1073/pnas.0500057102</pub-id><pub-id pub-id-type="pmid">16103358</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>MD</given-names></name><name><surname>McCarthy</surname> <given-names>DJ</given-names></name><name><surname>Smyth</surname> <given-names>GK</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>edgeR: a bioconductor package for differential expression analysis of digital gene expression data</article-title><source>Bioinformatics</source><volume>26</volume><fpage>139</fpage><lpage>140</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp616</pub-id><pub-id pub-id-type="pmid">19910308</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Rohrmann</surname> <given-names>GF</given-names></name></person-group><year iso-8601-date="2013">2013</year><source>Baculovirus Molecular Biology</source></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rokyta</surname> <given-names>DR</given-names></name><name><surname>Wichman</surname> <given-names>HA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Genic incompatibilities in two hybrid bacteriophages</article-title><source>Molecular Biology and Evolution</source><volume>26</volume><fpage>2831</fpage><lpage>2839</lpage><pub-id pub-id-type="doi">10.1093/molbev/msp199</pub-id><pub-id pub-id-type="pmid">19726536</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sabin</surname> <given-names>LR</given-names></name><name><surname>Hanna</surname> <given-names>SL</given-names></name><name><surname>Cherry</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Innate antiviral immunity in <italic>Drosophila</italic></article-title><source>Current Opinion in Immunology</source><volume>22</volume><fpage>4</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1016/j.coi.2010.01.007</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sackton</surname> <given-names>TB</given-names></name><name><surname>Lazzaro</surname> <given-names>BP</given-names></name><name><surname>Schlenke</surname> <given-names>TA</given-names></name><name><surname>Evans</surname> <given-names>JD</given-names></name><name><surname>Hultmark</surname> <given-names>D</given-names></name><name><surname>Clark</surname> <given-names>AG</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Dynamic evolution of the innate immune system in <italic>Drosophila</italic></article-title><source>Nature Genetics</source><volume>39</volume><fpage>1461</fpage><lpage>1468</lpage><pub-id pub-id-type="doi">10.1038/ng.2007.60</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sagulenko</surname> <given-names>P</given-names></name><name><surname>Puller</surname> <given-names>V</given-names></name><name><surname>Neher</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>TreeTime: maximum-likelihood phylodynamic analysis</article-title><source>Virus Evolution</source><volume>4</volume><elocation-id>vex042</elocation-id><pub-id pub-id-type="doi">10.1093/ve/vex042</pub-id><pub-id pub-id-type="pmid">29340210</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Scanlan</surname> <given-names>PD</given-names></name><name><surname>Buckling</surname> <given-names>A</given-names></name><name><surname>Hall</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Experimental evolution and bacterial resistance: (co)evolutionary costs and trade-offs as opportunities in phage therapy research</article-title><source>Bacteriophage</source><volume>5</volume><elocation-id>e1050153</elocation-id><pub-id pub-id-type="doi">10.1080/21597081.2015.1050153</pub-id><pub-id pub-id-type="pmid">26459626</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schulz</surname> <given-names>MH</given-names></name><name><surname>Zerbino</surname> <given-names>DR</given-names></name><name><surname>Vingron</surname> <given-names>M</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Oases: robust de novo RNA-seq assembly across the dynamic range of expression levels</article-title><source>Bioinformatics</source><volume>28</volume><fpage>1086</fpage><lpage>1092</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts094</pub-id><pub-id pub-id-type="pmid">22368243</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shin</surname> <given-names>J-H</given-names></name><name><surname>Blay</surname> <given-names>S</given-names></name><name><surname>Graham</surname> <given-names>J</given-names></name><name><surname>McNeney</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>LDheatmap : An R Function for Graphical Display of Pairwise Linkage Disequilibria Between Single Nucleotide Polymorphisms </article-title><source>Journal of Statistical Software</source><volume>16</volume><elocation-id>jss.v016.c03</elocation-id><pub-id pub-id-type="doi">10.18637/jss.v016.c03</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shultz</surname> <given-names>AJ</given-names></name><name><surname>Sackton</surname> <given-names>TB</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Immune genes are hotspots of shared positive selection across birds and mammals</article-title><source>eLife</source><volume>8</volume><elocation-id>e41815</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.41815</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Smit</surname> <given-names>AFA</given-names></name><name><surname>Hubley</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2008">2008</year><source>RepeatModeler Open-1.0</source></element-citation></ref><ref id="bib93"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Smit</surname> <given-names>AFA</given-names></name><name><surname>Hubley</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><source>RepeatMasker Open-4.0</source></element-citation></ref><ref id="bib94"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>NGC</given-names></name><name><surname>Eyre-Walker</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Adaptive protein evolution in <italic>Drosophila</italic></article-title><source>Nature</source><volume>415</volume><fpage>1022</fpage><lpage>1024</lpage><pub-id pub-id-type="doi">10.1038/4151022a</pub-id></element-citation></ref><ref id="bib95"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soetaert</surname> <given-names>K</given-names></name><name><surname>Petzoldt</surname> <given-names>T</given-names></name><name><surname>Setzer</surname> <given-names>RW</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Solving differential equations in <italic>R</italic> : package deSolve</article-title><source>Journal of Statistical Software</source><volume>33</volume><fpage>1</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.18637/jss.v033.i09</pub-id></element-citation></ref><ref id="bib96"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Souza</surname> <given-names>V</given-names></name><name><surname>Travisano</surname> <given-names>M</given-names></name><name><surname>Turner</surname> <given-names>PE</given-names></name><name><surname>Eguiarte</surname> <given-names>LE</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Does experimental evolution reflect patterns in natural populations? <italic>E. coli</italic> strains from long-term studies compared with wild isolates</article-title><source>Antonie Van Leeuwenhoek</source><volume>81</volume><fpage>143</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1023/a:1020594013195</pub-id><pub-id pub-id-type="pmid">12448713</pub-id></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stadler</surname> <given-names>T</given-names></name><name><surname>Kuhnert</surname> <given-names>D</given-names></name><name><surname>Bonhoeffer</surname> <given-names>S</given-names></name><name><surname>Drummond</surname> <given-names>AJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Birth-death skyline plot reveals temporal changes of epidemic spread in HIV and hepatitis C virus (HCV)</article-title><source>PNAS</source><volume>110</volume><fpage>228</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1073/pnas.1207965110</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stoletzki</surname> <given-names>N</given-names></name><name><surname>Eyre-Walker</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Estimation of the neutrality index</article-title><source>Molecular Biology and Evolution</source><volume>28</volume><fpage>63</fpage><lpage>70</lpage><pub-id pub-id-type="doi">10.1093/molbev/msq249</pub-id><pub-id pub-id-type="pmid">20837603</pub-id></element-citation></ref><ref id="bib99"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Unckless</surname> <given-names>RL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A DNA virus of <italic>Drosophila</italic></article-title><source>PLOS ONE</source><volume>6</volume><elocation-id>e26564</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0026564</pub-id><pub-id pub-id-type="pmid">22053195</pub-id></element-citation></ref><ref id="bib100"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Mierlo</surname> <given-names>JT</given-names></name><name><surname>Bronkhorst</surname> <given-names>AW</given-names></name><name><surname>Overheul</surname> <given-names>GJ</given-names></name><name><surname>Sadanandan</surname> <given-names>SA</given-names></name><name><surname>Ekström</surname> <given-names>JO</given-names></name><name><surname>Heestermans</surname> <given-names>M</given-names></name><name><surname>Hultmark</surname> <given-names>D</given-names></name><name><surname>Antoniewski</surname> <given-names>C</given-names></name><name><surname>van Rij</surname> <given-names>RP</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Convergent evolution of argonaute-2 slicer antagonism in two distinct insect RNA viruses</article-title><source>PLOS Pathogens</source><volume>8</volume><elocation-id>e1002872</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1002872</pub-id><pub-id pub-id-type="pmid">22916019</pub-id></element-citation></ref><ref id="bib101"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name><name><surname>Kleespies</surname> <given-names>RG</given-names></name><name><surname>Huger</surname> <given-names>AM</given-names></name><name><surname>Jehle</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The genome of Gryllus bimaculatus Nudivirus indicates an ancient diversification of baculovirus-related nonoccluded nudiviruses of insects</article-title><source>Journal of Virology</source><volume>81</volume><fpage>5395</fpage><lpage>5406</lpage><pub-id pub-id-type="doi">10.1128/JVI.02781-06</pub-id><pub-id pub-id-type="pmid">17360757</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name><name><surname>Kleespies</surname> <given-names>RG</given-names></name><name><surname>Ramle</surname> <given-names>MB</given-names></name><name><surname>Jehle</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Sequencing of the large dsDNA genome of oryctes Rhinoceros nudivirus using multiple displacement amplification of nanogram amounts of virus DNA</article-title><source>Journal of Virological Methods</source><volume>152</volume><fpage>106</fpage><lpage>108</lpage><pub-id pub-id-type="doi">10.1016/j.jviromet.2008.06.003</pub-id><pub-id pub-id-type="pmid">18598718</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Webster</surname> <given-names>CL</given-names></name><name><surname>Waldron</surname> <given-names>FM</given-names></name><name><surname>Robertson</surname> <given-names>S</given-names></name><name><surname>Crowson</surname> <given-names>D</given-names></name><name><surname>Ferrari</surname> <given-names>G</given-names></name><name><surname>Quintana</surname> <given-names>JF</given-names></name><name><surname>Brouqui</surname> <given-names>J-M</given-names></name><name><surname>Bayne</surname> <given-names>EH</given-names></name><name><surname>Longdon</surname> <given-names>B</given-names></name><name><surname>Buck</surname> <given-names>AH</given-names></name><name><surname>Lazzaro</surname> <given-names>BP</given-names></name><name><surname>Akorli</surname> <given-names>J</given-names></name><name><surname>Haddrill</surname> <given-names>PR</given-names></name><name><surname>Obbard</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The discovery, distribution, and evolution of viruses associated with <italic>Drosophila melanogaster</italic></article-title><source>PLOS Biology</source><volume>13</volume><elocation-id>e1002210</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1002210</pub-id><pub-id pub-id-type="pmid">26172158</pub-id></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>GC</given-names></name><name><surname>Nesse</surname> <given-names>RM</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>The dawn of darwinian medicine</article-title><source>The Quarterly Review of Biology</source><volume>66</volume><fpage>1</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1086/417048</pub-id><pub-id pub-id-type="pmid">2052670</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wilm</surname> <given-names>A</given-names></name><name><surname>Aw</surname> <given-names>PP</given-names></name><name><surname>Bertrand</surname> <given-names>D</given-names></name><name><surname>Yeo</surname> <given-names>GH</given-names></name><name><surname>Ong</surname> <given-names>SH</given-names></name><name><surname>Wong</surname> <given-names>CH</given-names></name><name><surname>Khor</surname> <given-names>CC</given-names></name><name><surname>Petric</surname> <given-names>R</given-names></name><name><surname>Hibberd</surname> <given-names>ML</given-names></name><name><surname>Nagarajan</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>LoFreq: a sequence-quality aware, ultra-sensitive variant caller for uncovering cell-population heterogeneity from high-throughput sequencing datasets</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>11189</fpage><lpage>11201</lpage><pub-id pub-id-type="doi">10.1093/nar/gks918</pub-id><pub-id pub-id-type="pmid">23066108</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>TD</given-names></name><name><surname>Nacu</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Fast and SNP-tolerant detection of complex variants and splicing in short reads</article-title><source>Bioinformatics</source><volume>26</volume><fpage>873</fpage><lpage>881</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq057</pub-id><pub-id pub-id-type="pmid">20147302</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zambon</surname> <given-names>RA</given-names></name><name><surname>Nandakumar</surname> <given-names>M</given-names></name><name><surname>Vakharia</surname> <given-names>VN</given-names></name><name><surname>Wu</surname> <given-names>LP</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The toll pathway is important for an antiviral response in <italic>Drosophila</italic></article-title><source>PNAS</source><volume>102</volume><fpage>7257</fpage><lpage>7262</lpage><pub-id pub-id-type="doi">10.1073/pnas.0409181102</pub-id><pub-id pub-id-type="pmid">15878994</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.58931.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Ebert</surname><given-names>Dieter</given-names></name><role>Reviewing Editor</role><aff><institution>University of Basel</institution><country>Switzerland</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Gagneux</surname><given-names>Sebastian</given-names> </name><role>Reviewer</role><aff><institution>University of Basel</institution><country>Switzerland</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>This study reports on a phenotypic and genetic polymorphism of DiNVirus in natural populations of <italic>Drosophila</italic>. The authors present a series of experiments and assessments to understand how the polymorphism evolved and what implication it has for the host and conclude that the observed polymorphism arose multiple times independently and that it is maintained in a polymorphic state. The results are very clear and convincing and provide an excellent example for the power of natural selection in shaping host-parasite interactions.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Recurrent evolution of two competing haplotypes in an insect DNA virus&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by three peer reviewers, one of whom is a member of our Board of Reviewing Editors, and the evaluation has been overseen by Diethard Tautz as the Senior Editor. The following individual involved in review of your submission has agreed to reveal their identity: Sebastian Gagneux (Reviewer #3).</p><p>The reviewers have discussed the reviews with one another and the Reviewing Editor has drafted this decision to help you prepare a revised submission.</p><p>As the editors have judged that your manuscript is of interest, but as described below that additional experiments are required before it is published, we would like to draw your attention to changes in our revision policy that we have made in response to COVID-19 (https://elifesciences.org/articles/57162). First, because many researchers have temporarily lost access to the labs, we will give authors as much time as they need to submit revised manuscripts. We are also offering, if you choose, to post the manuscript to bioRxiv (if it is not already there) along with this decision letter and a formal designation that the manuscript is &quot;in revision at <italic>eLife</italic>&quot;. Please let us know if you would like to pursue this option. (If your work is more suitable for medRxiv, you will need to post the preprint yourself, as the mechanisms for us to do so are still in development.)</p><p>Your manuscript presents an exciting finding of the evolution and biology of a natural viral pathogen of <italic>Drosophila</italic>. You suggest that two distinct haplotypes of the virus evolved multiple times in isolated host populations. These two haplotypes are characterised by 11 SNPs and associated phenotypic traits (differences in viral titer). The independent evolution of such a complex trait seems rather unusual, making this nice example of evolution of host-parasite interactions.</p><p>The three reviewers of this manuscripts have a lot of praise for the study. However, they also raise a number of important points that needs to be addressed. Most important, the substantive technical points raised by reviewer 2 need careful consideration. This includes explaining and resolving the inconsistencies in the analysis and the right choice of the methods. Reviewer 1 pointed out problems with the simulation. Below are the detailed reviews. A final decision will only be possible when the problems in the analysis are solved.</p><p><italic>Reviewer #1:</italic></p><p>This study reports on a phenotypic and genetic polymorphism of DiNVirus in <italic>Drosophila</italic>. The study is very appealing as this study system is a natural system (unlike <italic>D. melanogaster</italic>) with a well understood ecology and biogeography. The authors go through a series of experiments and assessments to understand how the polymorphism evolved and what implication it has for the host. The main conclusion is that the observed polymorphism arose multiple times independently and that it is maintained in a polymorphic state. The mechanism for this maintenance is not entirely clear. A simulation model is used to bring some light into this puzzle. For the most part, the study is solid and well carried out. Below are a number of points that may help the authors to present their material more clearly.</p><p>The Introduction is not to the point. The Introduction moves among various aspects of host-parasite interactions without giving the reader an idea where this is going. At various places topics are raised that then later dismissed. For example: the first two paragraphs are about host response to viruses. In the third paragraph it become system specific, but now switches mainly to the DiNV system. It is not clear where one is going here. The Introduction (sixth paragraph) is summed up with a very generic phrase without much perspective on what is going to follow. At this place I still do not know what this paper will be about. A much more targeted Introduction is necessary. What are the questions? What was driving this research? What are the hypotheses?</p><p>With 11 SNPs spread across the chromosome and obligate recombination every round of replication, the proportion of viral offspring with sub-optimal multi-SNP-genotypes must be huge. How can the right genotypes be maintained? Discuss.</p><p>The authors run computer simulations (called SIR models, but they seem actually to be SI models) to support their ideas about virus evolution. The simulation regarding the accumulation of mutation is fine and gives quantitative support for presented evolutionary scenario.</p><p>A second simulation is about the competition of the two viral types. This is by far the weakest part of the manuscript. I am not convinced about the value of this simulation. The outcome can be predicted from the assumptions of the model, several of them are very speculative. Without data on transmission over the course of the infection, the simulations are not very helpful. I suggest to leave this out. It is ok to speculate about this in the Discussion, but it makes the Results part heavy and less strong. Also, the associated figure (Figure 7) is hard to understand.</p><p>The authors stress at multiple place that it is likely that the increased virulence of the high type is traded-off against transmission. The evidence for this is much weaker than the strength of these statements suggests. I strongly suggest to tone this down.</p><p><italic>Reviewer #2:</italic></p><p>This manuscript presents an exciting analysis of the evolution and biology of a natural viral pathogen of <italic>Drosophila</italic>. In outline: the authors claim that two distinct haplotypes of the virus co-occur in multiple host populations, that these 'evolved' independently (i.e. separate origins) through convergence, and that they have alternative 'life histories' (high titer and virulence, versus low titer and virulence), which may permit their coexistence.</p><p>The practical experiments and sequencing data are substantial and appear sound, and the work is likely to be of interest to a very broad host-pathogen audience. However, while some of these headline claims are well-supported, I have a number of serious concerns about the analyses and interpretation of others, and a few key methodological details are missing. The analyses would need substantial revision, or at least additional checks, before I would be convinced by the story.</p><p>1) Is there any possibility of circularity in defining the 'high' and 'low' haplotypes?</p><p>Two 'types' are defined based on 11 linked SNPs that are described as having 'high' or 'low' titer phenotypes. First, this requires a more robust approach to define 'significance' as (if there is any LD) tests are not independent, and sample sizes relative to predictive SNPs are small. Phenotypes (titers) should be permuted across genotypes (within populations) a thousand times and the analyses re-run for each permutation, then the tails of this distribution used to define significance, as commonly done for e.g. the DGRP. Second, this feels like it is dangerously circular: supposing all 11 SNPs were false positives, and arbitrary haplotypes erroneously defined based on them? Post hoc analysis of the difference between the two haplotypes could still show Figure 1B, because those differences were what drove the (erroneous) detection of the SNPs. So would Figure 1C, since that result necessarily follows from the definition of the haplotypes.</p><p>To convince me that the 'two haplotypes' interpretation is real, for each of the randomisation replicates one would need to define a 'high' and a 'low' haplotype based on the any randomisation spurious 'significant' SNPs and re-run the rest of Figure 1 to demonstrate that the separation between the real haplotypes is greater than that between those that result from permutation tests. I appreciate that this will require some computing time, but the authors already have all of the code, so it shouldn't be more than an afternoon of 'hands on' time.</p><p>Since the haplotypes overlap in phenotype space (Figure 1B) and not all have all SNPs (Figure 1C), how were intermediate haplotypes (with &lt;11 SNPs) assigned to high or low?</p><p>2) Did the haplotypes evolve independently three or four times?</p><p>Of the 11 SNPs defining the haplotype, three are non-synonymous, five are in the UTRs of known virulence genes, and three are intergenic SNPs. Is it really credible that a specific base change has arisen and been selected for independently in each of four populations, at each of the 11 sites? i.e. always A-&gt;C being beneficial at a site, but never A-&gt;G at that site? Even for the non-coding ones? This is an extraordinary claim that would require extraordinary evidence, for which the rates of recombination and gene conversion seem at the heart of the argument.</p><p>In some places the authors appear to assume (or assert) that recombination is frequent, while in others they assume it is completely absent, and nowhere do they explicitly test for it. If it is common, then none of the tree-based analyses can be used. If it is absent, then it is very hard to explain the patterns of diversity or LD in Figure 1—figure supplement 3, and some of the simulations may be inappropriate.</p><p>The tree-based analyses seem to be the basis of the major claim that these haplotypes arose independently multiple times, and that the order of the mutations arising was similar. Any tree analysis absolutely requires the absence of recombination or gene conversion, and this needs to be explicitly tested for here (e.g. using GARD, or possibly if diversity is low by inferring the ARG using tsinfer). The failure of LD to decay with distance hints that recombination is absent, but gene conversion over very short distances could still shuffle mutations between haplotypes. However, assuming its branch lengths are in mutations, the shape of the tree in Figure 6A (many short tips below large crowns) shouts 'recombination' to me. Where did this tree come from? It is not ultrametric, so if it was inferred with BEAST this is not standard output.</p><p>If the authors do have evidence that recombination is absent, can they confirm that the other analyses (population size history from the SFS; simulations) also assumed zero recombination? Although zero recombination would make me worry even more about the p-values in the GWAS discovery of the SNPs that define the haplotypes – but permutation tests would help deal with that.</p><p>I am really confused about their view of recombination, because a couple of times they imply recombination may be common, for example by noting that recombination is required in Nudivirus replication – but then this would invalidate the tree-based analyses? Plus, mechanistic recombination is irrelevant without coinfection infection, since recombination only between identical haplotypes does not have any effect. The observation that no co-infections occur suggests recombination should be very rare – but that's not what the tree looks like. The sixth paragraph of the Discussion suggests that not enough thought has been given to the likelihood, or implications, of recombination.</p><p>Finally, if there is no recombination, then 'fully derived' haplotypes (those with all 11 SNPs) can only be as old as the youngest of the 11 SNPs, unless the SNPs arose multiple time within populations as well. Is that compatible with the patterns of diversity and the timescale? One check would be to mask those 11, re-infer the tree, and check that the clades are still monophyletic. If these arose by re-current mutations due to strong selection, they could be warping the tree – this might be akin to the problem of recurrent drug or MHC selected mutations in HIV trees, where the known sites are excluded before analysis.</p><p>If we believe the two haplotypes are real, a much more credible 'story' to me would be two distinct and potentially old haplotypes maintained by selection in the face of a low level of ongoing gene conversion (or recombination). Is there some aspect of the data that this does not fit? This seems just as good a 'story', just as interesting, and just as publishable.</p><p>3) The MK-like analyses</p><p>Although a relatively minor part of the story, the MK analysis is extremely unclear. In part this is because it doesn't appear in the Materials and methods.</p><p>i) What software was used? The supporting data implies VCFtools can do this (which surprised me) but the method it uses was not explained. Was SNIpre fitted with the original code? The text seem to imply that raw counts were used, which would not be suitable unless Ks was &lt;0.3</p><p>ii) Although divergence is required, we're not told how (or from what species) it was estimated. If Ks&gt;0.3, then something more than raw counts should certainly be used. If Ks&gt;0.8, I would strongly advise against doing any sort of MK analysis.</p><p>iii) Figure 4 gives 'difference from background'. What is meant by difference from background? But if the background is a signal of strong constraint, could a positive signal here mean relaxed constraint rather than positive selection. What is the evidence that it's positive selection rather than relaxed constraint?</p><p>iv) In Figure 4—figure supplement 1 the populations are presented separately, and differences among them are interpreted as differences in positive selection. But surely the divergence number must be massive, larger than the polymorphism number, so almost all of this variation is due to differences in constraint affecting the Pn/Ps ratio. Or is the method one that uses high frequency derived SNPs as evidence of positive selection? If so, how was ancestral state identified? This is probably not possible to do reliably if Ks&gt;0.3. We need to see some raw numbers, and a lot more detail on the methods.</p><p><italic>Reviewer #3:</italic></p><p>This is an interesting piece of work on the co-evolution of the <italic>Drosophilainnubila</italic> Nudivirus (DiNV) and its host. The authors analyzed several natural populations of <italic>D. innubila</italic>, some of which were partially infected with DiNV, and simultaneously characterized both the host and the infecting viral pathogen using a combination of DNA and RNA sequencing and various complementary analytical approaches. Using a GWAS approach, they discovered a viral variant with high virulence (High Type) that differed by 11 strongly linked SNPs from low virulence variants. They demonstrate that the High Type associates with a higher viral titer and increased host mortality, and also validate these findings using experimental infection assays. Using a transcriptomic approach, they show that the High Type overexpresses genes known to be linked to viral virulence, and this correlated with the under expression of host genes involved in antiviral immunity, indicating that the increased virulence of the High Type is at least partially due to the inhibition of host defense mechanisms. They further show that loci associated with differences in virulence were under strong selection for adaptation, particularly genes involved in the viral envelope and virulence proteins. Based on their reconstruction of the most likely evolutionary histories of the High Type in the different host populations, they further conclude that the High Type emerged multiple times independently, and yet, the High Type did not outcompete the Low virulence variants in these populations. This might indicate varying trade-offs between virulence and transmission in the High and Low Types that vary across these populations. Finally, they show that the same phenomenon for High Type evolution of DiNV can also be observed in other <italic>Drosophila</italic> species. I have just a few comments:</p><p>1) Throughout the manuscript, the authors switch back and forth between the present and past tense, which seems awkward from a stylistic point of, I'd suggest to stick to the past tense through-out.</p><p>2) The fact that the exact same 11 SNPs evolve multiple times independently in mostly the same order is interesting. The authors note that an expected alternative could be different SNPs emerging in the same genes, but they don't discuss the potential mechanism, by which the exact same SNPs seem to be preferred instead.</p><p>3) The authors found little evidence of mixed infection with both the High and Low Types and conclude potential in-compatibility. Please expand on the potential mechanisms of this.</p><p>4) Related to the above comment, the authors observe no &quot;hybrids&quot; of High and Low Types again, suggesting &quot;incompatibility&quot;. Please discuss the difference between this genetic/genomic versus ecological incompatibility referred to above, as well as the potential link between these two types of incompatibilities.</p><p>5) Please rephrase the first sentence of the Discussion (what is the deference between &quot;to better infect&quot; and &quot;optimizing the infection&quot;?).</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.58931.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Reviewer #1:</p><p>[…] Below are a number of points that may help the authors to present their material more clearly.</p><p>The Introduction is not to the point. The Introduction moves among various aspects of host-parasite interactions without giving the reader an idea where this is going. At various places topics are raised that then later dismissed. For example: the first two paragraphs are about host response to viruses. In the third paragraph it become system specific, but now switches mainly to the DiNV system. It is not clear where one is going here. The Introduction (sixth paragraph) is summed up with a very generic phrase without much perspective on what is going to follow. At this place I still do not know what this paper will be about. A much more targeted Introduction is necessary. What are the questions? What was driving this research? What are the hypotheses?</p></disp-quote><p>We have rewritten the Introduction following your suggestions, thank you for the comments regarding this. We have moved sections around and rewritten parts so there is a better flow between paragraphs. We have also better provided our initial hypotheses and how these led to our findings. Specifically, we use the first paragraph to describe the host/virus coevolution and how viruses suppress the host, we then outline why people study viruses and why the DiNV system is ideal for viral study (with details of the system). Finally, we use the last paragraph to discuss the experiments we performed and a basic outline of our results.</p><disp-quote content-type="editor-comment"><p>With 11 SNPs spread across the chromosome and obligate recombination every round of replication, the proportion of viral offspring with sub-optimal multi-SNP-genotypes must be huge. How can the right genotypes be maintained? Discuss.</p></disp-quote><p>We have further expanded upon this in both the Results and the Discussion. In the Results we highlight that though there is obligate recombination, the chance that two strains with differing genotypes infect the same cell in an actual organism is very low, so we cannot know the actual rate or recombination, but it is likely lower than the high rate suggested and would lead to a lower proportion of intermediate strains. In the Discussion we also cover this and highlight that one possible difference between the High and Low strains is that they could infect different tissues (as is seen between other nudivirus strains). If the two types preferentially infect different tissues, then the chance of them co-infecting the same cell is impossible and would result in our observed absence of recombination between the High and Low type. We also discuss incompatibilities that could occur in suboptimal genotypes which could be the cause of their absence in our survey, or other factors which could drive the maintenance of the full haplotypes.</p><disp-quote content-type="editor-comment"><p>The authors run computer simulations (called SIR models, but they seem actually to be SI models) to support their ideas about virus evolution. The simulation regarding the accumulation of mutation is fine and gives quantitative support for presented evolutionary scenario.</p><p>A second simulation is about the competition of the two viral types. This is by far the weakest part of the manuscript. I am not convinced about the value of this simulation. The outcome can be predicted from the assumptions of the model, several of them are very speculative. Without data on transmission over the course of the infection, the simulations are not very helpful. I suggest to leave this out. It is ok to speculate about this in the Discussion, but it makes the Results part heavy and less strong. Also, the associated figure (Figure 7) is hard to understand.</p></disp-quote><p>We have now removed the second set of simulations and associated figures regarding the possibility of virulence/transmission equilibrium, as we agree it is the weakest part of the manuscript and is unlikely, as it is an unstable equilibrium. We now discuss the idea of a virulence/transmission equilibrium in the Discussion, however, we also discuss all the possible things that could have resulted in us finding two viral types (incomplete sweep, trade-off, optimized to different hosts).</p><disp-quote content-type="editor-comment"><p>The authors stress at multiple place that it is likely that the increased virulence of the high type is traded-off against transmission. The evidence for this is much weaker than the strength of these statements suggests. I strongly suggest to tone this down.</p></disp-quote><p>We have rewritten our Discussion to tone this down, in our Results we describe a relationship we find between increasing titer and increasing virulence. Beyond this, in the Discussion we examine multiple possible causes for the two maintained strains. We hypothesize that the two strains could be neutral and are drifting, that the High type is partway through a sweep, that a trade-off allows them to exist alongside each other or that each virus type is optimized to different host species/genotypes (with migration between two types). Overall, we think we have removed the majority of our Results and Discussion where talk about the trade-off, as it is not the only suggested explanation.</p><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>[…] The practical experiments and sequencing data are substantial and appear sound, and the work is likely to be of interest to a very broad host-pathogen audience. However, while some of these headline claims are well-supported, I have a number of serious concerns about the analyses and interpretation of others, and a few key methodological details are missing. The analyses would need substantial revision, or at least additional checks, before I would be convinced by the story.</p><p>1) Is there any possibility of circularity in defining the 'high' and 'low' haplotypes?</p><p>Two 'types' are defined based on 11 linked SNPs that are described as having 'high' or 'low' titer phenotypes. First, this requires a more robust approach to define 'significance' as (if there is any LD) tests are not independent, and sample sizes relative to predictive SNPs are small. Phenotypes (titers) should be permuted across genotypes (within populations) a thousand times and the analyses re-run for each permutation, then the tails of this distribution used to define significance, as commonly done for e.g. the DGRP. Second, this feels like it is dangerously circular: supposing all 11 SNPs were false positives, and arbitrary haplotypes erroneously defined based on them? Post hoc analysis of the difference between the two haplotypes could still show Figure 1B, because those differences were what drove the (erroneous) detection of the SNPs. So would Figure 1C, since that result necessarily follows from the definition of the haplotypes.</p></disp-quote><p>There are two important considerations here so we will address them separately. We address the false negative problem below.</p><p>Defining significance since tests are not independent: Based on reviewer comments, we have significantly revised our approach at examining linkage disequilibrium and hope this clarifies some of the concerns. Given the signatures of LD, we did perform permutations and now include the permutation threshold in Figure 1A (slightly less stringent than the corrected P-value with a FDR of 0.01). So this was an important improvement but did not qualitatively change the result.</p><disp-quote content-type="editor-comment"><p>To convince me that the 'two haplotypes' interpretation is real, for each of the randomisation replicates one would need to define a 'high' and a 'low' haplotype based on the any randomisation spurious 'significant' SNPs and re-run the rest of Figure 1 to demonstrate that the separation between the real haplotypes is greater than that between those that result from permutation tests. I appreciate that this will require some computing time, but the authors already have all of the code, so it shouldn't be more than an afternoon of 'hands on' time.</p></disp-quote><p>The haplotype association problem: This is an important concern – if the first SNP is a false positive, they are all false positives. We think we have three convincing lines of evidence that these are not false positive SNPs. First, we have now performed the association study in the five populations (3 <italic>innubila</italic>, 1 <italic>azteca</italic> and 1 <italic>falleni</italic>) independently. Given the viral population structure among these populations (except maybe the sympatric <italic>innubila</italic>/<italic>azteca</italic>), these represent 5 independent tests and the haplotypes are significantly associated in all five populations. Thus a false negative would have to be a false positive 5 (or 4 if we don’t consider sympatric <italic>innubila</italic>/<italic>azteca</italic> as separate) times. Even if we consider our FDR threshold of 0.01 (these SNPs have lower P values), the likelihood of a false positive in 5 independent tests is 0.01^5 = 10^-10 or 1 in 10 billion (if 4 independent tests, that value drops to 10^-8 or 1 in 100 million). Second, we have followed the reviewer’s suggestions (we think) and permuted the titer associated with each strain 100,000 times, and each time binned strains by the most significant SNP and its 10 most closely linked SNPs, then find the difference in titer between two types. We find that the difference between the high and low type is significantly higher than by random chance for all SNPs, and no random combinations of alleles show the titer increase seen in Figure 1B. We have included the distribution of permuted differences between the artificial high and low types in Figure 1—figure supplement 2, with the true difference shown as a dotted line.</p><disp-quote content-type="editor-comment"><p>Since the haplotypes overlap in phenotype space (Figure 1B) and not all have all SNPs (Figure 1C), how were intermediate haplotypes (with &lt;11 SNPs) assigned to high or low?</p></disp-quote><p>We have now changed the manuscript to consider intermediate strains to be a third subset of DiNV and have excluded them from analyses comparing High to Low type, including Figure 1B.</p><disp-quote content-type="editor-comment"><p>2) Did the haplotypes evolve independently three or four times?</p></disp-quote><p>As we have clarified in the manuscript, the haplotype has recurrently evolved three times to our knowledge in the Sky Islands, and at least once more in another location.</p><disp-quote content-type="editor-comment"><p>Of the 11 SNPs defining the haplotype, three are non-synonymous, five are in the UTRs of known virulence genes, and three are intergenic SNPs. Is it really credible that a specific base change has arisen and been selected for independently in each of four populations, at each of the 11 sites? i.e. always A-&gt;C being beneficial at a site, but never A-&gt;G at that site? Even for the non-coding ones? This is an extraordinary claim that would require extraordinary evidence, for which the rates of recombination and gene conversion seem at the heart of the argument.</p></disp-quote><p>We agree it is extraordinary that specific base changes always arising is strange, though this is not uncommon for viruses, especially given the low, but ubiquitous levels of codon usage bias seen in viruses. Additionally, it is possible that specific nucleotide changes are necessary for regulatory differences. We have discussed this in the manuscript and have also searched for gene conversion and find no evidence of it between the significant SNPs and surrounding SNPs (Figure 5—figure supplement 3). Our argument for increased effective mutation rate coupled to titer also increases the likelihood of this occurring, as we discuss in the manuscript.</p><disp-quote content-type="editor-comment"><p>In some places the authors appear to assume (or assert) that recombination is frequent, while in others they assume it is completely absent, and nowhere do they explicitly test for it. If it is common, then none of the tree-based analyses can be used. If it is absent, then it is very hard to explain the patterns of diversity or LD in Figure 1—figure supplement 3, and some of the simulations may be inappropriate.</p></disp-quote><p>We now address recombination in the second section of the manuscript (after the associations). The reviewer is correct that establishing and being clear about the recombination landscape is crucial for the rest of the manuscript. We have inserted also clarifications in the manuscript in the Discussion. In summary, while recombination events occur anywhere in the genome and frequently, the effective recombination rate is not as extremely high as we may have intimated, as most recombination events are between identical genomes. Recombination between two distinct haplotypes (including the low and high titer haplotypes) would require that the two types infect the same cell and this may be rare even if a host is superinfected. Even when recombination occurs between the two haplotypes, we hypothesize that recombinant genomes may have incompatibilities and so may not be able to leave the host cell. Based on these comments we have also changed our Discussion in the manuscript and have more thoroughly searched for evidence of recombination events between different SNPs. We find no evidence of recombination between populations, meaning each SNP must recurrently evolve in each population, however the final haplotype may have formed by bringing the SNPs onto the same background.</p><disp-quote content-type="editor-comment"><p>The tree-based analyses seem to be the basis of the major claim that these haplotypes arose independently multiple times, and that the order of the mutations arising was similar. Any tree analysis absolutely requires the absence of recombination or gene conversion, and this needs to be explicitly tested for here (e.g. using GARD, or possibly if diversity is low by inferring the ARG using tsinfer). The failure of LD to decay with distance hints that recombination is absent, but gene conversion over very short distances could still shuffle mutations between haplotypes. However, assuming its branch lengths are in mutations, the shape of the tree in Figure 6A (many short tips below large crowns) shouts 'recombination' to me. Where did this tree come from? It is not ultrametric, so if it was inferred with BEAST this is not standard output.</p></disp-quote><p>The reviewer is correct, we have now used GARD to identify recombination events and find evidence of recombination events between significantly associated SNPs and background SNPs. We also see evidence of recombination in our r<sup>2</sup> analyses between background SNPs but not our significant SNPs. We have also regenerated the phylogeny in BEAST2 using BDsky/skyline (now in Figure 5, appropriate given the recombination findings). We have also generated a phylogeny in BEAST2 considering recombination and get a very similar phylogeny to that seen now in Figure 5. In all cases we find recurrent evolution of each high type SNP in each population, but the combinations of SNPs may either be due to recurrent mutation or recombination. As we find no evidence of recombination between populations, even if the full complement of 11 SNPs has not recurrently evolved on the same background, we still find evidence of recurrent recombination of each mutation in each population.</p><disp-quote content-type="editor-comment"><p>If the authors do have evidence that recombination is absent, can they confirm that the other analyses (population size history from the SFS; simulations) also assumed zero recombination? Although zero recombination would make me worry even more about the p-values in the GWAS discovery of the SNPs that define the haplotypes – but permutation tests would help deal with that.</p></disp-quote><p>To identify recombination events and measure associations of alleles, we have performed the following tests/used the following tools:</p><p>– GARD</p><p>– R2 in r</p><p>– 4 allele test between SNP combinations</p><p>We have also factored in recombination in the following analyses and still find support for recurrent evolution of each SNP (though SNPs may have recombined to form the final High Type):</p><p>– BEAST2</p><p>– TreeTimes</p><p>– DeSolve Simulations</p><p>– StairwayPlot</p><p>We find no evidence of recombination between populations, e.g. we find no recombinant haplotypes which are ½ CH and ½ PR ; however we find recombination within populations. The results shown in Figure 1—figure supplements 2 and 3, suggests recombination is on average quite common. Based on the GARD results and BEAST2 analysis, recombination could have occurred between intermediate types in each population, which may have generated the full High type. However, for this to occur each mutation must recurrently occur in each population on the intermediate type.</p><disp-quote content-type="editor-comment"><p>I am really confused about their view of recombination, because a couple of times they imply recombination may be common, for example by noting that recombination is required in Nudivirus replication – but then this would invalidate the tree-based analyses? Plus, mechanistic recombination is irrelevant without coinfection infection, since recombination only between identical haplotypes does not have any effect. The observation that no co-infections occur suggests recombination should be very rare – but that's not what the tree looks like. The sixth paragraph of the Discussion suggests that not enough thought has been given to the likelihood, or implications, of recombination.</p></disp-quote><p>We think the confusion is the difference between actual recombination (the physical process) and effective recombination (the genetic signature of recombination between two distinct haplotypes). Coinfection is required for effective recombination but not actual recombination. We expect recombination between neutral genotypes to be common if coinfections are common, but do not see recombination between the High and Low types, suggesting either they do not co-infect or recombinants between the two types are inviable. We have included this in the Discussion and reworked our discussion of recombination throughout.</p><disp-quote content-type="editor-comment"><p>Finally, if there is no recombination, then 'fully derived' haplotypes (those with all 11 SNPs) can only be as old as the youngest of the 11 SNPs, unless the SNPs arose multiple time within populations as well. Is that compatible with the patterns of diversity and the timescale? One check would be to mask those 11, re-infer the tree, and check that the clades are still monophyletic. If these arose by re-current mutations due to strong selection, they could be warping the tree – this might be akin to the problem of recurrent drug or MHC selected mutations in HIV trees, where the known sites are excluded before analysis.</p><p>If we believe the two haplotypes are real, a much more credible 'story' to me would be two distinct and potentially old haplotypes maintained by selection in the face of a low level of ongoing gene conversion (or recombination). Is there some aspect of the data that this does not fit? This seems just as good a 'story', just as interesting, and just as publishable.</p></disp-quote><p>In all cases, the phylogeny has been generated without the 11 focal SNPs but was also generated with the SNPs and is nearly identical in both cases, we have now mentioned this in the manuscript. The idea of the two old haplotypes being maintained in the face of gene conversion is interesting, and we now discuss this in the manuscript. However, the idea of long-term maintenance does not fit with any analyses we have performed, as High types do not cluster together. Additionally, we find the High type is present in the geographically distinct <italic>Drosophila falleni</italic> population and clusters completely separately from the innubila samples.</p><p>We do find some evidence of recombination events occurring around the significantly associated SNPs, so it is possible that multiple intermediate strains (with different complements of SNPs) have recurrently evolved in each population and have recombined to generate the full High type, or that the High type has evolved once in each population and is eroded as you described. We address this possibility in both the Results and the Discussion. We now have a dedicated section of our Results regarding detecting recombination and inferring if the High type SNPs were brought together by recombination. We are unable to conclude if the High type was formed exclusively by recombination (likely both sequential mutation and recombination play a role), but find no evidence of recombination between populations, which suggests that even if recombination brings the SNPs together onto the same background, the SNPs must have recurrently evolved in each population.</p><disp-quote content-type="editor-comment"><p>3) The MK-like analyses</p><p>Although a relatively minor part of the story, the MK analysis is extremely unclear. In part this is because it doesn't appear in the Materials and methods.</p><p>i) What software was used? The supporting data implies VCFtools can do this (which surprised me) but the method it uses was not explained. Was SNIpre fitted with the original code? The text seem to imply that raw counts were used, which would not be suitable unless Ks was &lt;0.3</p><p>ii) Although divergence is required, we're not told how (or from what species) it was estimated. If Ks&gt;0.3, then something more than raw counts should certainly be used. If Ks&gt;0.8, I would strongly advise against doing any sort of MK analysis.</p><p>iii) Figure 4 gives 'difference from background'. What is meant by difference from background? But if the background is a signal of strong constraint, could a positive signal here mean relaxed constraint rather than positive selection. What is the evidence that it's positive selection rather than relaxed constraint?</p><p>iv) In Figure 4—figure supplement 1 the populations are presented separately, and differences among them are interpreted as differences in positive selection. But surely the divergence number must be massive, larger than the polymorphism number, so almost all of this variation is due to differences in constraint affecting the Pn/Ps ratio. Or is the method one that uses high frequency derived SNPs as evidence of positive selection? If so, how was ancestral state identified? This is probably not possible to do reliably if Ks&gt;0.3. We need to see some raw numbers, and a lot more detail on the methods.</p></disp-quote><p>We apologise, the MK-based analyses methods were removed during the writing of the manuscript accidentally, as our results have been reconfigured as 2 manuscripts from one large manuscript, we have added them back. VCFtools was not used, we instead used the original SnIPRE code and SNPeff SNP assignments. We have also explained what we mean by difference from the background, specifically that we found the average statistical measure for nearby genes and compared each focal gene to that, to see how much the focal gene deferred from its surrounding average. We have attempted to answer all your points and questions in this new section of the Materials and methods. Average dS from the outgroups used was 0.133 and 0.279 between DiNV-Kallithea and DiNV-OrNV respectively, which we feel is enough divergence to adequately perform the analysis and is not too much to result in all differences being due to the Pn/Ps ratio. We also used these comparisons to identify the ancestral state, based on the Kallithea and OrNV variant. We have now included our measures of Dn/Ds and Pn/Ps as well as the estimated Selection Effect for our genes of interest as a supplementary figure. In this case we find that envelope proteins and AMPs have an excess of Dn/Ds per site compared to other genes, which is driving their elevated selection effect, while an excess of Pn/Ps and a deficit of Dn/Ds in unknown function DiNV genes is likely driving most other differences. We have chosen to use MK-based tests as opposed to Dn/Ds to identify if the changes are due to positive selection rather than relaxed constraint, as Dn/Ds is weighted by Pn/Ps to identify the proportion of substitutions fixed by selection.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>[…] 1) Throughout the manuscript, the authors switch back and forth between the present and past tense, which seems awkward from a stylistic point of, I'd suggest to stick to the past tense through-out.</p></disp-quote><p>We apologise for this, we have corrected this inconsistency.</p><disp-quote content-type="editor-comment"><p>2) The fact that the exact same 11 SNPs evolve multiple times independently in mostly the same order is interesting. The authors note that an expected alternative could be different SNPs emerging in the same genes, but they don't discuss the potential mechanism, by which the exact same SNPs seem to be preferred instead.</p></disp-quote><p>We have expanded our discussion of the compatibility, both from an evolution perspective and from a functional perspective. First, we hypothesize that the significantly associated SNPs have epistatic effects and thus must occur in a particular order to traverse the fitness landscape form Low type to the High type. In line with this, very few genes show signatures of adaptation in DiNV, which could further limit what beneficial mutations fix. We also discuss the different conditions which could lead to two types of DiNV being found in each population. We also hypothesize some methods these incompatibilities could occur, such as a change in the 19K/PIF-6 protein structure which limits membrane access for viral particles without this variant and could be rescued by a matching change in another protein. Or these two changes together are neutral but in combination prevent membrane access for the other viral type. We also find evidence of recombination events around the significantly associated SNPs, which could easily facilitate the recombination of variants onto the same background, without needing to traverse a lower fitness genotype combination.</p><disp-quote content-type="editor-comment"><p>3) The authors found little evidence of mixed infection with both the High and Low Types and conclude potential in-compatibility. Please expand on the potential mechanisms of this.</p></disp-quote><p>We have expanded our discussion of the compatibility, both from an evolution perspective and from a functional perspective, we have also highlighted similar studies where diverged viruses are unable to coninfect similar systems or are unable to produce viable recombinant particles.</p><disp-quote content-type="editor-comment"><p>4) Related to the above comment, the authors observe no &quot;hybrids&quot; of High and Low Types again, suggesting &quot;incompatibility&quot;. Please discuss the difference between this genetic/genomic versus ecological incompatibility referred to above, as well as the potential link between these two types of incompatibilities.</p></disp-quote><p>We have included this in our expanded section in the Discussion. Briefly we discuss how the incompatibility may not be caused by a functional effect, as most SNPs are upstream of genes, but could be due to an effect of the expression of the proteins in combination on viral particle formation, host fitness or even viral ability to transmit between hosts. We do not have zero intermediate types, but we have explained that these could be steps between two fitness peaks as the few intermediates without low fitness.</p><disp-quote content-type="editor-comment"><p>5) Please rephrase the first sentence of the Discussion (what is the deference between &quot;to better infect&quot; and &quot;optimizing the infection&quot;?).</p></disp-quote><p>We have attempted to clarify that we mean that viruses are evolving according to the transmission/virulence trade-off and so are not just evolving to propagate within hosts but to transmit to others, which usually requires the virus not kill the host before transmission can occur.</p></body></sub-article></article>