<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">87361</article-id><article-id pub-id-type="doi">10.7554/eLife.87361</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.87361.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group></article-categories><title-group><article-title>Viral genome sequence datasets display pervasive evidence of strand-specific substitution biases that are best described using non-reversible nucleotide substitution models</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name><surname>Sianga-Mete</surname><given-names>Rita</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6626-4007</contrib-id><email>rita@aims.ac.za</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Hartnady</surname><given-names>Penelope</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Mandikumba</surname><given-names>Wimbai Caroline</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Rutherford</surname><given-names>Kayleigh</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Currin</surname><given-names>Christopher Brian</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4809-5059</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Phelanyane</surname><given-names>Florence</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Stefan</surname><given-names>Sabina</given-names></name><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Weaver</surname><given-names>Steven</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Kosakovsky Pond</surname><given-names>Sergei L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4817-4029</contrib-id><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Martin</surname><given-names>Darren P</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8785-0870</contrib-id><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="fn" rid="con10"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03p74gp79</institution-id><institution>Division of Computational Biology, Institute of Infectious Diseases and Molecular Medicine, Department of Integrative Biomedical Sciences, Faculty of Health Sciences, University of Cape Town</institution></institution-wrap><addr-line><named-content content-type="city">Rondebosch</named-content></addr-line><country>South Africa</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03p74gp79</institution-id><institution>Department of Human Biology, Faculty of Health Sciences, University of Cape Town</institution></institution-wrap><addr-line><named-content content-type="city">Rondebosch</named-content></addr-line><country>South Africa</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03p74gp79</institution-id><institution>Centre for Infectious Disease and Epidemiology Research, School of Public Health and Family Medicine, University of Cape Town</institution></institution-wrap><addr-line><named-content content-type="city">Rondebosch</named-content></addr-line><country>South Africa</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05gq02987</institution-id><institution>Centre for Biomedical Engineering, School of Engineering, Brown University</institution></institution-wrap><addr-line><named-content content-type="city">Providence</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00kx1jb78</institution-id><institution>Institute for Genomics and Evolutionary Medicine, Department of Biology, Temple University</institution></institution-wrap><addr-line><named-content content-type="city">Philadelphia</named-content></addr-line><country>United States</country></aff><aff id="aff6"><label>6</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03p74gp79</institution-id><institution>Wellcome Center for Infectious Diseases Research in Africa, Institute of Infectious Disease and Molecular Medicine and Department of Medicine, University of Cape Town</institution></institution-wrap><addr-line><named-content content-type="city">Rondebosch</named-content></addr-line><country>South Africa</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Rokas</surname><given-names>Antonis</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02vm5rt34</institution-id><institution>Vanderbilt University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>30</day><month>09</month><year>2025</year></pub-date><volume>12</volume><elocation-id>RP87361</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-03-24"><day>24</day><month>03</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2022-12-29"><day>29</day><month>12</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.21203/rs.3.rs-2407778/v1"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-05-30"><day>30</day><month>05</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.87361.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2025-08-29"><day>29</day><month>08</month><year>2025</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.87361.2"/></event></pub-history><permissions><copyright-statement>© 2023, Sianga-Mete et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Sianga-Mete et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-87361-v1.pdf"/><abstract><p>Most phylogenetic trees are inferred using time-reversible evolutionary models that assume that the relative rates of substitution for any given pair of nucleotides are the same regardless of the direction of the substitutions. However, there is no reason to assume that the underlying biochemical mutational processes that cause substitutions are similarly symmetrical. We consider two non-reversible nucleotide substitution models: (1) a 6-rate non-reversible model (NREV6) that is applicable to analysing mutational processes in double-stranded genomes, in that complementary substitutions occur at identical rates and (2) a 12-rate non-reversible model (NREV12) that is applicable to analysing mutational processes in single-stranded (ss) genomes, in that all substitution types are free to occur at different rates. Using likelihood ratio and Akaike information criterion-based model tests, we show that, surprisingly, NREV12 provided a significantly better fit than the general time reversible (GTR) and NREV6 models to 21/31 dsRNA and 20/30 dsDNA datasets. As expected, however, NREV12 provided a significantly better fit to 24/33 ssDNA and 40/47 ssRNA datasets. We tested how non-reversibility impacts the accuracy with which phylogenetic trees are inferred. As simulated degrees of non-reversibility (DNRs) increased, the tree topology inferences using both NREV12 and GTR became more accurate, whereas inferred tree branch lengths became less accurate. We conclude that while non-reversible models should be helpful in the analysis of mutational processes in most virus species, there is no pressing need to use these models for routine phylogenetic inference.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>reversibility</kwd><kwd>mutations</kwd><kwd>12-rate nucelotide substitution model</kwd><kwd>non-reversibility</kwd><kwd>models of evolution</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Viruses</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution>South African Centre for Epidemiological Modelling and Analysis</institution></institution-wrap></funding-source><award-id>PhD Study Bursary</award-id><principal-award-recipient><name><surname>Sianga-Mete</surname><given-names>Rita</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>U.S. National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01 AI134384</award-id><principal-award-recipient><name><surname>Kosakovsky Pond</surname><given-names>Sergei L</given-names></name><name><surname>Weaver</surname><given-names>Steven</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>U.S. National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01 AI140970</award-id><principal-award-recipient><name><surname>Kosakovsky Pond</surname><given-names>Sergei L</given-names></name><name><surname>Weaver</surname><given-names>Steven</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>U.S. National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM144468</award-id><principal-award-recipient><name><surname>Kosakovsky Pond</surname><given-names>Sergei L</given-names></name><name><surname>Weaver</surname><given-names>Steven</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/029chgv08</institution-id><institution>Wellcome Trust</institution></institution-wrap></funding-source><award-id award-id-type="doi">10.35802/222574</award-id><principal-award-recipient><name><surname>Martin</surname><given-names>Darren P</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication. For the purpose of Open Access, the authors have applied a CC BY public copyright license to any Author Accepted Manuscript version arising from this submission.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Non-reversible nucleotide substitution models best describe strand-specific nucleotide substitution biases of viral genomes.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Modelling the nucleotide substitution processes that underlie the diversification of virus genome sequences lies at the heart of many viral evolutionary analyses. The most widely used nucleotide substitution models belong to the general time reversible (GTR) family (<xref ref-type="bibr" rid="bib39">Tavaré, 1986</xref>) and assume that the Markov process of evolution is time-reversible (<xref ref-type="bibr" rid="bib15">Hoff et al., 2016</xref>; <xref ref-type="bibr" rid="bib20">Liò and Goldman, 1998</xref>; <xref ref-type="bibr" rid="bib39">Tavaré, 1986</xref>).</p><p>The GTR model is defined by its instantaneous rate matrix <inline-formula><alternatives><mml:math id="inf1"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi mathvariant="bold-italic">Q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mi mathvariant="bold-italic">j</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft1">\begin{document}$\boldsymbol {Q_{ij}}$\end{document}</tex-math></alternatives></inline-formula> (<xref ref-type="disp-formula" rid="equ1">Equation 1</xref>), where <inline-formula><alternatives><mml:math id="inf2"><mml:mstyle><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">Q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mi mathvariant="bold-italic">j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math><tex-math id="inft2">\begin{document}$\boldsymbol {Q_{ij}}$\end{document}</tex-math></alternatives></inline-formula> defines the instantaneous rate of change from nucleotide <inline-formula><alternatives><mml:math id="inf3"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mo>∈</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi mathvariant="bold-italic">A</mml:mi><mml:mo mathvariant="bold">,</mml:mo><mml:mi mathvariant="bold-italic">C</mml:mi><mml:mo mathvariant="bold">,</mml:mo><mml:mi mathvariant="bold-italic">G</mml:mi><mml:mo mathvariant="bold">,</mml:mo><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft3">\begin{document}$\boldsymbol i\in \left\{\boldsymbol {A,C,G,T}\right\}$\end{document}</tex-math></alternatives></inline-formula> to nucleotide <inline-formula><alternatives><mml:math id="inf4"><mml:mstyle><mml:mrow><mml:mi mathvariant="bold-italic">j</mml:mi></mml:mrow></mml:mstyle></mml:math><tex-math id="inft4">\begin{document}$\boldsymbol j$\end{document}</tex-math></alternatives></inline-formula>; subject to the detailed balance condition: <inline-formula><alternatives><mml:math id="inf5"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">j</mml:mi><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi mathvariant="bold-italic">π</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">j</mml:mi><mml:mi mathvariant="bold-italic">i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi mathvariant="bold-italic">π</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft5">\begin{document}$\boldsymbol {q_{ji}\pi _{i}}=\boldsymbol {q_{ji}\pi _{j}}$\end{document}</tex-math></alternatives></inline-formula>, with rates <bold>q</bold> and equilibrium frequencies <bold>π</bold> (<xref ref-type="bibr" rid="bib37">Squartini and Arndt, 2008</xref>; <xref ref-type="bibr" rid="bib30">Posada, 2003</xref>). The instantaneous rate matrix of the GTR model includes six rate parameters (<italic>a</italic>, <italic>b</italic>, <italic>c</italic>, <italic>d</italic>, <italic>e</italic>, and <italic>f</italic>). Because only products of substitution rates and evolutionary times can be estimated, one of the rate parameters is set to 1 (e.g. <italic>b</italic>), or the entire matrix is normalised to yield one expected substitution per unit time.<disp-formula id="equ1"><label>(1)</label><alternatives><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="italic">Q</mml:mi></mml:mrow><mml:mo mathvariant="bold">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">i</mml:mi><mml:mi mathvariant="bold-italic">j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mo mathvariant="bold">=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>a</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>b</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>a</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>c</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>e</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>c</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>f</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>e</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>f</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="t1">\begin{document}$$\displaystyle \boldsymbol{{\it Q}=\left\{q_{i j}\right\}=\left(\begin{array}{cccc}- &amp; a \pi_C &amp; b \pi_G &amp; d \pi_T \\ a \pi_A &amp; - &amp; c \pi_G &amp; e \pi_T \\ b \pi_A &amp; c \pi_C &amp; - &amp; f \pi_T \\ d \pi_A &amp; e \pi_C &amp; f \pi_G &amp; - \end{array}\right)}$$\end{document}</tex-math></alternatives></disp-formula></p><p>The rate matrix in <xref ref-type="disp-formula" rid="equ1">Equation 1</xref> is symmetrical, e.g., the relative rate at which A changes to G is the same as the relative rate at which G changes to A.</p><p>Time-reversible nucleotide substitution models such as GTR form the basis of almost all nucleotide sequence-focused evolutionary analyses (including those involving eukaryotes, prokaryotes, and viruses) (<xref ref-type="bibr" rid="bib19">Lefort et al., 2017</xref>; <xref ref-type="bibr" rid="bib28">Posada and Crandall, 2001a</xref>; <xref ref-type="bibr" rid="bib29">Posada and Crandall, 2001b</xref>; <xref ref-type="bibr" rid="bib21">Minin et al., 2003</xref>).</p><p>The reliability of a phylogenetic tree constructed using a particular nucleotide sequence dataset should be maximised when the evolutionary models used to construct the tree accurately reflect the important aspects of the evolutionary process (<xref ref-type="bibr" rid="bib4">Buckley and Cunningham, 2002</xref>; <xref ref-type="bibr" rid="bib31">Ripplinger and Sullivan, 2008</xref>; <xref ref-type="bibr" rid="bib15">Hoff et al., 2016</xref>). The suitability of different models for describing the evolution of DNA or RNA sequences is, therefore, expected to depend to some degree on the biological and environmental contexts of the sequences being analysed.</p><p>Mutations in viral genomes arise due to diverse biotic (such as replication enzyme infidelities, RNA/DNA editing enzymes) and abiotic (such as ionising radiation, inorganic oxidisers, and chemical mutagens) factors (<xref ref-type="bibr" rid="bib34">Sanjuán and Domingo-Calap, 2016</xref>). Mutagenic chemical reactions or types of radiation that, for example, cause G to A or C to U mutations in DNA or RNA are not the same as those that cause A to G or U to C mutations (<xref ref-type="bibr" rid="bib6">Cheng et al., 1992</xref>;⁠ <xref ref-type="bibr" rid="bib22">Nguyen et al., 1992</xref>; <xref ref-type="bibr" rid="bib5">Chelico et al., 2006</xref>; <xref ref-type="bibr" rid="bib36">Sharma et al., 2016</xref>). It should not be expected, therefore, that the relative rates of G to A substitution will equal the relative rates of A to G substitution. Instead, in evolving double-stranded (ds) DNA and dsRNA molecules where both strands of the genome are in existence for similar amounts of time, both G to A and C to T substitutions should occur at relatively similar rates. Therefore, for nucleotide sequence datasets derived from any organisms with dsDNA or dsRNA genomes, a non-reversible nucleotide substitution model with a different relative substitution rate category for each of the six possible pairs of complementary nucleotide substitutions (e.g. NREV6 in <xref ref-type="disp-formula" rid="equ2">Equation 2</xref>), with <inline-formula><alternatives><mml:math id="inf6"><mml:mstyle><mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">A</mml:mi><mml:mi mathvariant="bold-italic">C</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi><mml:mi mathvariant="bold-italic">G</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">,</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">A</mml:mi><mml:mi mathvariant="bold-italic">G</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi><mml:mi mathvariant="bold-italic">C</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">,</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">A</mml:mi><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">T</mml:mi><mml:mi mathvariant="bold-italic">A</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">,</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">C</mml:mi><mml:mi mathvariant="bold-italic">G</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">G</mml:mi><mml:mi mathvariant="bold-italic">C</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">,</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">C</mml:mi><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">G</mml:mi><mml:mi mathvariant="bold-italic">A</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">,</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">G</mml:mi><mml:mi mathvariant="bold-italic">T</mml:mi></mml:mrow></mml:msub><mml:mo mathvariant="bold">=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">q</mml:mi><mml:mrow><mml:mi mathvariant="bold-italic">C</mml:mi><mml:mi mathvariant="bold-italic">A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math><tex-math id="inft6">\begin{document}${\boldsymbol {q_{AC}}}={\boldsymbol {q_{TG},q_{AG}=q_{TC},q_{AT}=q_{TA},q_{CG}=q_{GC},q_{CT}=q_{GA}, q_{GT}=q_{CA}}}$\end{document}</tex-math></alternatives></inline-formula>, might plausibly provide a better description of mutational processes than GTR (<xref ref-type="bibr" rid="bib1">Baele et al., 2010</xref>; <xref ref-type="bibr" rid="bib43">Wickner, 1993</xref>).<disp-formula id="equ2"><label>(2)</label><alternatives><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi mathvariant="italic">Q</mml:mi><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi mathvariant="italic">q</mml:mi><mml:mrow><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>a</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>b</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>c</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>e</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>e</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>f</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>c</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>b</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>a</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="t2">\begin{document}$$\displaystyle \boldsymbol{\it Q=\left\{q_{i j}\right\}=\left(\begin{array}{cccc}- &amp; a \pi_C &amp; b \pi_G &amp; c \pi_T \\ f \pi_A &amp; - &amp; d \pi_G &amp; e \pi_T \\ e \pi_A &amp; d \pi_C &amp; - &amp; f \pi_T \\ c \pi_A &amp; b \pi_C &amp; a \pi_G &amp; - \end{array}\right)}$$\end{document}</tex-math></alternatives></disp-formula></p><p>In the case of ssRNA viruses, ssDNA viruses, retroviruses, and dsRNA/dsDNA viruses where the two complementary genome strands do not exist for equal amounts of time (<xref ref-type="bibr" rid="bib46">Yu et al., 2004</xref>), a model where all 12 different substitutions occur at different rates might be best. Specifically, with ssRNA viruses, ssDNA viruses, and retroviruses, only one of the genome strands (called the virion strand) is packaged into viral particles for transmission and, in many dsRNA viruses, the genome strand that is translated into proteins (called the + strand) exists for longer during the life cycle than does the complementary (or –) strand (<xref ref-type="bibr" rid="bib3">Bruslind, 2020</xref>; <xref ref-type="bibr" rid="bib24">Onwubiko et al., 2020</xref>). In all these viruses, some degree of strand-specific substitution bias is expected to occur (<xref ref-type="bibr" rid="bib40">van der Walt et al., 2008</xref>; <xref ref-type="bibr" rid="bib26">Polak and Arndt, 2008</xref>) such that NREV6 might be anticipated to provide a poorer description of mutational processes than a model such as NREV12 (<xref ref-type="disp-formula" rid="equ3">Equation 3</xref>), where each of the 12 different types of substitution has a separate rate (<xref ref-type="bibr" rid="bib1">Baele et al., 2010</xref>).<disp-formula id="equ3"><label>(3)</label><alternatives><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="italic">Q</mml:mi><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi mathvariant="italic">q</mml:mi><mml:mrow><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>a</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>b</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>c</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>g</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>e</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>C</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>h</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>i</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mi>f</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>j</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>k</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mi>l</mml:mi><mml:msub><mml:mi>π</mml:mi><mml:mi>T</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math><tex-math id="t3">\begin{document}$$\displaystyle \boldsymbol {\it Q=\left\{q_{i j}\right\}=\left(\begin{array}{cccc} - &amp; a \pi_A &amp; b \pi_A &amp; c \pi_A \\ g \pi_C &amp; - &amp; d \pi_C &amp; e \pi_C \\ h \pi_G &amp; i \pi_G &amp; - &amp; f \pi_G \\ j \pi_T &amp; k \pi_T &amp; l \pi_T &amp; -\end{array}\right)}$$\end{document}</tex-math></alternatives></disp-formula></p><p>Because non-reversible models consider the directionality of evolution, they could, in some cases, be used to identify root nodes of phylogenetic trees (<xref ref-type="bibr" rid="bib45">Yap and Speed, 2005</xref>; <xref ref-type="bibr" rid="bib2">Boussau and Gouy, 2006</xref>). It is, however, unclear whether non-reversible models might, in certain situations at least, perform better than reversible models in the context of phylogenetic inference. Although it is possible to use non-reversible nucleotide substitution models such as NREV6 and NREV12 during maximum likelihood-based phylogenetic inference with computer programs such as IQ-TREE (<xref ref-type="bibr" rid="bib23">Nguyen et al., 2015</xref>), these models are not routinely used for phylogenetic inference. This is in part because non-reversible models render several commonly used algorithmic techniques for efficient likelihood computation inapplicable, making inference slower. It is also in part because it remains undetermined whether, under conditions where strand-specific substitution biases are evident, non-reversible models consistently yield substantially more accurate phylogenetic trees than reversible models.</p><p>Here, we present evidence that strand-specific nucleotide substitution biases are common within virus genomic sequence datasets such that NREV12 generally provides a significantly better fit than both GTR and NREV6 for such datasets. We then use simulations to demonstrate that whereas strand-specific nucleotide substitution biases reduce the accuracy of phylogenetic inference under both GTR and NREV12, when these biases become extreme, use of NREV12 can yield significantly more accurate phylogenetic trees than GTR.</p></sec><sec id="s2" sec-type="results|discussion"><title>Results and discussion</title><sec id="s2-1"><title>Non-reversible nucleotide substitution models generally provide a better fit than reversible models to virus sequence datasets</title><p>We tested for evidence of non-reversibility of the nucleotide substitution process in 141 virus sequence datasets (33 ssDNA virus datasets, 30 dsDNA virus datasets, 31 dsRNA virus datasets, and 47 ssRNA virus datasets), all consisting of either full genome sequences (for unsegmented viruses) or complete genome component sequences (for viruses with segmented genomes). Specifically, for each dataset, we compared the goodness-of-fit of the GTR+G, NREV6+G, and NREV12+G models (where G represents gamma-distributed nucleotide substitution rates among sites; <xref ref-type="bibr" rid="bib44">Yang, 1994</xref>).</p><p>Given that dsDNA viruses such as adenoviruses, papillomaviruses, and herpesviruses have both their DNA strands in existence for similar amounts of time before DNA-dependent-DNA polymerase enzymes copy both their + and – DNA strands during replication (<xref ref-type="bibr" rid="bib13">Hanson, 2009</xref>), we had anticipated that the best fitting substitution model for sequence datasets of these viruses would be NREV6. Using weighted small sample corrected Akaike information criterion (AIC-c) scores to reveal trends of model support (<xref ref-type="fig" rid="fig1">Figure 1</xref>), it is surprising that NREV12 was overall the best-supported model (illustrated by the redder hues around the top corner of the dsDNA plot in <xref ref-type="fig" rid="fig2">Figure 2</xref>). Out of the 30 dsDNA datasets considered, we found that NREV6 provided the best fit to five datasets (HPV18, HPV45, HPV16, BPV, and SV40) and GTR provided the best fit to five (Alphapapillomavirus 6, JC polyomavirus, DPV, RTBV, and DBAV). NREV12 was the best fitting model for the remaining 20 datasets (<xref ref-type="table" rid="table1">Table 1</xref>). Further, likelihood ratio tests (LRTs) revealed strong overall support for NREV12, with this model providing a significantly better fit (p&lt;0.05) than NREV6 for 25/30 of the dsDNA datasets and a significantly better fit than GTR for 24/30 of the datasets.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Ternary plots illustrating the relative fit of the NREV12, NREV6, and GTR nucleotide substitution models based on weighted small sample corrected Akaike information criterion (AIC-c) scores for 30 dsDNA, 31 dsRNA, 33 ssDNA, and 47 ssRNA virus nucleotide sequence datasets.</title><p>These plots were produced using the Akaike weights function with an overlaid density function (implemented in the qpcR package of RStudio; <xref ref-type="bibr" rid="bib32">Ritz and Spiess, 2008</xref>) to indicate point densities. Each model is represented by a corner of the triangles, and each circle represents the relative fit of each of the three models to a single nucleotide sequence dataset. The sides of the triangle represent model support axes ranging from 0% to 100%, with the position of a circle in relation to each of the sides of the triangle indicating the probability of models best describing the nucleotide sequence dataset that is represented by that point. Red colours represent a very high density of nucleotide sequence datasets that favour a particular model, blue colours indicate a lower, but still substantial, density of datasets that favour a particular model.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87361-fig1-v1.tif"/></fig><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Weighted Robinson-Foulds distances between inferred and true phylogenetic trees for datasets simulated with different degrees of nucleotide substitution non-reversibility and different average pairwise sequence identities (APIs) (~75%, ~80%, ~85%, ~90%, and ~95%).</title><p>‘ns’ above a pair of box and whisker plots indicates a paired t-test adjusted p-value of ≥0.05 and ‘*’ indicates a paired t-test adjusted p-value of &lt;0.05.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87361-fig2-v1.tif"/></fig><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Akaike information criterion (AIC) scores and likelihood ratio test (LRT) results for double-stranded DNA virus datasets.</title><p>The lowest small sample corrected AIC (AIC-c) scores indicating the best fitting models are in bold.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Virus family</th><th align="left" valign="bottom">Dataset</th><th align="left" valign="bottom">AIC score GTR</th><th align="left" valign="bottom">AIC score NREV-6</th><th align="left" valign="bottom">AIC score NREV-12</th><th align="left" valign="bottom">p-Value GTR vs NREV-12</th><th align="left" valign="bottom">p-Value NREV-6 vs NREV-12</th><th align="left" valign="bottom">DNR</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="10">Papillomaviridae</td><td align="left" valign="bottom">APPV 6</td><td align="left" valign="bottom"><bold>35099.5</bold></td><td align="left" valign="bottom">35108.0</td><td align="left" valign="bottom">35102.2</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.007</bold></td><td align="left" valign="bottom">0.089</td></tr><tr><td align="left" valign="bottom">HPV18_2</td><td align="left" valign="bottom">25202.9</td><td align="left" valign="bottom"><bold>25174.6</bold></td><td align="left" valign="bottom">25179.2</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.323</td></tr><tr><td align="left" valign="bottom">HPV45_2</td><td align="left" valign="bottom">23600.6</td><td align="left" valign="bottom"><bold>23599.0</bold></td><td align="left" valign="bottom">23602.9</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.285</td></tr><tr><td align="left" valign="bottom">HPV16_2</td><td align="left" valign="bottom">29734.0</td><td align="left" valign="bottom"><bold>29664.5</bold></td><td align="left" valign="bottom">29665.4</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.371</td></tr><tr><td align="left" valign="bottom">HPV31</td><td align="left" valign="bottom">24681.4</td><td align="left" valign="bottom">24677.3</td><td align="left" valign="bottom"><bold>24672.8</bold></td><td align="left" valign="bottom"><bold>0.002</bold></td><td align="left" valign="bottom"><bold>0.01</bold></td><td align="left" valign="bottom">0.165</td></tr><tr><td align="left" valign="bottom">HPV6_1</td><td align="left" valign="bottom">31199.1</td><td align="left" valign="bottom">31150.0</td><td align="left" valign="bottom"><bold>31141.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.451</td></tr><tr><td align="left" valign="bottom">LPV</td><td align="left" valign="bottom">67165.7</td><td align="left" valign="bottom">67188.1</td><td align="left" valign="bottom"><bold>67145.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.42</td></tr><tr><td align="left" valign="bottom">DPV</td><td align="left" valign="bottom"><bold>69829.7</bold></td><td align="left" valign="bottom">69889.2</td><td align="left" valign="bottom">69835.1</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.056</td></tr><tr><td align="left" valign="bottom">XPV</td><td align="left" valign="bottom">95455.6</td><td align="left" valign="bottom">95617.1</td><td align="left" valign="bottom"><bold>95452.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.072</td></tr><tr><td align="left" valign="bottom">BATV</td><td align="left" valign="bottom">134821</td><td align="left" valign="bottom">134511</td><td align="left" valign="bottom"><bold>133322</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.402</td></tr><tr><td align="left" valign="bottom" rowspan="4">Polyomaviridae</td><td align="left" valign="bottom">JC_2</td><td align="left" valign="bottom"><bold>51806.7</bold></td><td align="left" valign="bottom">51819.6</td><td align="left" valign="bottom">51812.0</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.003</bold></td><td align="left" valign="bottom">0.089</td></tr><tr><td align="left" valign="bottom">BK_2</td><td align="left" valign="bottom">21472.6</td><td align="left" valign="bottom">21472.7</td><td align="left" valign="bottom"><bold>21471.1</bold></td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom">0.244</td></tr><tr><td align="left" valign="bottom">SV40</td><td align="left" valign="bottom">16859.8</td><td align="left" valign="bottom"><bold>16858.0</bold></td><td align="left" valign="bottom">16858.4</td><td align="left" valign="bottom"><bold>0.037</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.567</td></tr><tr><td align="left" valign="bottom">BPV</td><td align="left" valign="bottom">148614.9</td><td align="left" valign="bottom"><bold>148573.8</bold></td><td align="left" valign="bottom">148585.2</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.064</td></tr><tr><td align="left" valign="bottom" rowspan="6">Caulimoviridae</td><td align="left" valign="bottom">CMV</td><td align="left" valign="bottom">124083.9</td><td align="left" valign="bottom">124221.0</td><td align="left" valign="bottom"><bold>123888.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.351</td></tr><tr><td align="left" valign="bottom">CSSV</td><td align="left" valign="bottom">145327.0</td><td align="left" valign="bottom">146575</td><td align="left" valign="bottom"><bold>145202</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.158</td></tr><tr><td align="left" valign="bottom">SVBV</td><td align="left" valign="bottom">138575</td><td align="left" valign="bottom">138488.1</td><td align="left" valign="bottom"><bold>138464.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.174</td></tr><tr><td align="left" valign="bottom">DBAV</td><td align="left" valign="bottom"><bold>46495.5</bold></td><td align="left" valign="bottom">46514.1</td><td align="left" valign="bottom">46502.0</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.0335</td></tr><tr><td align="left" valign="bottom">RTBV</td><td align="left" valign="bottom"><bold>54987.9</bold></td><td align="left" valign="bottom">55350.1</td><td align="left" valign="bottom">54991</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.082</td></tr><tr><td align="left" valign="bottom">BDV</td><td align="left" valign="bottom">376325.2</td><td align="left" valign="bottom">376647.6</td><td align="left" valign="bottom"><bold>376029.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.140</td></tr><tr><td align="left" valign="bottom">Siphoviridae</td><td align="left" valign="bottom">CLV</td><td align="left" valign="bottom">237362.3</td><td align="left" valign="bottom">237351.8</td><td align="left" valign="bottom"><bold>237348.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.01</bold></td><td align="left" valign="bottom">0.070</td></tr><tr><td align="left" valign="bottom">Tectiviridae</td><td align="left" valign="bottom">TTIV</td><td align="left" valign="bottom">913864.9</td><td align="left" valign="bottom">913915.4</td><td align="left" valign="bottom"><bold>913773.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.279</td></tr><tr><td align="left" valign="bottom" rowspan="8">Adenoviridae</td><td align="left" valign="bottom">FAV_C</td><td align="left" valign="bottom">3074086.7</td><td align="left" valign="bottom">3074207.5</td><td align="left" valign="bottom"><bold>3073739.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.169</td></tr><tr><td align="left" valign="bottom">FAV_E</td><td align="left" valign="bottom">103482.3</td><td align="left" valign="bottom">103222.7</td><td align="left" valign="bottom"><bold>102636.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.357</td></tr><tr><td align="left" valign="bottom">FAV_D</td><td align="left" valign="bottom">2326925.6</td><td align="left" valign="bottom">2325719.4</td><td align="left" valign="bottom"><bold>2324784.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.551</td></tr><tr><td align="left" valign="bottom">FAV_A</td><td align="left" valign="bottom">705328.5</td><td align="left" valign="bottom">705436.5</td><td align="left" valign="bottom"><bold>705197.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.645</td></tr><tr><td align="left" valign="bottom">HMAV_B</td><td align="left" valign="bottom">103796.7</td><td align="left" valign="bottom">103937.44</td><td align="left" valign="bottom"><bold>103753.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">10.890</td></tr><tr><td align="left" valign="bottom">HMAV_D</td><td align="left" valign="bottom">1748635.2</td><td align="left" valign="bottom">1749769</td><td align="left" valign="bottom"><bold>1748119.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.646</td></tr><tr><td align="left" valign="bottom">HMAV_C</td><td align="left" valign="bottom">2851144.5</td><td align="left" valign="bottom">2851357.1</td><td align="left" valign="bottom"><bold>2851133</bold></td><td align="left" valign="bottom"><bold>0.006</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.0225</td></tr><tr><td align="left" valign="bottom">HMAV_E</td><td align="left" valign="bottom">1915044.8</td><td align="left" valign="bottom">1915065.3</td><td align="left" valign="bottom"><bold>1914998</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.049</td></tr></tbody></table></table-wrap><p>As NREV6 was not the best fitting model for most of the dsDNA virus datasets, we infer that, in most dsDNA virus species, strand-specific substitution biases are not ignorable. Further, the datasets where NREV6 was not the best fit are from species in families containing other species where NREV6 was the best fit, indicating that such strand-specific substitution biases are unlikely to be a consequence of some broadly conserved feature of viral life cycles in these families (such as ssDNA replicative intermediates). It is instead plausible that these differences may relate to:</p><list list-type="roman-lower" id="list1"><list-item><p>differences in replication fidelity and/or proofreading efficiency on the leading and trailing DNA strands in some dsDNA virus species (<xref ref-type="bibr" rid="bib12">Grigoriev, 1999</xref>). These differences are common in eukaryotes (<xref ref-type="bibr" rid="bib25">Pavlov et al., 2002</xref>; <xref ref-type="bibr" rid="bib11">Furusawa, 2012</xref>) and prokaryotes (<xref ref-type="bibr" rid="bib10">Fijalkowska et al., 1998</xref>) and, considering that the replication processes of dsDNA viruses analysed here mirror those of their eukaryote hosts, it is perhaps unsurprising that most of these viruses also display some evidence of strand-specific substitution biases.</p></list-item><list-item><p>extra exposure to DNA damage of displaced template strands during unidirectional rolling circle replication in some papillomavirus species such as HPV16 could be a contributor to strand-specific nucleotide substitution biases (<xref ref-type="bibr" rid="bib17">Kusumoto-Matsuo et al., 2011</xref>).</p></list-item><list-item><p>extra time spent by non-coding strands in single-stranded dissociated states during RNA transcription in some papillomavirus and polyomavirus species (<xref ref-type="bibr" rid="bib9">Fernandes and Medeiros Fernandes, 2012</xref>): during transcription processes, the dissociated non-coding strand is transiently more exposed to damage than the coding strand (<xref ref-type="bibr" rid="bib42">Wei et al., 2010</xref>), which might also contribute to strand-specific substitution biases.</p></list-item></list><p>Similarly, and equally surprising, we found that NREV12 was overall the best supported model for dsRNA viruses (illustrated by the redder hues around the top corner of the dsRNA plot in <xref ref-type="fig" rid="fig1">Figure 1</xref>).</p><p>NREV6 fit only 2 of the 31 dsRNA datasets better than both NREV12 and GTR (Human rotavirus A set H and Fiji virus). NREV12 was found to be the best fitting model for 21/31 datasets and GTR was the best fitting of 8/31 (<xref ref-type="table" rid="table2">Table 2</xref>). In all three Birnaviridae family datasets (which contain virus species with two genome segments) and in 17/22 of Reoviridae family datasets (which contain virus species with 10–12 genome segments), the NREV12 model provided the best fit. Based on the LRTs, strong overall support for NREV12 was found, with this model providing a significantly better fit (p&lt;0.05) in 27/31 dsRNA virus datasets relative to NREV6 and 23/31 datasets relative to GTR.</p><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>Akaike information criterion (AIC) scores and likelihood ratio test (LRT) results for double-stranded RNA datasets.</title><p>The lowest small sample corrected AIC (AIC-c) scores indicating the best fitting models are in bold.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Virus family</th><th align="left" valign="bottom">Dataset</th><th align="left" valign="bottom">AIC score GTR</th><th align="left" valign="bottom">AIC score NREV-6</th><th align="left" valign="bottom">AIC score NREV-12</th><th align="left" valign="bottom">GTR vs NREV-12</th><th align="left" valign="bottom">NREV-6 vs NREV-12</th><th align="left" valign="bottom">DNR</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="4">Birnaviridae</td><td align="left" valign="bottom">AQBV</td><td align="left" valign="bottom">31754.9</td><td align="left" valign="bottom">31853.3</td><td align="left" valign="bottom"><bold>31721.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.219</td></tr><tr><td align="left" valign="bottom">GBV_A</td><td align="left" valign="bottom">47176.9</td><td align="left" valign="bottom">47347.2</td><td align="left" valign="bottom"><bold>47154.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.142</td></tr><tr><td align="left" valign="bottom">IPNV</td><td align="left" valign="bottom">79186.2</td><td align="left" valign="bottom">79221.9</td><td align="left" valign="bottom"><bold>79182.4</bold></td><td align="left" valign="bottom"><bold>0.0145</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.162</td></tr><tr><td align="left" valign="bottom">GBV_B</td><td align="left" valign="bottom">39313.7</td><td align="left" valign="bottom">39062.8</td><td align="left" valign="bottom"><bold>38938.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.201</td></tr><tr><td align="left" valign="bottom" rowspan="22">Reoviridae</td><td align="left" valign="bottom">BTV_A</td><td align="left" valign="bottom">34803.5</td><td align="left" valign="bottom">34895.1</td><td align="left" valign="bottom"><bold>34801.3</bold></td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.042</td></tr><tr><td align="left" valign="bottom">BTV_B</td><td align="left" valign="bottom">48849.9</td><td align="left" valign="bottom">48893.</td><td align="left" valign="bottom"><bold>48837.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.043</td></tr><tr><td align="left" valign="bottom">BTV_C</td><td align="left" valign="bottom">28350.9</td><td align="left" valign="bottom">28386.5</td><td align="left" valign="bottom"><bold>28350.8</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.061</td></tr><tr><td align="left" valign="bottom">BTV_D</td><td align="left" valign="bottom">24969.1</td><td align="left" valign="bottom">24947.3</td><td align="left" valign="bottom"><bold>24894.0</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.191</td></tr><tr><td align="left" valign="bottom">BTV_F</td><td align="left" valign="bottom">20622.7</td><td align="left" valign="bottom">20708.5</td><td align="left" valign="bottom"><bold>20610.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.067</td></tr><tr><td align="left" valign="bottom">BTV_G</td><td align="left" valign="bottom">63349.9</td><td align="left" valign="bottom">63485.0</td><td align="left" valign="bottom"><bold>63345.9</bold></td><td align="left" valign="bottom"><bold>0.00426</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.040</td></tr><tr><td align="left" valign="bottom">BTV_H</td><td align="left" valign="bottom">20596.7</td><td align="left" valign="bottom">20685.5</td><td align="left" valign="bottom"><bold>20586.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.118</td></tr><tr><td align="left" valign="bottom">BTV_I</td><td align="left" valign="bottom">17592.7</td><td align="left" valign="bottom">17622.5</td><td align="left" valign="bottom"><bold>17588.8</bold></td><td align="left" valign="bottom"><bold>0.01</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.095</td></tr><tr><td align="left" valign="bottom">BRVA_C</td><td align="left" valign="bottom">41206.7</td><td align="left" valign="bottom">41187.4</td><td align="left" valign="bottom"><bold>41137.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.128</td></tr><tr><td align="left" valign="bottom">HRVA_A</td><td align="left" valign="bottom"><bold>17030.5</bold></td><td align="left" valign="bottom">17043.2</td><td align="left" valign="bottom">17035.5</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.003</bold></td><td align="left" valign="bottom">0.036</td></tr><tr><td align="left" valign="bottom">HRVA_B</td><td align="left" valign="bottom"><bold>8275.1</bold></td><td align="left" valign="bottom">8280.3</td><td align="left" valign="bottom">8281.7</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.087</td></tr><tr><td align="left" valign="bottom">HRVA_C</td><td align="left" valign="bottom">12815.1</td><td align="left" valign="bottom">12842.6</td><td align="left" valign="bottom"><bold>12807.6</bold></td><td align="left" valign="bottom"><bold>0.003</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.132</td></tr><tr><td align="left" valign="bottom">HRVA_D2</td><td align="left" valign="bottom"><bold>8036.8</bold></td><td align="left" valign="bottom">8041.0</td><td align="left" valign="bottom">8043.7</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.057</td></tr><tr><td align="left" valign="bottom">HRVA_E</td><td align="left" valign="bottom"><bold>7045.9</bold></td><td align="left" valign="bottom">7056.1</td><td align="left" valign="bottom">7053.3</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.02</bold></td><td align="left" valign="bottom">0.102</td></tr><tr><td align="left" valign="bottom">HRVA_F</td><td align="left" valign="bottom"><bold>7046.0</bold></td><td align="left" valign="bottom">7056.7</td><td align="left" valign="bottom">7053.4</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.02</bold></td><td align="left" valign="bottom">0.0710</td></tr><tr><td align="left" valign="bottom">HRVA_G</td><td align="left" valign="bottom">18424.2</td><td align="left" valign="bottom">18434.0</td><td align="left" valign="bottom">18425.1</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.123</td></tr><tr><td align="left" valign="bottom">HRVA_H</td><td align="left" valign="bottom">20431.4</td><td align="left" valign="bottom"><bold>20413.8</bold>7</td><td align="left" valign="bottom">20420.5.6</td><td align="left" valign="bottom"><bold>0.002</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.163</td></tr><tr><td align="left" valign="bottom">PRVA_A</td><td align="left" valign="bottom">28540.7</td><td align="left" valign="bottom">28441.9</td><td align="left" valign="bottom"><bold>28398.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.204</td></tr><tr><td align="left" valign="bottom">PRVA_B</td><td align="left" valign="bottom">14757.7</td><td align="left" valign="bottom">14775.5</td><td align="left" valign="bottom"><bold>14732.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.351</td></tr><tr><td align="left" valign="bottom">HRVC_A</td><td align="left" valign="bottom">6713.2</td><td align="left" valign="bottom">6718.2</td><td align="left" valign="bottom"><bold>6712.3</bold></td><td align="left" valign="bottom"><bold>0.045</bold></td><td align="left" valign="bottom"><bold>0.007</bold></td><td align="left" valign="bottom">0.124</td></tr><tr><td align="left" valign="bottom">PTOV</td><td align="left" valign="bottom">202011.3</td><td align="left" valign="bottom">202106.5</td><td align="left" valign="bottom"><bold>201878.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.039</td></tr><tr><td align="left" valign="bottom">FJV_B</td><td align="left" valign="bottom">9274.1</td><td align="left" valign="bottom"><bold>9250.0</bold></td><td align="left" valign="bottom">9250.9</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.194</td></tr><tr><td align="left" valign="bottom" rowspan="2">Endornaviridae</td><td align="left" valign="bottom">EDV</td><td align="left" valign="bottom">1771992.8</td><td align="left" valign="bottom">1772689.1</td><td align="left" valign="bottom"><bold>1771950.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.121</td></tr><tr><td align="left" valign="bottom">BPAV</td><td align="left" valign="bottom"><bold>70386.5</bold></td><td align="left" valign="bottom">70540.2</td><td align="left" valign="bottom">70390.7</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.00</bold></td><td align="left" valign="bottom">0.047</td></tr><tr><td align="left" valign="bottom" rowspan="2">Totiviridae</td><td align="left" valign="bottom">TTV</td><td align="left" valign="bottom">617302.6</td><td align="left" valign="bottom">617462.6</td><td align="left" valign="bottom"><bold>617172.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.052</td></tr><tr><td align="left" valign="bottom">GDV</td><td align="left" valign="bottom"><bold>80435.8</bold></td><td align="left" valign="bottom">80396.5</td><td align="left" valign="bottom">80387.7</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>0.002</bold></td><td align="left" valign="bottom">0.109</td></tr><tr><td align="left" valign="bottom">Hypoviridae</td><td align="left" valign="bottom">HPV</td><td align="left" valign="bottom">66859.8</td><td align="left" valign="bottom">66899.8</td><td align="left" valign="bottom"><bold>66857.8</bold></td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.057</td></tr></tbody></table></table-wrap><p>We anticipated that NREV12 might fit many of these dsRNA datasets better than NREV6 simply because, during their infection cycles, the coding +strand of dsRNA viruses (the one from which protein translation occurs) tends to exist for longer periods within an infected cell than the non-coding –strand. Specifically, there are two main steps during double-stranded RNA virus replication (<xref ref-type="bibr" rid="bib43">Wickner, 1993</xref>). Firstly, synthesis of the viral +strands from a dsRNA template occurs in the cytoplasm within viral particles. These +strands exist within the cell for prolonged periods in the absence of complementary –strands and are used as templates for translation of viral proteins. In the second step, the +strands remaining after translation act as templates for –strand synthesis, resulting in the formation of new dsRNA molecules. The +strands of dsRNA viruses are therefore likely more impacted by mutational processes, which in turn could explain the pervasive strand-specific substitution biases seen in this group of viruses.</p><p>For the ssRNA and ssDNA viruses where one genome strand exists during the virus life cycle for far longer periods of time than the other such that complementary substitutions would not be expected to occur at similar rates, we anticipated that NREV12 should provide a better fit than both NREV6 and GTR. Indeed, for ssRNA viruses, NREV12 was a better fit than NREV6 and GTR for 40/47 of the ssRNA datasets and 24/33 of the ssDNA virus datasets (<xref ref-type="fig" rid="fig1">Figure 1</xref>). Of the nine ssDNA virus datasets where NREV12 was not the best fitting model, GTR fit 7/9 better and NREV6 fit 2/9 better (<xref ref-type="table" rid="table3">Table 3</xref>). Of the seven ssRNA datasets where NREV12 was not the best fitting model, GTR fit 6/7 better and NREV6 fit 1/7 better.</p><table-wrap id="table3" position="float"><label>Table 3.</label><caption><title>Small sample corrected Akaike information criterion (AIC-c) scores and likelihood ratio test (LRT) results for single-stranded DNA datasets.</title><p>The lowest AIC-c scores indicating the best fitting models are in bold.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Virus family</th><th align="left" valign="bottom">Dataset</th><th align="left" valign="bottom">AIC score GTR</th><th align="left" valign="bottom">AIC score NREV-6</th><th align="left" valign="bottom">AIC score NREV-12</th><th align="left" valign="bottom">p-Value GTR vs NREV-12</th><th align="left" valign="bottom">p-Value NREV-6 vs NREV-12</th><th align="left" valign="bottom">DNR</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="8">Nanoviridae</td><td align="left" valign="bottom">BBTV M</td><td align="left" valign="bottom">15044.3</td><td align="left" valign="bottom">15207.9</td><td align="left" valign="bottom"><bold>14984.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.662</td></tr><tr><td align="left" valign="bottom">BBTV N</td><td align="left" valign="bottom">10605.6</td><td align="left" valign="bottom">10686.2</td><td align="left" valign="bottom"><bold>10595.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.533</td></tr><tr><td align="left" valign="bottom">BBTV R</td><td align="left" valign="bottom">18484.5</td><td align="left" valign="bottom">18544</td><td align="left" valign="bottom"><bold>18480.8</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.609</td></tr><tr><td align="left" valign="bottom">BBTV S</td><td align="left" valign="bottom">12718.9</td><td align="left" valign="bottom">12757.2</td><td align="left" valign="bottom"><bold>12707.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.728</td></tr><tr><td align="left" valign="bottom">CCDV</td><td align="left" valign="bottom"><bold>38622.7</bold></td><td align="left" valign="bottom">38632.0</td><td align="left" valign="bottom">38630.5</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom">0.050</td></tr><tr><td align="left" valign="bottom">MDV</td><td align="left" valign="bottom">36232.8</td><td align="left" valign="bottom"><bold>36063</bold></td><td align="left" valign="bottom">36064</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.142</td></tr><tr><td align="left" valign="bottom">PYDV</td><td align="left" valign="bottom">56138.4</td><td align="left" valign="bottom">56076.6</td><td align="left" valign="bottom"><bold>56056.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.187</td></tr><tr><td align="left" valign="bottom">FBNS</td><td align="left" valign="bottom">100153.6</td><td align="left" valign="bottom">100135.6</td><td align="left" valign="bottom"><bold>100120.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.098</td></tr><tr><td align="left" valign="bottom" rowspan="8">Geminiviridae</td><td align="left" valign="bottom">Begomo 5</td><td align="left" valign="bottom"><bold>28192.1</bold></td><td align="left" valign="bottom">28311.9</td><td align="left" valign="bottom">28192.5</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.1995</td></tr><tr><td align="left" valign="bottom">Begomo 6</td><td align="left" valign="bottom">16743.0</td><td align="left" valign="bottom"><bold>16722.6</bold></td><td align="left" valign="bottom">16724.1</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.214</td></tr><tr><td align="left" valign="bottom">Begomo 9</td><td align="left" valign="bottom">8517.6</td><td align="left" valign="bottom">8540.8</td><td align="left" valign="bottom"><bold>8515.6</bold></td><td align="left" valign="bottom"><bold>0.03</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.312</td></tr><tr><td align="left" valign="bottom">Dicot 1</td><td align="left" valign="bottom">44730.7</td><td align="left" valign="bottom">44594.3</td><td align="left" valign="bottom"><bold>44583.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.200</td></tr><tr><td align="left" valign="bottom">Dicot 2</td><td align="left" valign="bottom"><bold>39909.9</bold></td><td align="left" valign="bottom">39919.8</td><td align="left" valign="bottom">39917.9</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.100</td></tr><tr><td align="left" valign="bottom">MSV</td><td align="left" valign="bottom"><bold>252645.3</bold></td><td align="left" valign="bottom">254347.5</td><td align="left" valign="bottom">254347.5</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.144</td></tr><tr><td align="left" valign="bottom">PanSV</td><td align="left" valign="bottom">94601.2</td><td align="left" valign="bottom">94600.3</td><td align="left" valign="bottom"><bold>94593.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.182</td></tr><tr><td align="left" valign="bottom">WDV</td><td align="left" valign="bottom">35301.7</td><td align="left" valign="bottom">35313.2</td><td align="left" valign="bottom"><bold>35253.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.1033</td></tr><tr><td align="left" valign="bottom" rowspan="7">Circoviridae</td><td align="left" valign="bottom">BFDV</td><td align="left" valign="bottom">17256.7</td><td align="left" valign="bottom">17262.7</td><td align="left" valign="bottom"><bold>17246.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.224</td></tr><tr><td align="left" valign="bottom">DG_CV</td><td align="left" valign="bottom"><bold>12754.8</bold></td><td align="left" valign="bottom">12779.5</td><td align="left" valign="bottom">12758.3</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.116</td></tr><tr><td align="left" valign="bottom">PiCV</td><td align="left" valign="bottom"><bold>19180.5</bold></td><td align="left" valign="bottom">19192.5</td><td align="left" valign="bottom">19191.0</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>0.04</bold></td><td align="left" valign="bottom">0.117</td></tr><tr><td align="left" valign="bottom">CCCC</td><td align="left" valign="bottom">84435.7</td><td align="left" valign="bottom">84377.4</td><td align="left" valign="bottom"><bold>84315.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.132</td></tr><tr><td align="left" valign="bottom">BTC</td><td align="left" valign="bottom">262910.4</td><td align="left" valign="bottom">262060.1</td><td align="left" valign="bottom"><bold>261985.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.178</td></tr><tr><td align="left" valign="bottom">POCV2</td><td align="left" valign="bottom">24940.9</td><td align="left" valign="bottom">24953.8</td><td align="left" valign="bottom"><bold>24915.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.162</td></tr><tr><td align="left" valign="bottom">CCV</td><td align="left" valign="bottom">90307.9</td><td align="left" valign="bottom">90301.5</td><td align="left" valign="bottom"><bold>90285.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.114</td></tr><tr><td align="left" valign="bottom" rowspan="2">Anelloviridae</td><td align="left" valign="bottom">TTV_1</td><td align="left" valign="bottom">825811</td><td align="left" valign="bottom">826800</td><td align="left" valign="bottom"><bold>825292</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.513</td></tr><tr><td align="left" valign="bottom">TTSV</td><td align="left" valign="bottom">332287.9</td><td align="left" valign="bottom">332397.4</td><td align="left" valign="bottom"><bold>332258.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">1.560</td></tr><tr><td align="left" valign="bottom" rowspan="5">Parvoviridae</td><td align="left" valign="bottom">MVM</td><td align="left" valign="bottom">26756.3</td><td align="left" valign="bottom">26743.9</td><td align="left" valign="bottom"><bold>26686.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.148</td></tr><tr><td align="left" valign="bottom">HPV</td><td align="left" valign="bottom">67051.2</td><td align="left" valign="bottom">67080.1</td><td align="left" valign="bottom"><bold>67001.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.235</td></tr><tr><td align="left" valign="bottom">CPV</td><td align="left" valign="bottom">85731</td><td align="left" valign="bottom">85695</td><td align="left" valign="bottom"><bold>85689.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>0.007</bold></td><td align="left" valign="bottom">0.062</td></tr><tr><td align="left" valign="bottom">PPV</td><td align="left" valign="bottom">163006.8</td><td align="left" valign="bottom">163090.7</td><td align="left" valign="bottom"><bold>162995.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.143</td></tr><tr><td align="left" valign="bottom">CAV_P</td><td align="left" valign="bottom">37073.3</td><td align="left" valign="bottom">37115.5</td><td align="left" valign="bottom"><bold>37065.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.162</td></tr><tr><td align="left" valign="bottom">Microviridae</td><td align="left" valign="bottom">BMV</td><td align="left" valign="bottom">31175.3</td><td align="left" valign="bottom">31164.8</td><td align="left" valign="bottom"><bold>31147.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.188</td></tr><tr><td align="left" valign="bottom" rowspan="2">Pleolipoviridae</td><td align="left" valign="bottom">APV</td><td align="left" valign="bottom">85700.2</td><td align="left" valign="bottom">85617.4</td><td align="left" valign="bottom"><bold>85402.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.204</td></tr><tr><td align="left" valign="bottom">BPV</td><td align="left" valign="bottom">204797.5</td><td align="left" valign="bottom">204802.3</td><td align="left" valign="bottom"><bold>204796.7</bold></td><td align="left" valign="bottom"><bold>0.04</bold></td><td align="left" valign="bottom"><bold>0.007</bold></td><td align="left" valign="bottom">0.064</td></tr></tbody></table></table-wrap><p>Based on the LRTs, strong overall support for NREV12 was found with this model providing a significantly better fit (p&lt;0.05) than NREV6 for 45/47 of the ssRNA virus datasets (<xref ref-type="table" rid="table4">Table 4</xref>) and 31/33 of the ssDNA virus datasets (<xref ref-type="table" rid="table3">Table 3</xref>). Similarly, based on LRTs, NREV12 provided a significantly better fit than GTR for 27/33 of the ssDNA virus datasets (<xref ref-type="table" rid="table3">Table 3</xref>) and 40/47 of the ssRNA virus datasets (<xref ref-type="table" rid="table4">Table 4</xref>).</p><table-wrap id="table4" position="float"><label>Table 4.</label><caption><title>Small sample corrected Akaike information criterion (AIC-c) scores and likelihood ratio test (LRT) results for single-stranded RNA datasets.</title><p>The lowest AIC-c scores indicating the best fitting models are in bold.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Virus family</th><th align="left" valign="bottom">Dataset</th><th align="left" valign="bottom">AIC score GTR</th><th align="left" valign="bottom">AIC score NREV-6</th><th align="left" valign="bottom">AIC score NREV-12</th><th align="left" valign="bottom">p-Value GTR vs NREV-12</th><th align="left" valign="bottom">p-Value NREV-6 vs NREV-12</th><th align="left" valign="bottom">DNR</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="7">Astroviridae</td><td align="left" valign="bottom">HAV</td><td align="left" valign="bottom">94580.7</td><td align="left" valign="bottom">94926.3</td><td align="left" valign="bottom"><bold>94548.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.096</td></tr><tr><td align="left" valign="bottom">BAV</td><td align="left" valign="bottom">188307.1</td><td align="left" valign="bottom">188572.9</td><td align="left" valign="bottom"><bold>188144.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.108</td></tr><tr><td align="left" valign="bottom">MMV</td><td align="left" valign="bottom">281072.2</td><td align="left" valign="bottom">281094.5</td><td align="left" valign="bottom"><bold>281076.9</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.072</td></tr><tr><td align="left" valign="bottom">PAV</td><td align="left" valign="bottom">150626.88</td><td align="left" valign="bottom">150827.6</td><td align="left" valign="bottom"><bold>150609.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.069</td></tr><tr><td align="left" valign="bottom">CKV</td><td align="left" valign="bottom">90902.3</td><td align="left" valign="bottom">91233.1</td><td align="left" valign="bottom"><bold>90873.0</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.083</td></tr><tr><td align="left" valign="bottom">GA</td><td align="left" valign="bottom">64998.5</td><td align="left" valign="bottom">65223.9</td><td align="left" valign="bottom"><bold>64975.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.110</td></tr><tr><td align="left" valign="bottom">CAV_A</td><td align="left" valign="bottom">85558.8</td><td align="left" valign="bottom">85617.4</td><td align="left" valign="bottom"><bold>85547.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.01</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.076</td></tr><tr><td align="left" valign="bottom" rowspan="5">Bromoviridae</td><td align="left" valign="bottom">CMV RNA1</td><td align="left" valign="bottom">34197.5</td><td align="left" valign="bottom">34198.8</td><td align="left" valign="bottom"><bold>34147.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.124</td></tr><tr><td align="left" valign="bottom">CMV RNA2</td><td align="left" valign="bottom"><bold>31398.2</bold></td><td align="left" valign="bottom">31455.9</td><td align="left" valign="bottom">31388.7</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.091</td></tr><tr><td align="left" valign="bottom">CMV RNA3</td><td align="left" valign="bottom"><bold>24337.2</bold></td><td align="left" valign="bottom">24360.3</td><td align="left" valign="bottom">24343.9</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.073</td></tr><tr><td align="left" valign="bottom">AMS</td><td align="left" valign="bottom"><bold>24337.2</bold></td><td align="left" valign="bottom">24360.3</td><td align="left" valign="bottom">24343.9</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.073</td></tr><tr><td align="left" valign="bottom">PSV</td><td align="left" valign="bottom">67707</td><td align="left" valign="bottom">67786.5</td><td align="left" valign="bottom"><bold>67691</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.048</td></tr><tr><td align="left" valign="bottom" rowspan="3">Caliciviridae</td><td align="left" valign="bottom">LAV</td><td align="left" valign="bottom">73042.8</td><td align="left" valign="bottom">73102.4</td><td align="left" valign="bottom"><bold>72984.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.120</td></tr><tr><td align="left" valign="bottom">NoV</td><td align="left" valign="bottom">207667.2</td><td align="left" valign="bottom">207777.5</td><td align="left" valign="bottom"><bold>207660</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.047</td></tr><tr><td align="left" valign="bottom">VSV</td><td align="left" valign="bottom">235936.4</td><td align="left" valign="bottom">236051.4</td><td align="left" valign="bottom"><bold>235913.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.046</td></tr><tr><td align="left" valign="bottom">Closteroviridae</td><td align="left" valign="bottom">CTV</td><td align="left" valign="bottom"><bold>30062.2</bold></td><td align="left" valign="bottom">29980.4</td><td align="left" valign="bottom">29960.1</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.272</td></tr><tr><td align="left" valign="bottom" rowspan="2">Flaviviridae</td><td align="left" valign="bottom">DGV_T1</td><td align="left" valign="bottom">69771.9</td><td align="left" valign="bottom">70030.5</td><td align="left" valign="bottom"><bold>69776.2</bold></td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.063</td></tr><tr><td align="left" valign="bottom">JEV</td><td align="left" valign="bottom">146920.8</td><td align="left" valign="bottom">148101.5</td><td align="left" valign="bottom"><bold>146885.5</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.091</td></tr><tr><td align="left" valign="bottom" rowspan="2">Hepeviridae</td><td align="left" valign="bottom">HPVE1</td><td align="left" valign="bottom">200439.5</td><td align="left" valign="bottom">200863.8</td><td align="left" valign="bottom"><bold>200179.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.073</td></tr><tr><td align="left" valign="bottom">HPVE2</td><td align="left" valign="bottom">155709.1</td><td align="left" valign="bottom">155983.8</td><td align="left" valign="bottom"><bold>155518.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.088</td></tr><tr><td align="left" valign="bottom" rowspan="8">Picornaviridae</td><td align="left" valign="bottom">ENV_A</td><td align="left" valign="bottom">552287.9</td><td align="left" valign="bottom">553535.5</td><td align="left" valign="bottom"><bold>551794.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.061</td></tr><tr><td align="left" valign="bottom">HRV_A</td><td align="left" valign="bottom">102218.7</td><td align="left" valign="bottom">102267.0</td><td align="left" valign="bottom"><bold>101550.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.285</td></tr><tr><td align="left" valign="bottom">AIV</td><td align="left" valign="bottom">101073.1</td><td align="left" valign="bottom">101136.7</td><td align="left" valign="bottom"><bold>101052.2</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.093</td></tr><tr><td align="left" valign="bottom">AHP</td><td align="left" valign="bottom">139635.7</td><td align="left" valign="bottom">140119.6</td><td align="left" valign="bottom"><bold>139506.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.170</td></tr><tr><td align="left" valign="bottom">ECV</td><td align="left" valign="bottom">82078.9</td><td align="left" valign="bottom">82181.0</td><td align="left" valign="bottom"><bold>82065.8</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.066</td></tr><tr><td align="left" valign="bottom">CDV</td><td align="left" valign="bottom">130551.3</td><td align="left" valign="bottom">130896.7</td><td align="left" valign="bottom"><bold>130478.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.086</td></tr><tr><td align="left" valign="bottom">TCV</td><td align="left" valign="bottom">53027.3</td><td align="left" valign="bottom">53029</td><td align="left" valign="bottom"><bold>53023</bold></td><td align="left" valign="bottom"><bold>0.0151</bold></td><td align="left" valign="bottom"><bold>0.0422</bold></td><td align="left" valign="bottom">0.033</td></tr><tr><td align="left" valign="bottom">FMDV</td><td align="left" valign="bottom">455180.6</td><td align="left" valign="bottom">455582.6</td><td align="left" valign="bottom"><bold>454806.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.117</td></tr><tr><td align="left" valign="bottom">Fusariviridae</td><td align="left" valign="bottom">FRV</td><td align="left" valign="bottom"><bold>52413.1</bold></td><td align="left" valign="bottom">52470.6</td><td align="left" valign="bottom">52418.4</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.076</td></tr><tr><td align="left" valign="bottom" rowspan="11">Retroviridae</td><td align="left" valign="bottom">HIV1_setA</td><td align="left" valign="bottom">344014.4</td><td align="left" valign="bottom">344295.1</td><td align="left" valign="bottom"><bold>343669.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.237</td></tr><tr><td align="left" valign="bottom">HIV1_M</td><td align="left" valign="bottom">80764.1</td><td align="left" valign="bottom">80829.5</td><td align="left" valign="bottom"><bold>80668.1</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.442</td></tr><tr><td align="left" valign="bottom">HIV1_setC</td><td align="left" valign="bottom">180575.0</td><td align="left" valign="bottom">180702.3</td><td align="left" valign="bottom"><bold>180494.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.107</td></tr><tr><td align="left" valign="bottom">HIV1_setD</td><td align="left" valign="bottom">298489.9</td><td align="left" valign="bottom">298695.3</td><td align="left" valign="bottom"><bold>298260.6</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.133</td></tr><tr><td align="left" valign="bottom">HIV1_setE</td><td align="left" valign="bottom">289111.3</td><td align="left" valign="bottom">289292.1</td><td align="left" valign="bottom"><bold>288941.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.112</td></tr><tr><td align="left" valign="bottom">HIV1_setF</td><td align="left" valign="bottom">214375.9</td><td align="left" valign="bottom">214692.2</td><td align="left" valign="bottom"><bold>214289.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.148</td></tr><tr><td align="left" valign="bottom">EIV</td><td align="left" valign="bottom">126149</td><td align="left" valign="bottom">126365.4</td><td align="left" valign="bottom"><bold>125300</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.192</td></tr><tr><td align="left" valign="bottom">BIV</td><td align="left" valign="bottom"><bold>24505.2</bold></td><td align="left" valign="bottom">24506.9</td><td align="left" valign="bottom">24513.2</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">0.15</td></tr><tr><td align="left" valign="bottom">FIV</td><td align="left" valign="bottom">164542.1</td><td align="left" valign="bottom">164487.9</td><td align="left" valign="bottom"><bold>164260.4</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.114</td></tr><tr><td align="left" valign="bottom">CAV</td><td align="left" valign="bottom">351329.9</td><td align="left" valign="bottom">351871.5</td><td align="left" valign="bottom"><bold>350721.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.174</td></tr><tr><td align="left" valign="bottom">SIV</td><td align="left" valign="bottom">110731.2</td><td align="left" valign="bottom">110816</td><td align="left" valign="bottom"><bold>110663.3</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.144</td></tr><tr><td align="left" valign="bottom">Filoviridae</td><td align="left" valign="bottom">Ebola_2</td><td align="left" valign="bottom">53147.3</td><td align="left" valign="bottom"><bold>53143.0</bold></td><td align="left" valign="bottom">53149.9</td><td align="left" valign="bottom">&gt;0.05</td><td align="left" valign="bottom">&gt;0.50</td><td align="left" valign="bottom">0.264</td></tr><tr><td align="left" valign="bottom" rowspan="2">Orthomyxo-viridae</td><td align="left" valign="bottom">Flu A 2</td><td align="left" valign="bottom">82872.8</td><td align="left" valign="bottom">83010.2</td><td align="left" valign="bottom"><bold>82849.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.27</td></tr><tr><td align="left" valign="bottom">Flu B 1</td><td align="left" valign="bottom">50090.4</td><td align="left" valign="bottom">50144.1</td><td align="left" valign="bottom"><bold>50060.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.311</td></tr><tr><td align="left" valign="bottom" rowspan="4">Coronaviridae</td><td align="left" valign="bottom">SARS-COV1</td><td align="left" valign="bottom">214715.3</td><td align="left" valign="bottom">214968.5</td><td align="left" valign="bottom"><bold>214644.39</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.198</td></tr><tr><td align="left" valign="bottom">SARS-COV2</td><td align="left" valign="bottom">15715.4.2</td><td align="left" valign="bottom">15715.6</td><td align="left" valign="bottom"><bold>15696.7</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">1.536</td></tr><tr><td align="left" valign="bottom">SARB</td><td align="left" valign="bottom">573966.3</td><td align="left" valign="bottom">573815.1</td><td align="left" valign="bottom"><bold>572517.0</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.301</td></tr><tr><td align="left" valign="bottom">MERS-COV</td><td align="left" valign="bottom">516683.2</td><td align="left" valign="bottom">516983.4</td><td align="left" valign="bottom"><bold>516608.9</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom"><bold>&lt;0.001</bold></td><td align="left" valign="bottom">0.169</td></tr></tbody></table></table-wrap><p>We found that the degree of non-reversibility (DNR) estimates alone did not cleanly differentiate between datasets for which NREV12 was or was not best supported (<xref ref-type="table" rid="table1 table2 table3 table4">Tables 1–4</xref>). For the 107 nucleotide sequence datasets with a model preference of NREV12, 10 had estimated DNRs that were greater than 0.5, 13 had DNRs between 0.25 and 0.5, and 84 had DNRs between 0.0225 and 0.25. For the 10 nucleotide sequence datasets with a model preference of NREV6, one had an estimated DNR greater than 0.5, four had estimated DNRs between 0.25 and 0.5, and five had estimated DNRs between 0.064 and 0.25 (<xref ref-type="fig" rid="fig1">Figure 1</xref>). For the 24 nucleotide sequence datasets with a model preference of GTR, none had estimated DNRs greater than 0.5, one had an estimated DNR between 0.25 and 0.5, and the remainder had estimated DNRs between 0.0335 and 0.25.</p><p>The dsDNA virus dataset with the highest DNR was Human mastadenovirus D (DNR = 0.646), the dsRNA virus dataset with the highest estimated DNR was Porcine_rotavirus_B (0.351), the ssRNA virus dataset with the highest DNR was SARS-CoV-2 (DNR = 1.536), and the ssDNA virus dataset with the highest DNR was Torque teno sus virus (DNR = 1.56).</p><p>Therefore, while NREV12 appears to be generally more appropriate than either NREV6 or GTR for describing mutational processes in ssRNA, ssDNA, dsDNA, and dsRNA viruses, this might only be particularly relevant from a practical perspective when datasets of these viruses yield DNR estimates that are greater than 0.25. For such datasets, NREV12 (and possibly NREV 6 in some instances) might be especially useful for both determining the direction of evolution across phylogenetic trees (i.e. it could potentially be used to root these trees) and for quantifying genomic strand-specific nucleotide substitution biases (<xref ref-type="bibr" rid="bib14">Harkins et al., 2009</xref>).</p></sec><sec id="s2-2"><title>Assessing the impacts of model misspecification on phylogenetic tree inference</title><p>To determine whether it is worthwhile to use NREV12 rather than GTR for phylogenetic inference when NREV12 is the best fitting nucleotide substitution model, we used simulated datasets to compare the accuracy of phylogenetic trees inferred using these models.</p><p>We found that, regardless of dataset diversity and the nucleotide substitution model used, phylogenetic inference tended to become less accurate (i.e. weighted Robinson-Foulds [wRF] scores increased) as DNR increased (<xref ref-type="fig" rid="fig2">Figure 2</xref>). This tendency was, however, more pronounced when using a (mis-specified) GTR model than when using a (correctly specified) NREV12 model with, for any given dataset having DNR&gt;0, the use of NREV12 tending to yield more accurate phylogenetic trees than when GTR was used. There were, however, only statistically significant improvements (p&lt;0.05, paired t-test) in the accuracy of phylogenetic trees inferred using NREV12 relative to those inferred using GTR in lower diversity datasets (i.e. those with average pairwise nucleotide sequence identities (APIs) of 85%, 90%, and 95%) and then only for DNR&gt;8. It is noteworthy that the highest estimated DNR in any of the empirical datasets that we analysed was 1.56 more than fourfold lower than the point where statistically significant differences in phylogenetic inference accuracies became apparent in the simulated datasets.</p></sec></sec><sec id="s3" sec-type="materials|methods"><title>Materials and methods</title><sec id="s3-1"><title>Virus sequence datasets and phylogenetic trees</title><p>We obtained viral nucleotide sequences from the National Centre for Biotechnology Information Taxonomy database (<ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov/taxonomy">http://www.ncbi.nlm.nih.gov/taxonomy</ext-link>) and the Los Alamos National Laboratory HIV sequence database (<ext-link ext-link-type="uri" xlink:href="https://www.hiv.lanl.gov/content/index">https://www.hiv.lanl.gov/content/index</ext-link>). These included gene and whole-genome sequences for viruses with ssRNA, ssDNA, dsRNA, and dsDNA genomes (datasets are summarised in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). An outgroup sequence from a closely related virus species was added to each dataset to help root phylogenetic trees. The sequences in each of the datasets were aligned using MUSCLE (<xref ref-type="bibr" rid="bib8">Edgar, 2004</xref>) implemented in Aliview (<xref ref-type="bibr" rid="bib18">Larsson, 2014</xref>)⁠, and maximum likelihood phylogenetic trees were constructed from each alignment using RAxML v8.2 (<xref ref-type="bibr" rid="bib38">Stamatakis, 2016</xref>)⁠.</p></sec><sec id="s3-2"><title>Model testing</title><p>We evaluated the fit of NREV12, NREV6, and GTR to the 141 individual sequence datasets using a previously published model test (<xref ref-type="bibr" rid="bib14">Harkins et al., 2009</xref>) implemented as a HyPhy package module (<xref ref-type="bibr" rid="bib27">Pond and Muse, 2005</xref>). This script (<ext-link ext-link-type="uri" xlink:href="https://github.com/veg/hyphy-analyses/tree/master/NucleotideNonREV">https://github.com/veg/hyphy-analyses/tree/master/NucleotideNonREV</ext-link>; <xref ref-type="bibr" rid="bib7">Delport and Kosakovsky Pond, 2023</xref>) is also available as a module in the Datamonkey web server (<xref ref-type="bibr" rid="bib41">Weaver et al., 2018</xref>), which takes as input a rooted maximum likelihood phylogenetic tree (minus the rooting sequence) and its corresponding nucleotide sequence alignment. The three models described above: GTR, NREV6, and NREV12 are then fitted to the data using maximum likelihood (ML). The equilibrium frequencies (EF) of the GTR model match those empirically observed in the alignment, while EFs for NREV6 and NREV12 are inferred by the model, satisfying the condition <inline-formula><alternatives><mml:math id="inf7"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi mathvariant="bold-italic">π</mml:mi><mml:mi mathvariant="bold-italic">Q</mml:mi><mml:mo mathvariant="bold">=</mml:mo><mml:mn mathvariant="bold">0</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft7">\begin{document}$\boldsymbol {\pi Q=0}$\end{document}</tex-math></alternatives></inline-formula>. An additional model NREV12 + F is estimated, where the distribution of nucleotides at the root of the tree is estimated by maximum likelihood, instead of being set to the empirical frequencies. Nested models were compared by the LRT with significance evaluated using the <inline-formula><alternatives><mml:math id="inf8"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msubsup><mml:mi>χ</mml:mi><mml:mi>d</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft8">\begin{document}$\chi _d^2$\end{document}</tex-math></alternatives></inline-formula>  distribution with <italic>d</italic>=difference in degrees of freedom. For all models, we also computed the small AIC-c score.</p></sec><sec id="s3-3"><title>Quantification of non-reversibility</title><p>We further defined the DNR as the absolute difference between the relative rate differences of two nucleotide pairs; i.e., for two nucleotides, <italic>x</italic> and <italic>y</italic>, there exists a relative rate of <italic>x</italic> to <italic>y</italic> substitutions that we will refer to as <italic>m</italic>, and a relative rate of <italic>y</italic> to <italic>x</italic> substitutions that we will refer to as <italic>n</italic>. Under the NREV12 model, the DNR between <italic>x</italic> and <italic>y</italic> is defined simply as the absolute difference between <italic>m</italic> and <italic>n</italic>: (|<italic>m-n</italic>|) and will hereby be referred to as <inline-formula><alternatives><mml:math id="inf9"><mml:mstyle><mml:mrow><mml:mi>i</mml:mi><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>N</mml:mi><mml:mi>R</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math><tex-math id="inft9">\begin{document}$ij_{DNR}$\end{document}</tex-math></alternatives></inline-formula>, where <inline-formula><alternatives><mml:math id="inf10"><mml:mstyle><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:mstyle></mml:math><tex-math id="inft10">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf11"><mml:mstyle><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:mstyle></mml:math><tex-math id="inft11">\begin{document}$j$\end{document}</tex-math></alternatives></inline-formula> are two nucleotides. We use DNR as a mathematical representation of the degree to which the rates of all pairs of reverse substitutions differ from one another. For each of the 140 individual viral alignments, we calculated the average DNR of the six <inline-formula><alternatives><mml:math id="inf12"><mml:mstyle><mml:mrow><mml:mi>i</mml:mi><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>N</mml:mi><mml:mi>R</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math><tex-math id="inft12">\begin{document}$ij_{DNR}$\end{document}</tex-math></alternatives></inline-formula> estimates inferred using the NREV12 model.</p></sec><sec id="s3-4"><title>Simulations for testing the impact of non-reversible evolution on phylogenetic inference</title><p>We tested the accuracy of phylogenetic tree inference under reversible and non-reversible models using simulated datasets with varying APIs evolved under the NREV12 model with different DNRs. The goal of these tests was not to exhaustively evaluate model misspecification issues during phylogenetic tree inference, but rather to check, in instances where viral taxa are known to be evolving in a detectably non-reversible manner (i.e. where NREV12 or NREV6 fits the data better than GTR), whether not accounting for this might decrease the accuracy of phylogenetic inference. Using IQ-TREE, a phylogenetic inference program that has the option to apply an NREV12-like model (referred to in IQ-TREE as the UNREST model), a phylogenetic tree was inferred from an alignment of real sequences (Avian Leukosis virus) (<xref ref-type="fig" rid="fig3">Figure 3</xref>) with an API of ~90%. The branch lengths on this tree were then scaled to create four other phylogenetic trees representing sequences with approximately 95%, 85%, 80%, and 75% API. These five trees are hereafter referred to as ‘true’ trees, and each individual tree was used as the starting point of a different set of simulations.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Phylogenetic tree inferred from an alignment of real sequences (Avian Leukosis virus) that was used to simulate datasets with degrees of non-reversibility (DNRs) varying from 0 to 20.</title><p>The alignment of Avian Leukosis virus had an average sequence identity (API) of ~90%, and the branches of this tree were scaled to produce four other trees reflecting branch tip sequences with approximate pairwise identities of ~75%, ~80%, ~85%, and~95%.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87361-fig3-v1.tif"/></fig><p>Phylogenetic trees were inferred from these 5500 simulated datasets and compared to the phylogenetic trees used to simulate the datasets (i.e. the true trees) using wRF distances to assess the impact of varying DNR on the accuracy of phylogenetic inference. We further tested whether the accuracy of phylogenetic inference could be improved for sequences that had evolved under DNR &gt;0 by using NREV12 instead of GTR. Specifically, for every simulated dataset, a phylogenetic tree was inferred using GTR, and another using NREV12 and the wRF distances of each of these trees to the true tree was determined. For each of the analysed DNRs, a paired t-test was then used to compare the wRF scores of trees inferred using GTR and NREV12. We were particularly interested in determining whether trees inferred using a mis-specified model (i.e. GTR in this case) would be less accurate than trees inferred with a correctly specified model (i.e. NREV12).</p><p>To test whether failure to account for non-reversibility might decrease the accuracy of phylogenetic inference, we simulated the evolution of 5500 nucleotide sequence alignments evolved non-reversibly under varying DNR along the five true phylogenetic trees: 100 datasets per true tree per simulated DNR. Specifically, simulations were done using HyPhy (<xref ref-type="bibr" rid="bib27">Pond and Muse, 2005</xref>), with relative rates ranging from a completely reversible matrix (<xref ref-type="disp-formula" rid="equ4">Equation 4</xref>)<disp-formula id="equ4"><label>(4)</label><alternatives><mml:math id="m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi mathvariant="italic">Q</mml:mi><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi mathvariant="italic">q</mml:mi><mml:mrow><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mo mathvariant="italic">=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="center center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mn>0.166</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0.14</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.166</mml:mn></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mn>0.131</mml:mn></mml:mtd><mml:mtd><mml:mn>1.101</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0.131</mml:mn></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd><mml:mtd><mml:mn>0.188</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0.14</mml:mn></mml:mtd><mml:mtd><mml:mn>1.101</mml:mn></mml:mtd><mml:mtd><mml:mn>0.188</mml:mn></mml:mtd><mml:mtd><mml:mo>−</mml:mo></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="t4">\begin{document}$$\displaystyle \boldsymbol{\it Q=\left\{q_{i j}\right\}=\left(\begin{array}{cccc} - &amp; 0.166 &amp; 1 &amp; 0.14 \\ 0.166 &amp; - &amp; 0.131 &amp; 1.101 \\ 1 &amp; 0.131 &amp; - &amp; 0.188 \\ 0.14 &amp; 1.101 &amp; 0.188 &amp; - \end{array}\right)}$$\end{document}</tex-math></alternatives></disp-formula></p><p>representing DNR = 0 – through matrices with DNR = 2, 4, 6, 8, 10, 12, 14, 16, 18, and 20 (<xref ref-type="table" rid="table5">Table 5</xref>). These baselines-simulated substitution rates are reflective of those seen in empirical viral nucleotide sequence datasets.</p><table-wrap id="table5" position="float"><label>Table 5.</label><caption><title>Relative rate change for C to A, G to A, A to T, G to C, T to G, and C to T mutations under the 11 degrees of non-reversibility alongside the maintained rates for A to C, A to G, T to A, C to G, G to T, and T to C.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom" rowspan="2">Degree of non-reversibility (DNR)</th><th align="left" valign="bottom" colspan="12">Relative rates of different nucleotide substitution types (from-to)</th></tr><tr><th align="left" valign="bottom">C-A</th><th align="left" valign="bottom">A-C</th><th align="left" valign="bottom">G-A</th><th align="left" valign="bottom">A-G</th><th align="left" valign="bottom">A-T</th><th align="left" valign="bottom">T-A</th><th align="left" valign="bottom">G-C</th><th align="left" valign="bottom">C-G</th><th align="left" valign="bottom">T-G</th><th align="left" valign="bottom">G-T</th><th align="left" valign="bottom">C-T</th><th align="left" valign="bottom">T-C</th></tr></thead><tbody><tr><td align="left" valign="bottom">0</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">1.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">2</td><td align="left" valign="bottom">2.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">3</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">2.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">2.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">2.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">3.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">4</td><td align="left" valign="bottom">4.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">5</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">4.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">4.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">4.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">5.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">6</td><td align="left" valign="bottom">6.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">7</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">6.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">6.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">6.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">7.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">8</td><td align="left" valign="bottom">8.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">9</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">8.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">8.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">8.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">9.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">10</td><td align="left" valign="bottom">10.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">11</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">10.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">10.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">10.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">11.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">12</td><td align="left" valign="bottom">12.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">13</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">12.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">12.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">12.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">13.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">14</td><td align="left" valign="bottom">14.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">15</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">14.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">14.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">14.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">15.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">16</td><td align="left" valign="bottom">16.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">17</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">16.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">16.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">16.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">17.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">18</td><td align="left" valign="bottom">18.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">19</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">18.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">18.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">18.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">19.101</td><td align="left" valign="bottom">1.101</td></tr><tr><td align="left" valign="bottom">20</td><td align="left" valign="bottom">20.166</td><td align="left" valign="bottom">0.166</td><td align="left" valign="bottom">21</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">20.14</td><td align="left" valign="bottom">0.14</td><td align="left" valign="bottom">20.131</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">20.118</td><td align="left" valign="bottom">0.118</td><td align="left" valign="bottom">21.101</td><td align="left" valign="bottom">1.101</td></tr></tbody></table></table-wrap><p>At each DNR, the relative rates used conformed to standard measures of non-reversibility under the Kolmogorov conditions according to which non-reversibly evolving sequence datasets should yield three irreversibility indices (IRI1, IRI2, and IRI3) that are all non-zero (<xref ref-type="bibr" rid="bib37">Squartini and Arndt, 2008</xref>). It should be noted that all simulations under NREV12 were performed under the stationarity criterion: <inline-formula><alternatives><mml:math id="inf13"><mml:mstyle><mml:mrow><mml:mrow><mml:mi>π</mml:mi><mml:msup><mml:mi mathvariant="normal">e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Q</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>π</mml:mi></mml:mrow></mml:mrow></mml:mstyle></mml:math><tex-math id="inft13">\begin{document}$\rm {\pi e^{Qt}=\pi}$\end{document}</tex-math></alternatives></inline-formula> (where <italic>Q</italic> is the rate matrix, <italic>π</italic> is the nucleotide frequency distribution, and <italic>t</italic>≥0).</p></sec><sec id="s3-5"><title>Quantifying the accuracy of phylogenetic inferences</title><p>We used the wRF (implemented in the R phangorn package; <xref ref-type="bibr" rid="bib35">Schliep, 2011</xref>⁠) to quantify differences between the true trees used to simulate datasets and the trees inferred from these datasets using the GTR or NREV12 models. wRF considers differences between both the topology and branch lengths of actual and inferred trees (<xref ref-type="bibr" rid="bib16">Kuhner and Yamato, 2015</xref>; <xref ref-type="bibr" rid="bib33">Robinson and Foulds, 1981</xref>).</p></sec><sec id="s3-6"><title>Conclusion</title><p>The non-reversible nucleotide substitution model, NREV12, provides a substantially better fit to most virus nucleotide sequence datasets than does the widely used reversible substitution model, GTR. NREV12 also provides a better fit to most virus nucleotide sequence datasets than does NREV6; a non-reversible model that would be expected to best describe the evolution of double-stranded genome sequences that display no strand-specific nucleotide substitution biases. This suggests that, contrary to our expectations, substantial strand-specific nucleotide substitution biases (i.e. estimated DNRs&gt;0.25) are common during viral evolution irrespective of genome type. Such biases should be expected for any viruses where one genome strand either is in existence for substantially longer periods of time than the other, or is more exposed to mutagenic processes than the other during transmission, replication, or gene expression.</p><p>We had anticipated that, given evidence of sequences evolving both non-reversibly and with strand-specific substitution biases, inferring trees using a model such as NREV12 that appropriately accounts for this might: (1) minimise the impact of increasing DNR on the accuracy of phylogenetic inference (i.e. wRF scores presented in blue in <xref ref-type="fig" rid="fig2">Figure 2</xref> might have been expected to not increase with increasing DNR) and (2) yield significantly more accurate phylogenetic inferences than when using GTR for all datasets where NREV12 was the most appropriate model and DNRs were greater than zero. However, increasing DNR clearly decreased the accuracy of phylogenetic inference even when using NREV12, and, for datasets where DNRs were greater than zero, using GTR did not consistently yield significantly less accurate phylogenetic inferences than those attained using NREV12. From a practical perspective, choosing a non-reversible nucleotide substitution model to construct phylogenetic trees from virus genome sequences that display strand-specific nucleotide substitution biases is not guaranteed to yield more accurate phylogenetic trees. Nevertheless, in instances where strand-specific substitution biases are higher than ~0.5 (such as are found in our SARS-CoV-2, Torque teno sus virus, and Banana bunchy top virus datasets), it may be prudent to select a model such as NREV12 (such as is implemented in programs like IQ-TREE) over GTR as the better of two suboptimal choices.</p><p>The lack of available data regarding the proportions of viral life cycles during which genomes exist in single- and double-stranded states makes it difficult to rationally predict the situations where the use of models such as GTR, NREV6, and NREV12 might be most justified: particularly in light of the poor overall performance of NREV6 and GTR relative to NREV12 with respect to describing mutational processes in viral genome sequence datasets. We therefore recommend case-by-case assessments of NREV12 vs NREV6 vs GTR model fit when deciding whether it is appropriate to consider the application of non-reversible models for phylogenetic inference and/or phylogenetic model-based analyses such as those intended to test for evidence of natural selection or the existence of molecular clocks.</p></sec><sec id="s3-7"><title>Declarations</title><sec id="s3-7-1"><title>Ethics</title><p>The University of Cape Town ethics committee declared that this research did not need ethics approval due to the use of freely accessible nucleotide sequences obtained from the National Centre for Biotechnology Information Taxonomy database (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/taxonomy">https://www.ncbi.nlm.nih.gov/taxonomy</ext-link>) and the Los Alamos National Laboratory HIV sequence database (<ext-link ext-link-type="uri" xlink:href="https://www.hiv.lanl.gov/content/index">https://www.hiv.lanl.gov/content/index</ext-link>).</p></sec></sec></sec></body><back><sec sec-type="additional-information" id="s4"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Formal analysis, Visualization, Methodology, Writing – original draft, Project administration</p></fn><fn fn-type="con" id="con2"><p>Investigation</p></fn><fn fn-type="con" id="con3"><p>Investigation</p></fn><fn fn-type="con" id="con4"><p>Investigation</p></fn><fn fn-type="con" id="con5"><p>Methodology</p></fn><fn fn-type="con" id="con6"><p>Investigation</p></fn><fn fn-type="con" id="con7"><p>Investigation</p></fn><fn fn-type="con" id="con8"><p>Methodology</p></fn><fn fn-type="con" id="con9"><p>Supervision, Methodology</p></fn><fn fn-type="con" id="con10"><p>Supervision, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s5"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-87361-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Summary of the viral genome component datasets used in the study.</title></caption><media xlink:href="elife-87361-supp1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s6"><title>Data availability</title><p>We obtained viral nucleotide sequences from the National Centre for Biotechnology Information Taxonomy database (<ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov/taxonomy">http://www.ncbi.nlm.nih.gov/taxonomy</ext-link>) and the Los Alamos National Laboratory HIV sequence database (<ext-link ext-link-type="uri" xlink:href="https://www.hiv.lanl.gov/content/index">https://www.hiv.lanl.gov/content/index</ext-link>). These included gene and whole-genome sequences for viruses with ssRNA, ssDNA, dsRNA, and dsDNA genomes (datasets are summarized in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>).</p></sec><ack id="ack"><title>Acknowledgements</title><p>The authors wish to thank the South African National Research Foundation (NRF) for funding this research under the South African Centre for Epidemiological Modelling and Analysis (SACEMA) bursary. SLKP and SW were supported in part by the U.S. National Institutes of Health (R01 AI134384 and AI140970, GM144468). DPM is supported by the Wellcome Trust (222574/Z/21/Z).</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baele</surname><given-names>G</given-names></name><name><surname>Van de Peer</surname><given-names>Y</given-names></name><name><surname>Vansteelandt</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Using non-reversible context-dependent evolutionary models to study substitution patterns in primate non-coding sequences</article-title><source>Journal of Molecular Evolution</source><volume>71</volume><fpage>34</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1007/s00239-010-9362-y</pub-id><pub-id pub-id-type="pmid">20623275</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boussau</surname><given-names>B</given-names></name><name><surname>Gouy</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Efficient likelihood computations with nonreversible models of evolution</article-title><source>Systematic Biology</source><volume>55</volume><fpage>756</fpage><lpage>768</lpage><pub-id pub-id-type="doi">10.1080/10635150600975218</pub-id><pub-id pub-id-type="pmid">17060197</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Bruslind</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Retrieved from General Microbiology</source><publisher-name>Oregon State University</publisher-name></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buckley</surname><given-names>TR</given-names></name><name><surname>Cunningham</surname><given-names>CW</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>The effects of nucleotide substitution model assumptions on estimates of nonparametric bootstrap support</article-title><source>Molecular Biology and Evolution</source><volume>19</volume><fpage>394</fpage><lpage>405</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a004094</pub-id><pub-id pub-id-type="pmid">11919280</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chelico</surname><given-names>L</given-names></name><name><surname>Pham</surname><given-names>P</given-names></name><name><surname>Calabrese</surname><given-names>P</given-names></name><name><surname>Goodman</surname><given-names>MF</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>APOBEC3G DNA deaminase acts processively 3′ → 5′ on single-stranded DNA</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>13</volume><fpage>392</fpage><lpage>399</lpage><pub-id pub-id-type="doi">10.1038/nsmb1086</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname><given-names>KC</given-names></name><name><surname>Cahill</surname><given-names>DS</given-names></name><name><surname>Kasai</surname><given-names>H</given-names></name><name><surname>Nishimura</surname><given-names>S</given-names></name><name><surname>Loeb</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>8-Hydroxyguanine, an abundant form of oxidative DNA damage, causes G----T and A----C substitutions</article-title><source>The Journal of Biological Chemistry</source><volume>267</volume><fpage>166</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1016/S0021-9258(18)48474-8</pub-id><pub-id pub-id-type="pmid">1730583</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Delport</surname><given-names>W</given-names></name><name><surname>Kosakovsky Pond</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Hyphy-analysis</data-title><version designator="d143ea6">d143ea6</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/veg/hyphy-analyses/tree/master/NucleotideNonREV">https://github.com/veg/hyphy-analyses/tree/master/NucleotideNonREV</ext-link></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>1792</fpage><lpage>1797</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id><pub-id pub-id-type="pmid">15034147</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Fernandes</surname><given-names>JV</given-names></name><name><surname>Medeiros Fernandes</surname><given-names>TA</given-names></name></person-group><year iso-8601-date="2012">2012</year><chapter-title>Human papillomavirus: biology and pathogenesis</chapter-title><person-group person-group-type="editor"><name><surname>Vanden Broeck</surname><given-names>D</given-names></name></person-group><source>Human Papillomavirus and Related Diseases-From Bench to Bedside-A Clinical Perspective</source><publisher-name>IntechOpen</publisher-name><pub-id pub-id-type="doi">10.5772/27154</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fijalkowska</surname><given-names>IJ</given-names></name><name><surname>Jonczyk</surname><given-names>P</given-names></name><name><surname>Tkaczyk</surname><given-names>MM</given-names></name><name><surname>Bialoskorska</surname><given-names>M</given-names></name><name><surname>Schaaper</surname><given-names>RM</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Unequal fidelity of leading strand and lagging strand DNA replication on the <italic>Escherichia coli</italic> chromosome</article-title><source>PNAS</source><volume>95</volume><fpage>10020</fpage><lpage>10025</lpage><pub-id pub-id-type="doi">10.1073/pnas.95.17.10020</pub-id><pub-id pub-id-type="pmid">9707593</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Furusawa</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Implications of fidelity difference between the leading and the lagging strand of DNA for the acceleration of evolution</article-title><source>Frontiers in Oncology</source><volume>2</volume><elocation-id>144</elocation-id><pub-id pub-id-type="doi">10.3389/fonc.2012.00144</pub-id><pub-id pub-id-type="pmid">23087905</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grigoriev</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Strand-specific compositional asymmetries in double-stranded DNA viruses</article-title><source>Virus Research</source><volume>60</volume><fpage>1</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1016/s0168-1702(98)00139-7</pub-id><pub-id pub-id-type="pmid">10225270</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hanson</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2009">2009</year><chapter-title>Isolation of viral DNA from cultures</chapter-title><source>Handbook of Nucleic Acid Purification</source><publisher-name>CRC press</publisher-name><fpage>1</fpage><lpage>18</lpage><pub-id pub-id-type="doi">10.1201/9781420070972.pt1</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harkins</surname><given-names>GW</given-names></name><name><surname>Delport</surname><given-names>W</given-names></name><name><surname>Duffy</surname><given-names>S</given-names></name><name><surname>Wood</surname><given-names>N</given-names></name><name><surname>Monjane</surname><given-names>AL</given-names></name><name><surname>Owor</surname><given-names>BE</given-names></name><name><surname>Donaldson</surname><given-names>L</given-names></name><name><surname>Saumtally</surname><given-names>S</given-names></name><name><surname>Triton</surname><given-names>G</given-names></name><name><surname>Briddon</surname><given-names>RW</given-names></name><name><surname>Shepherd</surname><given-names>DN</given-names></name><name><surname>Rybicki</surname><given-names>EP</given-names></name><name><surname>Martin</surname><given-names>DP</given-names></name><name><surname>Varsani</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Experimental evidence indicating that mastreviruses probably did not co-diverge with their hosts</article-title><source>Virology Journal</source><volume>6</volume><fpage>1</fpage><lpage>14</lpage><pub-id pub-id-type="doi">10.1186/1743-422X-6-104</pub-id><pub-id pub-id-type="pmid">19607673</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hoff</surname><given-names>M</given-names></name><name><surname>Orf</surname><given-names>S</given-names></name><name><surname>Riehm</surname><given-names>B</given-names></name><name><surname>Darriba</surname><given-names>D</given-names></name><name><surname>Stamatakis</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Does the choice of nucleotide substitution models matter topologically?</article-title><source>BMC Bioinformatics</source><volume>17</volume><elocation-id>143</elocation-id><pub-id pub-id-type="doi">10.1186/s12859-016-0985-x</pub-id><pub-id pub-id-type="pmid">27009141</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kuhner</surname><given-names>MK</given-names></name><name><surname>Yamato</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Practical performance of tree comparison metrics</article-title><source>Systematic Biology</source><volume>64</volume><fpage>205</fpage><lpage>214</lpage><pub-id pub-id-type="doi">10.1093/sysbio/syu085</pub-id><pub-id pub-id-type="pmid">25378436</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kusumoto-Matsuo</surname><given-names>R</given-names></name><name><surname>Kanda</surname><given-names>T</given-names></name><name><surname>Kukimoto</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Rolling circle replication of human papillomavirus type 16 DNA in epithelial cell extracts</article-title><source>Genes to Cells</source><volume>16</volume><fpage>23</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2443.2010.01458.x</pub-id><pub-id pub-id-type="pmid">21059156</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Larsson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>AliView: a fast and lightweight alignment viewer and editor for large datasets</article-title><source>Bioinformatics</source><volume>30</volume><fpage>3276</fpage><lpage>3278</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btu531</pub-id><pub-id pub-id-type="pmid">25095880</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lefort</surname><given-names>V</given-names></name><name><surname>Longueville</surname><given-names>JE</given-names></name><name><surname>Gascuel</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>SMS: Smart Model Selection in PhyML</article-title><source>Molecular Biology and Evolution</source><volume>34</volume><fpage>2422</fpage><lpage>2424</lpage><pub-id pub-id-type="doi">10.1093/molbev/msx149</pub-id><pub-id pub-id-type="pmid">28472384</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liò</surname><given-names>P</given-names></name><name><surname>Goldman</surname><given-names>N</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Models of molecular evolution and phylogeny</article-title><source>Genome Research</source><volume>8</volume><fpage>1233</fpage><lpage>1244</lpage><pub-id pub-id-type="doi">10.1101/gr.8.12.1233</pub-id><pub-id pub-id-type="pmid">9872979</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minin</surname><given-names>V</given-names></name><name><surname>Abdo</surname><given-names>Z</given-names></name><name><surname>Joyce</surname><given-names>P</given-names></name><name><surname>Sullivan</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Performance-based selection of likelihood models for phylogeny estimation</article-title><source>Systematic Biology</source><volume>52</volume><fpage>674</fpage><lpage>683</lpage><pub-id pub-id-type="doi">10.1080/10635150390235494</pub-id><pub-id pub-id-type="pmid">14530134</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname><given-names>T</given-names></name><name><surname>Brunson</surname><given-names>D</given-names></name><name><surname>Crespi</surname><given-names>CL</given-names></name><name><surname>Penman</surname><given-names>BW</given-names></name><name><surname>Wishnok</surname><given-names>JS</given-names></name><name><surname>Tannenbaum</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>DNA damage and mutation in human cells exposed to nitric oxide in vitro</article-title><source>PNAS</source><volume>89</volume><fpage>3030</fpage><lpage>3034</lpage><pub-id pub-id-type="doi">10.1073/pnas.89.7.3030</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname><given-names>L-T</given-names></name><name><surname>Schmidt</surname><given-names>HA</given-names></name><name><surname>von Haeseler</surname><given-names>A</given-names></name><name><surname>Minh</surname><given-names>BQ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>IQ-TREE: a fast and effective stochastic algorithm for estimating maximum-likelihood phylogenies</article-title><source>Molecular Biology and Evolution</source><volume>32</volume><fpage>268</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1093/molbev/msu300</pub-id><pub-id pub-id-type="pmid">25371430</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Onwubiko</surname><given-names>NO</given-names></name><name><surname>Borst</surname><given-names>A</given-names></name><name><surname>Diaz</surname><given-names>SA</given-names></name><name><surname>Passkowski</surname><given-names>K</given-names></name><name><surname>Scheffel</surname><given-names>F</given-names></name><name><surname>Tessmer</surname><given-names>I</given-names></name><name><surname>Nasheuer</surname><given-names>HP</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SV40 T antigen interactions with ssDNA and replication protein A: a regulatory role of T antigen monomers in lagging strand DNA replication</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>3657</fpage><lpage>3677</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa138</pub-id><pub-id pub-id-type="pmid">32128579</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pavlov</surname><given-names>YI</given-names></name><name><surname>Newlon</surname><given-names>CS</given-names></name><name><surname>Kunkel</surname><given-names>TA</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Yeast origins establish a strand bias for replicational mutagenesis</article-title><source>Molecular Cell</source><volume>10</volume><fpage>207</fpage><lpage>213</lpage><pub-id pub-id-type="doi">10.1016/s1097-2765(02)00567-1</pub-id><pub-id pub-id-type="pmid">12150920</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Polak</surname><given-names>P</given-names></name><name><surname>Arndt</surname><given-names>PF</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Transcription induces strand-specific mutations at the 5’ end of human genes</article-title><source>Genome Research</source><volume>18</volume><fpage>1216</fpage><lpage>1223</lpage><pub-id pub-id-type="doi">10.1101/gr.076570.108</pub-id><pub-id pub-id-type="pmid">18463301</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Pond</surname><given-names>SL</given-names></name><name><surname>Muse</surname><given-names>SV</given-names></name></person-group><year iso-8601-date="2005">2005</year><chapter-title>HyPhy: hypothesis testing using phylogenies</chapter-title><person-group person-group-type="editor"><name><surname>Muse</surname><given-names>SV</given-names></name></person-group><source>Statistical Methods in Molecular Evolution</source><publisher-name>Oxford University Press</publisher-name><fpage>125</fpage><lpage>181</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bti079</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Posada</surname><given-names>D</given-names></name><name><surname>Crandall</surname><given-names>KA</given-names></name></person-group><year iso-8601-date="2001">2001a</year><article-title>Selecting models of nucleotide substitution: an application to human immunodeficiency virus 1 (HIV-1)</article-title><source>Molecular Biology and Evolution</source><volume>18</volume><fpage>897</fpage><lpage>906</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a003890</pub-id><pub-id pub-id-type="pmid">11371577</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Posada</surname><given-names>D</given-names></name><name><surname>Crandall</surname><given-names>KA</given-names></name></person-group><year iso-8601-date="2001">2001b</year><article-title>Selecting the best-fit model of nucleotide substitution</article-title><source>Systematic Biology</source><volume>50</volume><fpage>580</fpage><lpage>601</lpage><pub-id pub-id-type="pmid">12116655</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Posada</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Using MODELTEST and PAUP* to select a model of nucleotide substitution</article-title><source>Current Protocols in Bioinformatics</source><volume>Chapter 6</volume><elocation-id>Unit</elocation-id><pub-id pub-id-type="doi">10.1002/0471250953.bi0605s00</pub-id><pub-id pub-id-type="pmid">18428705</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ripplinger</surname><given-names>J</given-names></name><name><surname>Sullivan</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Does choice in model selection affect maximum likelihood analysis?</article-title><source>Systematic Biology</source><volume>57</volume><fpage>76</fpage><lpage>85</lpage><pub-id pub-id-type="doi">10.1080/10635150801898920</pub-id><pub-id pub-id-type="pmid">18275003</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ritz</surname><given-names>C</given-names></name><name><surname>Spiess</surname><given-names>AN</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>qpcR: an R package for sigmoidal model selection in quantitative real-time polymerase chain reaction analysis</article-title><source>Bioinformatics</source><volume>24</volume><fpage>1549</fpage><lpage>1551</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btn227</pub-id><pub-id pub-id-type="pmid">18482995</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname><given-names>DF</given-names></name><name><surname>Foulds</surname><given-names>LR</given-names></name></person-group><year iso-8601-date="1981">1981</year><article-title>Comparison of phylogenetic trees</article-title><source>Mathematical Biosciences</source><volume>53</volume><fpage>131</fpage><lpage>147</lpage><pub-id pub-id-type="doi">10.1016/0025-5564(81)90043-2</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sanjuán</surname><given-names>R</given-names></name><name><surname>Domingo-Calap</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mechanisms of viral mutation</article-title><source>Cellular and Molecular Life Sciences</source><volume>73</volume><fpage>4433</fpage><lpage>4448</lpage><pub-id pub-id-type="doi">10.1007/s00018-016-2299-6</pub-id><pub-id pub-id-type="pmid">27392606</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schliep</surname><given-names>KP</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>phangorn: phylogenetic analysis in R</article-title><source>Bioinformatics</source><volume>27</volume><fpage>592</fpage><lpage>593</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq706</pub-id><pub-id pub-id-type="pmid">21169378</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname><given-names>S</given-names></name><name><surname>Patnaik</surname><given-names>SK</given-names></name><name><surname>Taggart</surname><given-names>RT</given-names></name><name><surname>Baysal</surname><given-names>BE</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The double-domain cytidine deaminase APOBEC3G is a cellular site-specific RNA editing enzyme</article-title><source>Scientific Reports</source><volume>6</volume><elocation-id>39100</elocation-id><pub-id pub-id-type="doi">10.1038/srep39100</pub-id><pub-id pub-id-type="pmid">27974822</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Squartini</surname><given-names>F</given-names></name><name><surname>Arndt</surname><given-names>PF</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Quantifying the stationarity and time reversibility of the nucleotide substitution process</article-title><source>Molecular Biology and Evolution</source><volume>25</volume><fpage>2525</fpage><lpage>2535</lpage><pub-id pub-id-type="doi">10.1093/molbev/msn169</pub-id><pub-id pub-id-type="pmid">18682605</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Stamatakis</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><data-title>The raxml v8. 2. x manual</data-title><version designator="v8. 2">v8. 2</version><source>Exelixis</source><ext-link ext-link-type="uri" xlink:href="https://cme.h-its.org/exelixis/resource/download/NewManual.pdf">https://cme.h-its.org/exelixis/resource/download/NewManual.pdf</ext-link></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tavaré</surname><given-names>S</given-names></name></person-group><year iso-8601-date="1986">1986</year><article-title>Some probabilistic and statistical problems in the analysis of DNA sequences</article-title><source>Lectures on Mathematics in the Life Sciences</source><volume>17</volume><fpage>57</fpage><lpage>86</lpage></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van der Walt</surname><given-names>E</given-names></name><name><surname>Martin</surname><given-names>DP</given-names></name><name><surname>Varsani</surname><given-names>A</given-names></name><name><surname>Polston</surname><given-names>JE</given-names></name><name><surname>Rybicki</surname><given-names>EP</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Experimental observations of rapid Maize streak virus evolution reveal a strand-specific nucleotide substitution bias</article-title><source>Virology Journal</source><volume>5</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1186/1743-422X-5-104</pub-id><pub-id pub-id-type="pmid">18816368</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weaver</surname><given-names>S</given-names></name><name><surname>Shank</surname><given-names>SD</given-names></name><name><surname>Spielman</surname><given-names>SJ</given-names></name><name><surname>Li</surname><given-names>M</given-names></name><name><surname>Muse</surname><given-names>SV</given-names></name><name><surname>Kosakovsky Pond</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Datamonkey 2.0: A Modern Web Application for Characterizing Selective and Other Evolutionary Processes</article-title><source>Molecular Biology and Evolution</source><volume>35</volume><fpage>773</fpage><lpage>777</lpage><pub-id pub-id-type="doi">10.1093/molbev/msx335</pub-id><pub-id pub-id-type="pmid">29301006</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname><given-names>S-J</given-names></name><name><surname>Shi</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>X-X</given-names></name><name><surname>Sharkey</surname><given-names>MJ</given-names></name><name><surname>van Achterberg</surname><given-names>C</given-names></name><name><surname>Ye</surname><given-names>G-Y</given-names></name><name><surname>He</surname><given-names>J-H</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>New views on strand asymmetry in insect mitochondrial genomes</article-title><source>PLOS ONE</source><volume>5</volume><elocation-id>e12708</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0012708</pub-id><pub-id pub-id-type="pmid">20856815</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wickner</surname><given-names>RB</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Double-stranded RNA virus replication and packaging</article-title><source>The Journal of Biological Chemistry</source><volume>268</volume><fpage>3797</fpage><lpage>3800</lpage><pub-id pub-id-type="pmid">8440674</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Maximum likelihood phylogenetic estimation from DNA sequences with variable rates over sites: approximate methods</article-title><source>Journal of Molecular Evolution</source><volume>39</volume><fpage>306</fpage><lpage>314</lpage><pub-id pub-id-type="doi">10.1007/BF00160154</pub-id><pub-id pub-id-type="pmid">7932792</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yap</surname><given-names>VB</given-names></name><name><surname>Speed</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Rooting a phylogenetic tree with nonreversible substitution models</article-title><source>BMC Evolutionary Biology</source><volume>5</volume><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1186/1471-2148-5-2</pub-id><pub-id pub-id-type="pmid">15629063</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>Q</given-names></name><name><surname>König</surname><given-names>R</given-names></name><name><surname>Pillai</surname><given-names>S</given-names></name><name><surname>Chiles</surname><given-names>K</given-names></name><name><surname>Kearney</surname><given-names>M</given-names></name><name><surname>Palmer</surname><given-names>S</given-names></name><name><surname>Richman</surname><given-names>D</given-names></name><name><surname>Coffin</surname><given-names>JM</given-names></name><name><surname>Landau</surname><given-names>NR</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Single-strand specificity of APOBEC3G accounts for minus-strand deamination of the HIV genome</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>11</volume><fpage>435</fpage><lpage>442</lpage><pub-id pub-id-type="doi">10.1038/nsmb758</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87361.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Rokas</surname><given-names>Antonis</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Vanderbilt University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Valuable</kwd></kwd-group></front-stub><body><p>This <bold>valuable</bold> study revisits the effects of substitution model selection on phylogenetics by comparing reversible and non-reversible DNA substitution models. The authors provide <bold>solid</bold> evidence that (1) it can be beneficial to include non-time-reversible models in addition to general time-reversible models when inferring phylogenetic trees out of simulated viral genome sequence data sets, and that (2) non time-reversible models may fit the real data better than the reversible substitution models commonly used in phylogenetics, a finding consistent with previous work.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87361.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>The study by Sianga-Mete et al revisits the effects of substitution model selection on phylogenetics by comparing reversible and non-reversible DNA substitution models. This topic is not new, previous works already showed that non-reversible, and also covarion, substitution models can fit the real data better than the reversible substitution models commonly used in phylogenetics. In this regard, the results of the present study are not surprising.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87361.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>The authors evaluate whether non time reversible models fit better data presenting strand-specific substitution biases than time reversible models. Specifically, the authors consider what they call NREV6 and NREV12 as candidate non time-reversible models. On the one hand, they show that AIC tends to select NREV12 more often than GTR on real virus data sets. On the other hand, they show using simulated data that NREV12 leads to inferred trees that are closer to the true generating tree when the data incorporates a certain degree of non time-reversibility. Based on these two experimental results, the authors conclude that &quot;We show that non-reversible models such as NREV12 should be evaluated during the model selection phase of phylogenetic analyses involving viral genomic sequences&quot;. This is a valuable finding, and I agree that this is potentially good practice. However, I miss an experiment that links the two findings to support the conclusion: in particular, an experiment that solves the following question: does the best-fit model also lead to better tree topologies?</p><p>[Editors' note: the reviewers were sent the revised submission and rebuttal and based on their response, an amended eLife Assessment has been formulated.]</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87361.3.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Sianga</surname><given-names>Rita</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Western Cape</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Hartnady</surname><given-names>Penelope</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Cape Town</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Mandikumba</surname><given-names>Wimbai Caroline</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Western Cape</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Rutherford</surname><given-names>Kayleigh</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Western Cape</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Currin</surname><given-names>Christopher Brian</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Cape Town</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Phelanyane</surname><given-names>Florence</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Western Cape</named-content></addr-line><country>South Africa</country></aff></contrib><contrib contrib-type="author"><name><surname>Stefan</surname><given-names>Sabina</given-names></name><role specific-use="author">Author</role><aff><institution>Brown University</institution><addr-line><named-content content-type="city">Providence</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Weaver</surname><given-names>Steven</given-names></name><role specific-use="author">Author</role><aff><institution>Temple University</institution><addr-line><named-content content-type="city">Pennsylvania</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Kosakovsky Pond</surname><given-names>Sergei L</given-names></name><role specific-use="author">Author</role><aff><institution>Temple University</institution><addr-line><named-content content-type="city">Pennsylvania</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Martin</surname><given-names>Darren P</given-names></name><role specific-use="author">Author</role><aff><institution>University of Cape Town</institution><addr-line><named-content content-type="city">Cape Town</named-content></addr-line><country>South Africa</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews</p><disp-quote content-type="editor-comment"><p><bold>eLife Assessment</bold></p><p>This valuable study revisits the effects of substitution model selection on phylogenetics by comparing reversible and non-reversible DNA substitution models. The authors provide evidence that (1) non time-reversible models sometimes perform better than general time-reversible models when inferring phylogenetic trees out of simulated viral genome sequence data sets, and that (2) non time-reversible models can fit the real data better than the reversible substitution models commonly used in phylogenetics, a finding consistent with previous work. However, the methods are incomplete in supporting the main conclusion of the manuscript, that is that non time-reversible models should be incorporated in the model selection process for these data sets.</p></disp-quote><p>The non-reversible models should be incorporated in the selection model process not because the significantly perform better but only because the do not perform worse than the reversible models and that true biochemical processes of nucleotide substitution does support the science of non-reversibility.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Public Review):</bold></p><p>The study by Sianga-Mete et al revisits the effects of substitution model selection on phylogenetics by comparing reversible and non-reversible DNA substitution models. This topic is not new, previous works already showed that non-reversible, and also covarion, substitution models can fit the real data better than the reversible substitution models commonly used in phylogenetics. In this regard, the results of the present study are not surprising. Specific comments are shown below.</p></disp-quote><p>True</p><disp-quote content-type="editor-comment"><p>It is well known that non-reversible models can fit the real data better than the commonly used reversible substitution models, see for example,</p><p><ext-link ext-link-type="uri" xlink:href="https://academic.oup.com/sysbio/article/71/5/1110/6525257">https://academic.oup.com/sysbio/article/71/5/1110/6525257</ext-link></p><p><ext-link ext-link-type="uri" xlink:href="https://onlinelibrary.wiley.com/doi/10.1111/jeb.14147?af=R">https://onlinelibrary.wiley.com/doi/10.1111/jeb.14147?af=R</ext-link></p><p>The manuscript indicates that the results (better fitting of non-reversible models compared to reversible models) are surprising but I do not think so, I think the results would be surprising if the reversible models provide a better fitting.</p><p>I think the introduction of the manuscript should be increased with more information about non-reversible models and the diverse previous studies that already evaluated them. Also I think the manuscript should indicate that the results are not surprising, or more clearly justify why they are surprising.</p></disp-quote><p>The surprise in the findings is in NREV12 performing better than NREV6 for double stranded DNA viruses as it was expected that NREV6 would perform better given the biochemical processes discussed in the introduction.</p><disp-quote content-type="editor-comment"><p>In the introduction and/or discussion I missed a discussion about the recent works on the influence of substitution model selection on phylogenetic tree reconstruction. Some works indicated that substitution model selection is not necessary for phylogenetic tree reconstruction,</p><p><ext-link ext-link-type="uri" xlink:href="https://academic.oup.com/mbe/article/37/7/2110/5810088">https://academic.oup.com/mbe/article/37/7/2110/5810088</ext-link></p><p><ext-link ext-link-type="uri" xlink:href="https://www.nature.com/articles/s41467-019-08822-w">https://www.nature.com/articles/s41467-019-08822-w</ext-link></p><p><ext-link ext-link-type="uri" xlink:href="https://academic.oup.com/mbe/article/35/9/2307/5040133">https://academic.oup.com/mbe/article/35/9/2307/5040133</ext-link></p><p>While others indicated that substitution model selection is recommended for phylogenetic tree reconstruction,</p><p><ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com/science/article/pii/S0378111923001774">https://www.sciencedirect.com/science/article/pii/S0378111923001774</ext-link></p><p><ext-link ext-link-type="uri" xlink:href="https://academic.oup.com/sysbio/article/53/2/278/1690801">https://academic.oup.com/sysbio/article/53/2/278/1690801</ext-link></p><p><ext-link ext-link-type="uri" xlink:href="https://academic.oup.com/mbe/article/33/1/255/2579471">https://academic.oup.com/mbe/article/33/1/255/2579471</ext-link></p><p>The results of the present study seem to support this second view. I think this study could be improved by providing a discussion about this aspect, including the specific contribution of this study to that.</p></disp-quote><p>In our conclusion we have stated that:</p><p>The lack of available data regarding the proportions of viral life cycles during which genomes exist in single and double stranded states makes it difficult to rationally predict the situations where the use of models such as GTR, NREV6 and NREV12 might be most justified: particularly in light of the poor over-all performance of NREV6 and GTR relative to NREV12 with respect to describing mutational processes in viral genome sequence datasets. We therefore recommend case-by-case assessments of NREV12 vs NREV6 vs GTR model fit when deciding whether it is appropriate to consider the application of non-reversible models for phylogenetic inference and/or phylogenetic model-based analyses such as those intended to test for evidence of natural section or the existence of molecular clocks.</p><disp-quote content-type="editor-comment"><p>The real data was downloaded from Los Alamos HIV database. I am wondering if there were any criterion for selecting the sequences or if just all the sequences of the database for every studied virus category were analysed. Also, was any quality filter applied? How gaps and ambiguous nucleotides were considered? Notice that these aspects could affect the fitting of the models with the data.</p></disp-quote><p>We selected varying number of sequences of the database for every studied virus type. Using the software aliview we did quality filter by re-aligning the sequences per virus type.</p><disp-quote content-type="editor-comment"><p>How the non-reversible model and the data are compared considering the non-reversible substitution process? In particular, given an input MSA, how to know if the nucleotide substitution goes from state x to state y or from state y to state x in the real data if there is not a reference (i.e., wild type) sequence? All the sequences are mutants and one may not have a reference to identify the direction of the mutation, which is required for the non-reversible model. Maybe one could consider that the most abundant state is the wild type state but that may not be the case in reality. I think this is a main problem for the practical application of non-reversible substitution models in phylogenetics.</p></disp-quote><p>True</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Recommendations for the authors):</bold></p><p>The reversible and non-reversible models used in this study assume that all the sites evolve under the same substitution matrix, which can be unrealistic. This aspect could be mentioned.</p></disp-quote><p>Done</p><disp-quote content-type="editor-comment"><p>The manuscript indicates that &quot;a phylogenetic tree was inferred from an alignment of real sequences (Avian Leukosis virus) with an average sequence identity (API) of ~90%.&quot;. I was wondering under which substitution model that phylogenetic tree reconstruction was performed? could the use of that model bias posterior results in terms of favoring results based on such a model?</p></disp-quote><p>We have stated that the GTR+G model was used to reconstruct the tree. The use of the GTR+G model could yes bias the posterior results as we have stated in the paper too.</p><disp-quote content-type="editor-comment"><p>I was wondering which specific R function was used to calculate the weighted Robinson-Foulds metric. I think this should be included in the manuscript.</p></disp-quote><p>We stated that We used the weighted Robinson-Foulds metric (wRF; implemented in the R phangorn package (Schliep, 2011)⁠)</p><disp-quote content-type="editor-comment"><p>Despite a minority, several datasets fitted better with a reversible model than with a non-reversible model. I think that should be clearly indicated. In addition, in my opinion the AIC does not enough penalizes the number of parameters of the models and favors the non-reversible models over the reversible models, but this is only my opinion based on the definition of AIC and it is not supported. Thus, I think the comparison between phylogenetic trees reconstructed under different substitution models was a good idea (but see also my second major comment).</p></disp-quote><p>Noted</p><disp-quote content-type="editor-comment"><p>When comparing phylogenetic trees I was wondering if one should consider the effect of the estimation method and quality of the studied data? For example, should bootstrap values be estimated for all the ancestral nodes and only ancestral nodes with high support be evaluated in the comparison among trees?</p></disp-quote><p>Yes the estimation method and quality of the studied data should be considered. When using RF unlike wRF this will not matter but for weighted RF it does. When building the trees, using RaxML only high support nodes are added to the tree.</p><disp-quote content-type="editor-comment"><p>In Figure 3, I do not see (by eye) significant differences among the models. I see in the legend that the statistical evaluation was based on a t test but I am not much convinced. Maybe it is only my view. Exactly, which pairs of datasets are evaluated with the t test? Next, I would expect that the influence of the substitution model on the phylogenetic tree reconstruction is higher at large levels of nucleotide diversity because with more substitution events there is more information to see the effects of the model. However, the t test seems to show that differences are only at low levels of nucleotide diversity (and large DNR), what could be the cause of this?</p></disp-quote><p>The paired T-tests compares the wRF distances of the inferred tree real tree and the trees simulated using the GTR model verses the wRF distances of the inferred true tree from the trees simulated using the NREV12 model.</p><p>The reason why the influence of the NREV12 model on the tree reconstructed is not significantly higher at large levels of nucleotide diversity could be because at a certain level the DNR are simply unrealistic.</p><disp-quote content-type="editor-comment"><p>Can the user perform substitution model selection (i.e., AIC) among reversible and non-reversible substitution models with IQTREE? If yes, then doing that should be the recommendation from this study, correct?</p><p>But, can DNR be estimated from a real dataset? DNR seems to be the key factor (Figure 3) for the phylogenetic analysis under a proper model.</p></disp-quote><p>Substitution model selection can be performed among reversible and non-reversible using both HyPhy and IQTREE. And we have recommended that model tests should be done as a first step before tree building. Estimating DNR from real datasets requires a substation rate matrix of a non-reversible.</p><disp-quote content-type="editor-comment"><p>The manuscript has many text errors (including typos and incorrect citations). For example, many citations in page 20 show &quot;Error! Reference source not found.&quot;. I think authors should double check the manuscript before submitting. Also, some text is not formally written. For example, &quot;G represents gamma-distributed rates&quot;, rates of what? The text should be clear for readers that are not familiar with the topic (i.e., G represents gamma-distributed substitution rates among sites). In general, I recommend a detailed revision of the whole text of the manuscript.</p></disp-quote><p>Done</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>The authors evaluate whether non time reversible models fit better data presenting strand-specific substitution biases than time reversible models. Specifically, the authors consider what they call NREV6 and NREV12 as candidate non time-reversible models. On the one hand, they show that AIC tends to select NREV12 more often than GTR on real virus data sets. On the other hand, they show using simulated data that NREV12 leads to inferred trees that are closer to the true generating tree when the data incorporates a certain degree of non time-reversibility.</p><p>Based on these two experimental results, the authors conclude that &quot;We show that non-reversible models such as NREV12 should be evaluated during the model selection phase of phylogenetic analyses involving viral genomic sequences&quot;. This is a valuable finding, and I agree that this is potentially good practice.</p><p>However, I miss an experiment that links the two findings to support the conclusion: in particular, an experiment that solves the following question: does the best-fit model also lead to better tree topologies?</p></disp-quote><p>By NREV12 leading to inferred trees that are closer to the true generating tree as compared to GTR, it then shows that the best-fit model in this case being NREV12 leads to better tree topologies.</p><disp-quote content-type="editor-comment"><p>On simulated data, the significance of the difference between GTR and NREV12 inferences is evaluated using a paired t test. I miss a rationale or a reference to support that a paired t test is suitable to measure the significance of the differences of the wRF distance. Also, the results show that on average NREV12 performs better than GTR, but a pairwise comparison would be more informative: for how many sequence alignments does NREV12 perform better than GTR?</p></disp-quote><p>We have used the popular paired t-test as it is the most widely used when comparing means values between two matched samples where the difference of each mean pair is normally distributed. And the wRF distances do match the guidelines above.</p><p>The paired t-test contains the pairwise comparison and the boxplots side by side show the pairwise wRF comparisions.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Recommendations for the authors):</bold></p><p>The authors reference Baele et al., 2010 for describing NREV6 and NREV12. I suggest using the same name used in the referenced paper: GNR-SYM and GNR respectively. Although I do not think there is a standard name for these models, I would use a previously used one.</p></disp-quote><p>We have built studies based on the names NREV6 and NREV12. We would like to keep the naming as standard for our studies.</p><disp-quote content-type="editor-comment"><p>GTR and NREV12 models are already described in many other papers. I do not see the need to include such an extensive description. Also, a reference should be included to the discrete Gamma rate categories [1]</p></disp-quote><p>We included the extensive description to enable other readers who are not super familiar with these models better understanding since we have given the models our own naming different from those used in other papers.</p><p>We have added referencing for the discrete gamma rate as recommended. (Yang, 1994)</p><disp-quote content-type="editor-comment"><p>To evaluate the exhaustiveness and correctness of the results, I would recommend publishing as supplementary material the simulated data sets or the scripts for generating the data set, the scripts or command lines for the analysis, and the versions of the software used (e.g., IQTREE). Also, to strongly support the main conclusion of the manuscript, I suggest adding to the simulations section results the RF-distances of the best-fit selected model under AIC, AICc, and BIC as well.</p></disp-quote><p>We can go ahead and submit all the needed datasets. The simulated data RF-Distances results are available and will be submitted. We cannot however add them to the main document as this will create very long data tables.</p><disp-quote content-type="editor-comment"><p>In some instances, it is mentioned that the selection criterion used is AIC, while in others, AIC-c is referenced. Even in the table captions, both terms are mixed. It should be made clearer which criterion is being employed, as AIC is not suitable for addressing the overparameterization of evolutionary models, given that it does not account for the sample size. A previous pre-print of this article [2] does not mention AIC-c, but also explicitly includes the formulas for AIC that do not take the sample size into account, and reports the same results as this manuscript, what indicates that AIC and not AIC-c was used here. This should be clarified. It is recommended to use AIC-c instead of AIC, especially if the sample size to model parameters ratio is low [3]. Two things may be appointed here: some authors consider tree branch lengths as model free parameters and others do not. In this paper it is not specified how the model parameters are counted. AIC tends to select more parameterized models than AIC-c, and overparameterization can lead to different tree inferences, as evidenced in Hoff et al., 2016. Therefore, it is expected that NREV12 is more frequently selected than NREV6 and GTR.</p><p>In my opinion, a pairwise comparison between GTR and NREV12 performance is of great interest here, and the whiskers plots are not useful. Scatterplots would display the results better.</p></disp-quote><p>Boxplots are meant to offer a simplified view of the results as the paired t-tests does all of the comparisons. We shall provide the scatter plots as supplementary information so that readers can get full detailed plots as recommended.</p><disp-quote content-type="editor-comment"><p>Some references are missing.</p></disp-quote><p>Missing references added</p></body></sub-article></article>