<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">61805</article-id><article-id pub-id-type="doi">10.7554/eLife.61805</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Epidemiology and Global Health</subject></subj-group><subj-group subj-group-type="heading"><subject>Microbiology and Infectious Disease</subject></subj-group></article-categories><title-group><article-title>In-host population dynamics of <italic>Mycobacterium tuberculosis</italic> complex during active disease</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-127554"><name><surname>Vargas</surname><given-names>Roger</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-7116-5211</contrib-id><email>roger_vargas@g.harvard.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201404"><name><surname>Freschi</surname><given-names>Luca</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201403"><name><surname>Marin</surname><given-names>Maximillian</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201399"><name><surname>Epperson</surname><given-names>L Elaine</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201398"><name><surname>Smith</surname><given-names>Melissa</given-names></name><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201402"><name><surname>Oussenko</surname><given-names>Irina</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201405"><name><surname>Durbin</surname><given-names>David</given-names></name><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201400"><name><surname>Strong</surname><given-names>Michael</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201401"><name><surname>Salfinger</surname><given-names>Max</given-names></name><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-201397"><name><surname>Farhat</surname><given-names>Maha Reda</given-names></name><email>Maha_Farhat@hms.harvard.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con10"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Department of Systems Biology, Harvard Medical School</institution><addr-line><named-content content-type="city">Boston</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Biomedical Informatics, Harvard Medical School</institution><addr-line><named-content content-type="city">Boston</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>Center for Genes, Environment and Health, Center for Genes, National Jewish Health</institution><addr-line><named-content content-type="city">Denver</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution>Department of Genetics and Genomic Sciences, Icahn School of Medicine at Mount Sinai</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution>Icahn Institute of Data Sciences and Genomics Technology</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff6"><label>6</label><institution>Mycobacteriology Reference Laboratory, Advanced Diagnostic Laboratories, National Jewish Health</institution><addr-line><named-content content-type="city">Denver</named-content></addr-line><country>United States</country></aff><aff id="aff7"><label>7</label><institution>College of Public Health, University of South Florida</institution><addr-line><named-content content-type="city">Tampa</named-content></addr-line><country>United States</country></aff><aff id="aff8"><label>8</label><institution>Morsani College of Medicine, University of South Florida</institution><addr-line><named-content content-type="city">Tampa</named-content></addr-line><country>United States</country></aff><aff id="aff9"><label>9</label><institution>Pulmonary and Critical Care Medicine, Massachusetts General Hospital</institution><addr-line><named-content content-type="city">Boston</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Kana</surname><given-names>Bavesh D</given-names></name><role>Reviewing Editor</role><aff><institution>University of the Witwatersrand</institution><country>South Africa</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Kana</surname><given-names>Bavesh D</given-names></name><role>Senior Editor</role><aff><institution>University of the Witwatersrand</institution><country>South Africa</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>01</day><month>02</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e61805</elocation-id><history><date date-type="received" iso-8601-date="2020-08-05"><day>05</day><month>08</month><year>2020</year></date><date date-type="accepted" iso-8601-date="2021-01-25"><day>25</day><month>01</month><year>2021</year></date></history><permissions><copyright-statement>© 2021, Vargas et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Vargas et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-61805-v2.pdf"/><abstract><p>Tuberculosis (TB) is a leading cause of death globally. Understanding the population dynamics of TB’s causative agent <italic>Mycobacterium tuberculosis</italic> complex (Mtbc) in-host is vital for understanding the efficacy of antibiotic treatment. We use longitudinally collected clinical Mtbc isolates that underwent Whole-Genome Sequencing from the sputa of 200 patients to investigate Mtbc diversity during the course of active TB disease after excluding 107 cases suspected of reinfection, mixed infection or contamination. Of the 178/200 patients with persistent clonal infection &gt;2 months, 27 developed new resistance mutations between sampling with 20/27 occurring in patients with pre-existing resistance. Low abundance resistance variants at a purity of ≥19% in the first isolate predict fixation in the subsequent sample. We identify significant in-host variation in 27 genes, including antibiotic resistance genes, metabolic genes and genes known to modulate host innate immunity and confirm several to be under positive selection by assessing phylogenetic convergence across a genetically diverse sample of 20,352 isolates.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>genomics</kwd><kwd>infectious disease</kwd><kwd>Mycobacterium tuberculosis</kwd><kwd>sequencing</kwd><kwd>antibiotic resistance</kwd><kwd>microbial evolution</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Other</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>DGE1745303</award-id><principal-award-recipient><name><surname>Vargas</surname><given-names>Roger</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>K01 ES026835</award-id><principal-award-recipient><name><surname>Farhat</surname><given-names>Maha Reda</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01 AI55765</award-id><principal-award-recipient><name><surname>Farhat</surname><given-names>Maha Reda</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Bulk whole genome sequencing data can be used to study the genetic variation present in pathogenic bacterial populations over the time-course of a single infection within a host.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Tuberculosis (TB) and its causative pathogen <italic>Mycobacterium tuberculosis</italic> complex (Mtbc) remain a major public health threat (<xref ref-type="bibr" rid="bib89">World Health Organization, 2018</xref>). Yet the majority of individuals exposed to Mtbc clear or contain the infection, and only 5–10% of those infected develop active TB disease at some point in their lifetime (<xref ref-type="bibr" rid="bib59">Pai et al., 2016</xref>). While basic human immune mechanisms to Mtbc have been identified, attempts at effective vaccine development guided by these mechanisms have repeatedly failed (<xref ref-type="bibr" rid="bib24">Ernst, 2018</xref>). Hence, global efforts in disease control currently focus on scale up of directly observed therapy but achieving a universal and sustained cure remains a challenge. Mtbc is an obligate human pathogen (<xref ref-type="bibr" rid="bib31">Gagneux, 2018</xref>). Infection and disease involve a complex human host-pathogen interaction that is both physically and temporally heterogeneous (<xref ref-type="bibr" rid="bib50">Lin et al., 2014</xref>). Consequently, all selective forces acting on Mtbc will originate within the host, and the study of temporal dynamics of this is likely to inform antibiotic treatment (<xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>) and rational vaccine design (<xref ref-type="bibr" rid="bib24">Ernst, 2018</xref>).</p><p>Little is known about selection at short timescales, such as within single infections. Drug pressure may select for resistance-conferring mutations, thus an understanding of how the frequency of minor alleles changes longitudinally can inform optimal drug treatment (<xref ref-type="bibr" rid="bib19">Didelot et al., 2016</xref>; <xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>; <xref ref-type="bibr" rid="bib92">Zhang et al., 2016</xref>). Mtbc’s interaction with host immunity or metabolic pressures imposed by persistent active human infection may also exert selective pressures, the detection of which can inform vaccine design or host directed therapeutics. To elucidate these temporal dynamics, we aimed to study how genomic diversity arises in-host in Mtbc populations, employing a longitudinal sampling scheme from patients with active TB disease enriched for treatment failure and relapse.</p><p>The application of genome sequencing technologies to Mtbc isolates cultured from clinical samples has highlighted that infection consists of populations of Mtbc bacteria rather than single clones devoid of diversity (<xref ref-type="bibr" rid="bib17">Copin et al., 2016</xref>; <xref ref-type="bibr" rid="bib19">Didelot et al., 2016</xref>; <xref ref-type="bibr" rid="bib48">Lieberman et al., 2014</xref>; <xref ref-type="bibr" rid="bib47">Lieberman et al., 2011</xref>; <xref ref-type="bibr" rid="bib52">Marvig et al., 2015</xref>). Differences in observed allele frequencies captured using genome sequencing (<xref ref-type="fig" rid="fig1">Figure 1A</xref>) may represent a difference in the genetic composition of the infecting population, commonly referred to as heterogeneity. Mtbc population heterogeneity might be present within a host because (1) the host is infected with multiple strains or is re-infected by a new strain (consistent with mixed infection or re-infection) or (2) genetic diversity arises within the Mtbc population during infection due to selection or drift (<xref ref-type="bibr" rid="bib30">Ford et al., 2012</xref>; <xref ref-type="bibr" rid="bib34">Guerra-Assunção et al., 2015</xref>; <xref ref-type="bibr" rid="bib49">Lieberman et al., 2016</xref>). However, non-uniform sampling (<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>), selection during the in vitro culture process (<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>), laboratory contamination (<xref ref-type="bibr" rid="bib32">Goig et al., 2020</xref>; <xref ref-type="bibr" rid="bib90">Wyllie et al., 2018</xref>), sequencing error and mapping error all represent examples of experimental error that give rise to heterogeneity of low significance to host-pathogen interactions.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Selection of patients with longitudinal clonal infection.</title><p>(<bold>A</bold>) Allele frequency change between paired isolates <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mi>A</mml:mi><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mi>A</mml:mi><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>B</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>. (<bold>B</bold>) The F2 measure &gt;0.04 (Materials and methods) was used to identify and exclude isolate pairs with evidence for mixed strain growth at any time point. (<bold>C</bold>) Replicate and longitudinal pairs with fixed SNP (fSNP) distance of &gt;7 were excluded. For longitudinal isolates, fSNP &gt;7 was assessed as consistent with Mtbc reinfection with a different strain.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Filtering out laboratory-contaminated samples and patients with mixed infections.</title><p>We implemented several filters to mitigate the effects of contamination from laboratory error or samples from co-infected hosts (Materials and methods). Our analysis included three types of replicate pairs (S2, C2, P3) and longitudinal pairs from eight studies (P, C, W, T, B, G, X, H) (Materials and methods). At each step, we filtered out any pair of isolates if at least one isolate failed to pass the filter in place (indicated by dashed arrows). First, we used Kraken to filter out isolates that had less than 95% of reads taxonomically classified under MTBC. Second, we filtered out isolates that did not meet the F2 threshold. Third, we filtered out isolate pairs that had a genetic distance greater than seven fixed SNPs. Our final filtered isolate pair sets included 62 replicate isolate pairs and 200 longitudinal isolate pairs.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig1-figsupp1-v2.tif"/></fig></fig-group><p>Here, we present a framework to overcome these barriers and demonstrate the use of longitudinally collected isolates, pooled sweeps of colonies cultured from sputa, to investigate true in-host diversity with implications for Mtbc treatment. We analyzed 614 paired longitudinal isolates representing 307 patients from eight studies (<xref ref-type="bibr" rid="bib7">Bryant et al., 2013</xref>; <xref ref-type="bibr" rid="bib8">Casali et al., 2016</xref>; <xref ref-type="bibr" rid="bib34">Guerra-Assunção et al., 2015</xref>; <xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>; <xref ref-type="bibr" rid="bib85">Walker et al., 2013</xref>; <xref ref-type="bibr" rid="bib87">Witney et al., 2017</xref>; <xref ref-type="bibr" rid="bib91">Xu et al., 2018</xref>). Many patients, despite undergoing treatment, remained culture positive at 2 months intervals or longer meeting microbiological criteria for delayed culture conversion, treatment failure or relapse (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>–<xref ref-type="supplementary-material" rid="supp2">2</xref>). Our sample consisted of 178 patients fulfilling these criteria, which allowed us to overcome the small sample size problem present in prior studies. We provide a proof of concept that whole-genome sequencing (WGS) can aide in predicting resistance amplification and demonstrate that in addition to loci involved in the acquisition of antibiotic resistance, loci implicated in modulation of innate host-immunity appear to be under positive selection.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Identifying clonal Mtbc populations in-host</title><p>Of the 307 patients with longitudinal samples collected (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>–<xref ref-type="supplementary-material" rid="supp2">2</xref>), 32 patients had evidence for isolate microbiological contamination at any time point (<xref ref-type="bibr" rid="bib32">Goig et al., 2020</xref>) and were excluded. We found evidence for mixed infection with two or more Mtbc lineages (<xref ref-type="bibr" rid="bib90">Wyllie et al., 2018</xref>) for 31 patients (<xref ref-type="fig" rid="fig1">Figure 1B</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>); 44 patients had evidence for re-infection with a different Mtbc strain between the first and second time points, using a pairwise genetic distance &gt;7 fixed SNPs (fSNPs) (Materials and methods, <xref ref-type="fig" rid="fig1">Figure 1C</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Median fSNP distance for the 44 patients identified as reinfection was 708 (IQR 250–1086). The remaining 200 patients were accordingly identified as having persistent or relapsed clonal infection (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Isolates from these infections spanned five of the eight known Mtbc lineages (Figure 5A). We implemented WGS SNP calling filters to minimize the likelihood of false positive SNP calls and validated calls with simulation and PacBio long-read data. We required that no indels be present in any of the reads supporting any SNP call, dropped SNP calls in repetitive regions and enforced a read depth ≥25x and alternate allele depth of ≥5 reads. We estimated the false error rate of our analysis pipeline for detecting allele frequency changes between sampling times at ≤0.053 using a control dataset of 82 isolate pairs (162 total) that were in vitro technical or biological replicates (Materials and methods, Figure 4 and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>).</p></sec><sec id="s2-2"><title>In-host pathogen dynamics in antibiotic resistance loci</title><p>Of the 200 patients with clonal infection, we had complete treatment data on 127 patients. Six of the 127 patients had isolates sampled &lt;2 months apart, and the remaining 121 had an outcome at the second sampling consistent with delayed culture conversion, failure or relapse of their clonal infection, hitherto treatment failure for brevity. Treatment regimen details are provided in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>. Of the other 73 patients, 49 patients were sampled ≥2 months apart during treatment but regimen details or interruptions were not available, for these patients the outcome may have been either default or failure. The remaining 24 patients had inadequate treatment data to confirm treatment outcome. We conducted all analyses focused on antibiotic resistance loci on 200 patients with isolate date data and separately on the 121-patient subset with confirmed failure (the latter detailed in Appendix 3). For all 200 cases, the order of sampling was available, but for 195/200 (119/121 confirmed failure patients) we also had the exact dates of sampling which were required for some analyses.</p><p>Resistance mutations found at low frequencies in-host may indicate the impending development of clinical resistance (<xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>; <xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>; <xref ref-type="bibr" rid="bib92">Zhang et al., 2016</xref>).﻿ To investigate temporal dynamics related to antibiotic pressure, we identified non-synonymous and intergenic SNPs within a set of 36 predetermined resistance loci associated with antibiotic resistance (<xref ref-type="bibr" rid="bib27">Farhat et al., 2016</xref>; <xref ref-type="bibr" rid="bib25">Farhat et al., 2013</xref>; <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) that changed in allele frequency by ≥5% (<xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>) and ensuring that support of the alternate allele was ≥5 reads at each time point (Materials and methods). We detected 1939 such SNPs across our sample of 200 patients (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), 1774 were non-synonymous, 91 were intergenic, and 74 occurred within the <italic>rrs</italic> region (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Allele frequency dynamics within antibiotic resistance loci.</title><p>(<bold>A</bold>) The antibiotic resistance genes <italic>embB</italic>, <italic>katG</italic>, <italic>gyrA</italic>, <italic>ethA</italic>, and <italic>pncA</italic> demonstrate evidence for competing clones during infection (other examples found are displayed in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). Each mutant allele is labeled with amino acid encoded by the reference allele, H37Rv codon position, and amino acid encoded by the mutant allele. (<bold>B</bold>) The allele frequency trajectories for SNPs that occur in patients over the course of infection can be used to study the prediction of further antibiotic resistance using the frequency of alternate alleles detected in the longitudinal isolates collected from patients. (<bold>C</bold>) Plot of true positive rate (TPR) and false positive rate (FPR) for detecting eventual fixation of a resistance allele (<inline-formula><mml:math id="inf2"><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo> <mml:mi/></mml:math></inline-formula> 75%) as a function of initial allele frequency (<inline-formula><mml:math id="inf3"><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Mutant allele trajectories consistent with clonal interference.</title><p>Several examples of co-occurring mutant alleles and their allele frequency trajectories between longitudinal isolate collection demonstrate genetic diversity patterns consistent with competing clones in-host. Each mutant allele is labeled with amino acid encoded by the reference allele, H37Rv codon position, and amino acid encoded by the mutant allele.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig2-figsupp1-v2.tif"/></fig></fig-group><p>We searched for signs of selection by identifying clonal interference, or evidence of competition between strains with different drug resistance mutations (<xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>; <xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>; <xref ref-type="bibr" rid="bib92">Zhang et al., 2016</xref>). We characterized this in longitudinal isolates fulfilling three criteria: (i) isolates containing multiple resistance SNPs in the same gene within the same patient, (ii) at alternate allele frequencies that change in opposing directions over time, and (iii) the alternate (mutant) allele frequency was intermediate to high at ≥40% in at least one isolate (<xref ref-type="bibr" rid="bib27">Farhat et al., 2016</xref>) for at least one of the co-occurring SNPs. This identified 11 cases of clonal interference (<xref ref-type="fig" rid="fig2">Figure 2A</xref> and <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>), demonstrating most often the fixation of a single allele in the second isolate from a mixture of multiple alleles at lower frequencies in the first isolate collected.</p></sec><sec id="s2-3"><title>Allele frequency &gt;19% predicts subsequent fixation of resistance variants</title><p>We aimed to measure the lowest AR allele frequency that can accurately predict the fixation of resistance alleles later in time (<xref ref-type="bibr" rid="bib22">Dreyer et al., 2020</xref>; <xref ref-type="bibr" rid="bib74">Sun et al., 2012</xref>; <xref ref-type="bibr" rid="bib92">Zhang et al., 2016</xref>). We examined all 1919 SNPs that varied by at least 5% in allele frequency (AF), and discarded 20 SNPs that were fixed at AF &gt;75% in both isolates. We calculated the true positive rate (TPR) and false positive rate (FPR) for varying values of AF at the first time point (<inline-formula><mml:math id="inf4"><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo><mml:mo>∈</mml:mo><mml:mfenced close="}" open="{" separators="|"><mml:mrow><mml:mn>0,1</mml:mn><mml:mo>,</mml:mo> <mml:mi/><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>⋯</mml:mo><mml:mo>,</mml:mo><mml:mn>99,100</mml:mn></mml:mrow></mml:mfenced><mml:mi>%</mml:mi></mml:math></inline-formula> (<xref ref-type="fig" rid="fig2">Figure 2C</xref>, Materials and methods) allowing a maximum FPR of 5%. We found the optimal classification threshold to be <inline-formula><mml:math id="inf5"><mml:msubsup><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>*</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>19</mml:mn><mml:mi>%</mml:mi></mml:math></inline-formula> with an associated sensitivity of 27.0% and a specificity of 95.8%. Of the total 37 alleles that became fixed at the second time point, 10 (from seven patients) had a frequency between 19% and 75% at the first time point, two were detected at the first time point but had AF &lt;19%, and the remaining majority, or 25, were undetectable (i.e. had support of &lt;5 reads) at the first time point. Taken together, we find a high turnover of low-frequency alleles in loci associated with antibiotic resistance but that mutant alleles in these loci that rise to a frequency of 19% are predicted to fix in-host with a sensitivity of 27.0% and specificity of 95.8%.</p></sec><sec id="s2-4"><title>Determinants of antibiotic resistance acquisition and microbiological treatment failure</title><p>We identified overall rates of resistance acquisition by focusing on AR SNPs with moderate to high ΔAF ≥ 40% given prior evidence of association between such SNPs and phenotypic resistance (<xref ref-type="bibr" rid="bib27">Farhat et al., 2016</xref>).</p><p>Twenty-seven AR SNPs were acquired in the 178 patients with persistent or relapsed clonal infection ≥2 months (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Among the set of 119 patients with confirmed failure and known isolate sampling date, 9% (11/119) of these patients acquired ≥1 AR SNP. Of the 11, 9 patients received fewer than four effective drugs. We examined the relationship between pre-existing resistance and new AR acquisition. Pre-existing resistance was defined as ≥1 fixed AR SNPs in the first isolate (<xref ref-type="bibr" rid="bib27">Farhat et al., 2016</xref>) (Materials and methods). Two hundred fifty-nine pre-existing AR SNPs were identified with 41% (73/178) of failure patients harboring resistance to any drug at the first sampling (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, <xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>). The majority of this resistance was MDR (multidrug resistance to at least isoniazid and a rifamycin), 64% (47/73) (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). New resistance acquisition occurred mostly in patients with pre-existing resistance 20/27 (74%) (<inline-formula><mml:math id="inf6"><mml:mi>O</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mn>5.28</mml:mn><mml:mo>,</mml:mo> <mml:mi/> <mml:mi/><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>2.2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> Fisher’s exact test) or pre-existing MDR (<inline-formula><mml:math id="inf7"><mml:mi>O</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mn>3.85</mml:mn><mml:mo>,</mml:mo> <mml:mi/> <mml:mi/><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>3.4</mml:mn> <mml:mi/><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> Fisher’s exact test). Among the set of 195/200 patients with clonal infection and sampling date, AR acquisition was more likely as the time between sampling increased with the OR of AR acquisition being 1.023 per 30day increment (<inline-formula><mml:math id="inf8"><mml:mn>95</mml:mn><mml:mi>%</mml:mi> <mml:mi/><mml:mi>C</mml:mi><mml:mi>I</mml:mi> <mml:mi/><mml:mn>1.002</mml:mn><mml:mo>,</mml:mo> <mml:mi/><mml:mn>1.045</mml:mn><mml:mo>,</mml:mo> <mml:mi/> <mml:mi/><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.035</mml:mn></mml:math></inline-formula> Logistic Regression).</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Pre-existing resistance is associated with resistance amplification.</title><p>We called heterozygous SNPs (hSNP) in each isolate from a patient with clonal infection classified as failing treatment (N = 178). We defined hSNPs as a SNP called in an isolate with an alternate allele frequency between 25% and 75% (Materials and methods). (<bold>A</bold>) The number of hSNPs called in the second sample isolated vs the number of hSNPs called in the first sample isolated from each of 178 patients (median T1 = 13.5 hSNPs, median T2 = 13.5 hSNPs). The dashed line is y = x. Red denotes 27/178 patients who had an antibiotic resistance in-host SNP arise between sampling (median T1 = 15.0 hSNPs, median T2 = 11.0 hSNPs), blue denotes 5/178 patients who had a putative host-adaptive in-host SNP (Rv1944c, Rv0095c, <italic>PPE18</italic>, <italic>PPE54</italic>, <italic>PPE60</italic>) arise between sampling (median T1 = 19.0 hSNPs, median T2 = 6.0 hSNPs). (<bold>B–C</bold>) Among patients who fail treatment, (<bold>B</bold>) patients with pre-existing mutations that confer antibiotic resistance and (<bold>C</bold>) those that have pre-existing MDR are more likely to acquire antibiotic resistance mutations throughout the course of infection.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig3-v2.tif"/></fig><p>We also quantified genome-wide Mtbc diversity in-host among the patients with persistent or relapsed infection for ≥2months. We reasoned that if these patients are not on or not adherent to effective antibiotic treatment, their effective pathogen population size may be large and prone to more genetic drift or turnover of minority variants with and without selection (<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>). We counted the number SNPs with an alternate allele frequency between 25% and 75% (Materials and methods) at each time point as a conservative estimate of the number of segregating sites in each population. We found this count to strongly correlate between the first and second time point (<xref ref-type="fig" rid="fig3">Figure 3A</xref>) suggesting that minor allele diversity is maintained in-host in patients without effective therapy (median T1=13.5 hSNPs, median T2=13.5 hSNPs, <inline-formula><mml:math id="inf9"><mml:msup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0.426</mml:mn><mml:mo>,</mml:mo> <mml:mi/> <mml:mi/><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>5.97</mml:mn> <mml:mi/><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>23</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> Linear Regression).</p></sec><sec id="s2-5"><title>Genome-wide in-host diversity</title><p>Beyond antibiotic pressure, selective forces acting on the infecting Mtbc population in-host are largely unknown. To investigate this reliably across the entire Mtbc genome, we first examined the genome-wide allele frequency distribution for both technical replicates (in vitro technical or biological replicates, sample size m=62 after exclusions) and in-host longitudinal pairs (<xref ref-type="fig" rid="fig4">Figure 4</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). We detected five SNPs in <italic>glpK</italic> (with ΔAF ≥ 25%) among five replicate pairs (mean ΔAF = 45%) consistent with an adaptive role for <italic>glpK</italic> mutations in vitro (<xref ref-type="bibr" rid="bib62">Pethe et al., 2010</xref>; <xref ref-type="bibr" rid="bib81">Vargas and Farhat, 2020</xref>) and accordingly excluded this gene from further analysis (Materials and methods). The genome-wide AF distribution in both replicate and longitudinal pairs demonstrated an abundance of SNPs with low ΔAF likely resulting from noise or technical factors. To clearly distinguish signal related to in-host factors from noise, we determined the ΔAF threshold above which SNPs/isolate-pair were rare among technical replicates that is, constituted 5% or less of total SNPs (<xref ref-type="fig" rid="fig4">Figure 4</xref>). We determined this ΔAF threshold to be 70% and selected 174 SNPs that developed in-host (in-host SNPs) among the 200 TB cases (<xref ref-type="fig" rid="fig4">Figure 4C</xref>, <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). Using archived MTBC isolates, we observe that changes in allele frequency are common among replicate isolates and changes in frequency of 70% are indicative of in-host evolution.</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Replicate pairs reveal levels of biological noise associated with repeated sampling.</title><p>(<bold>A, B</bold>) We analyzed the distribution of ΔAF for all SNPs detected across all replicate pairs <inline-formula><mml:math id="inf10"><mml:mo>(</mml:mo><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>62</mml:mn><mml:mo>)</mml:mo></mml:math></inline-formula> and longitudinal pairs <inline-formula><mml:math id="inf11"><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>200</mml:mn><mml:mo>)</mml:mo></mml:math></inline-formula> for SNPs where ΔAF ≥25%. (<bold>B</bold>) SNPs were detectable at lower levels of ΔAF for both types of isolate pairs, but SNPs with higher values of ΔAF were only found in longitudinal pairs. (<bold>C</bold>) To determine a ΔAF threshold for calling SNPs representative of changes in bacterial population composition in-host, we calculated the average number of SNPs per pair of isolates at different ΔAF thresholds for both replicate and longitudinal pairs. At a ΔAF threshold of 70% the number of SNPs between replicate pairs represents 5.27% of the SNPs detected amongst all replicate and longitudinal pairs, weighted by the number of pairs in each group.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig4-v2.tif"/></fig></sec><sec id="s2-6"><title>Characteristics of mutations in-host</title><p>Of the 174 SNPs, 112 were non-synonymous, 42 synonymous, and 21 were intergenic (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). The 153/174 coding SNPs were distributed across 127/3,886 genes and were observed in 71/200 patients (<xref ref-type="fig" rid="fig5">Figure 5B</xref> and <xref ref-type="fig" rid="fig5">Figure 5D</xref>). We analyzed the spectrum of mutations and found the GC &gt; AT nucleotide transition to be the most common. The GC &gt; AT transition is putatively due to oxidative damage including the deamination of cytosine/5-methyl-cytosine or the formation of 8-oxoguanine (<xref ref-type="bibr" rid="bib20">Dillon et al., 2015</xref>; <xref ref-type="bibr" rid="bib29">Ford et al., 2011</xref>). The transversion AT &gt; TA was the least common substitution (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). We expected the number of SNPs detected between longitudinal isolates to increase with time between isolate collection. Regressing the number of SNPs per patient on the timing between isolate collection (for 195 patients with isolate collection dates) (<xref ref-type="fig" rid="fig6">Figure 6B</xref>), we found SNPs to accumulate at an average rate of 0.56 SNPs per genome per year (<inline-formula><mml:math id="inf12"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>7</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>12</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) consistent with prior in vivo estimates (<xref ref-type="bibr" rid="bib29">Ford et al., 2011</xref>; <xref ref-type="bibr" rid="bib85">Walker et al., 2013</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Genome-wide diversity in 200 clonal Mtbc infections.</title><p>(<bold>A</bold>) Distribution of five major Mtbc lineages among the 200 clonal Mtbc infections. (<bold>B</bold>) Distribution of 153 in-host SNPs within coding regions among the 200 longitudinal isolate pairs across the 4.41 Mbp Mtbc genome (blue circles: synonymous, red circles: non-synonymous). Blue and red circles on the innermost black ring indicate the locations of SNPs detected in one patient; circles on the next ring represent SNPs detected in two patients. The <inline-formula><mml:math id="inf13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> (p-value) of the mutational density test (Materials and methods) by gene is plotted in the outermost, red and green, regions. Labeled yellow circles represent genes significant at the bonferroni-corrected cutoff (<inline-formula><mml:math id="inf14"><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn><mml:mo>/</mml:mo><mml:mn>3,886</mml:mn></mml:math></inline-formula>). (<bold>C</bold>) Distribution of ΔAF by SNP type: sSNP: synonymous, nSNP: non-synonymous, iSNP: intergenic. (<bold>D</bold>) Heat-map of SNPs per gene (rows) and patient (columns). Colored circles across columns indicate the strain phylogenetic lineage (as represented in <bold>A</bold>). Gene names colored according to gene category (<xref ref-type="fig" rid="fig6">Figure 6D</xref>) with parentheses indicating the number of patients with an SNP in a given gene. *Indicates genes in which SNPs are detected within multiple patients.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>In-host SNP detected in <italic>PPE18</italic>.</title><p>IGV (<xref ref-type="bibr" rid="bib76">Thorvaldsdóttir et al., 2013</xref>) image of BAM alignment (reads sorted by start location) for longitudinal clinical isolates that were cultured from sputum collected from patient P000183 (<xref ref-type="bibr" rid="bib85">Walker et al., 2013</xref>). (<bold>A</bold>) 500 and (<bold>B</bold>) 3000 basepair windows centered at reference position 1339741. Isolate 1 is the BAM alignment for the isolate collected in 2003 and the reference position 1339741 matches the reference allele (<bold>C</bold>). Isolate 2 is the BAM alignment for isolate collected in 2008 and reference position 1339741 supports an alternate allele (<bold>G</bold>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig5-figsupp1-v2.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>In-host SNP detected in <italic>PPE54</italic>.</title><p>IGV (<xref ref-type="bibr" rid="bib76">Thorvaldsdóttir et al., 2013</xref>) image of BAM alignment (reads sorted by start location) for longitudinal clinical isolates that were cultured from sputum collected from patient P09 (<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>). (<bold>A</bold>) 500 and (<bold>B</bold>) 3000 basepair windows centered at reference position 3730411. Isolate 1 is the BAM alignment for the isolate collected first and the reference position 3730411 matches the reference allele (<bold>G</bold>). Isolate 2 is the BAM alignment for isolate collected 24 weeks after isolate 1 and reference position 3730411 supports an alternate allele (<bold>A</bold>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig5-figsupp2-v2.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>In-host SNPs detected in <italic>PPE60</italic>.</title><p>IGV (<xref ref-type="bibr" rid="bib76">Thorvaldsdóttir et al., 2013</xref>) image of BAM alignment (reads sorted by start location) for longitudinal clinical isolates that were cultured from sputum collected from patient 3096 (<xref ref-type="bibr" rid="bib28">Farhat et al., 2019</xref>). (<bold>A</bold>) 500 and (<bold>B</bold>) 3000 basepair windows centered at reference position 3895269. Isolate 1 is the BAM alignment for the isolate collected on September 25, 2001 and the reference positions 3895269, 3895281, 3895282 match the reference alleles (G,T,G respectively). Isolate 2 is the BAM alignment for isolate collected on October 18, 2002 and reference positions 3895269, 3895281, 3895282 support alternate alleles (C,C,A respectively).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig5-figsupp3-v2.tif"/></fig></fig-group><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>PE/PPE genes vary considerably within host while putative antigens remain conserved.</title><p>(<bold>A</bold>) Mutational spectrum of in-host SNPs. (<bold>B</bold>) In-host SNP counts vs. time between isolate collection (195/200 patients with dates shown, *W [<xref ref-type="bibr" rid="bib85">Walker et al., 2013</xref>] isolates only had year of collection). (<bold>C</bold>) Boxplots of nucleotide diversity by gene within each of five non-redundant categories (see text; <inline-formula><mml:math id="inf15"><mml:mi>n</mml:mi><mml:mo>=</mml:mo></mml:math></inline-formula> number of genes). (<bold>D</bold>) Average nucleotide diversity across genes by category. Nucleotide diversity in epitope and non-epitope region (Materials and methods) of each gene in the Antigen (<bold>E, F</bold>) and PE/PPE (<bold>G, H</bold>) gene categories. (<bold>l, J</bold>) PE/PPE genes separated into three non-redundant categories: PE, PE-PGRS, and PPE. (<bold>J</bold>) The average nucleotide diversity by category. (<bold>I</bold>) Box plot of nucleotide diversity by gene.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig6-v2.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Basic characteristics of epitopes used in analysis.</title><p>We downloaded a set of 2031 epitope peptide sequences from IEDB (<xref ref-type="bibr" rid="bib83">Vita et al., 2015</xref>) and used BLASTP to map these peptide sequences to H37Rv imposing an e-value cut-off of 0.01 (Materials and methods). (<bold>A</bold>) The distribution of e-values and (<bold>B</bold>) distribution of peptide lengths for the retained epitope mappings.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig6-figsupp1-v2.tif"/></fig><fig id="fig6s2" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 2.</label><caption><title>Most T cell epitopes remain conserved in-host during active TB disease.</title><p>No SNPs were detected in-host for a vast majority of CD4<sup>+</sup> and CD8<sup>+</sup> T cell epitopes, however, 1 SNP was detected in a small number <inline-formula><mml:math id="inf16"><mml:mfenced separators="|"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:mfenced></mml:math></inline-formula> of overlapping epitopes in PPE18. A list of these epitopes is given in <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig6-figsupp2-v2.tif"/></fig></fig-group></sec><sec id="s2-7"><title>Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions</title><p>Several SNPs detected were in the GC-rich repetitive PE/PPE gene family (<xref ref-type="bibr" rid="bib4">Brennan and Delogu, 2002</xref>). Variants called on these genes are commonly excluded from comparative genomic analyses (<xref ref-type="bibr" rid="bib8">Casali et al., 2016</xref>; <xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>; <xref ref-type="bibr" rid="bib17">Copin et al., 2016</xref>; <xref ref-type="bibr" rid="bib18">Coscolla et al., 2015</xref>) due to the limitations of short-read sequencing data and the possibility of making spurious variant calls; however, the rates at which these false calls occur has not been evaluated. We reasoned that our stringent filtering criteria, quality of sequencing data and depth of coverage allowed us to reliably detect variants in these regions of the genome, with the potential to uncover variation in these understudied regions of the genome.</p><p>We took several approaches to test the rate of false-positives for the single base-pair mutations observed in our analysis (Materials and methods). First, we introduced the mutant alleles observed in-host (<xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>) into a set of Mtbc reference genomes belonging to different lineages and simulated short read sequencing data from these modified genomes (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>). We then used our variant calling pipeline to call bases from this simulated data. We observed a high recall rate of the introduced mutant alleles and a very low number of false positive base calls (zero in most cases) within the loci containing modified alleles (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>). Second, we assessed the congruence in variant calls between short-read Illumina data and long-read PacBio data for a set of isolates that underwent sequencing with both technologies (Materials and methods). Unlike Illumina generated reads, PacBio reads are much longer and have randomly distributed error profiles (<xref ref-type="bibr" rid="bib65">Rhoads and Au, 2015</xref>). With high coverage, PacBio sequencing can reliably reconstruct full microbial genomes and identify SNPs in repetitive regions. The comparison with PacBio assemblies confirmed empirically a low rate of false positive base calls in genomic regions where we observed in-host SNPs (Materials and methods). Third, we confirmed the five phylogenetically convergent in-host SNPs in PPE genes <italic>PPE18</italic>, <italic>PPE54</italic>, and <italic>PPE60</italic> (see below) through manual inspection of the read alignment (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplements 1</xref>–<xref ref-type="fig" rid="fig5s3">3</xref>).</p></sec><sec id="s2-8"><title>Antibiotic resistance and PE/PPE genes vary while antigens remain conserved</title><p>To understand how different classes of proteins evolve in-host, we separated Mtbc genes into five non-redundant categories (Materials and methods). The vast majority of genes in each category did not vary within patients (<xref ref-type="fig" rid="fig6">Figure 6C</xref>). Antibiotic resistance genes were on average the most diverse category while Essential genes varied the least (<xref ref-type="fig" rid="fig6">Figure 6D</xref>). Antigen genes appeared to be as conserved as were both Essential (<inline-formula><mml:math id="inf17"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.49</mml:mn></mml:math></inline-formula> Mann-Whitney U-test) and Non-Essential genes (<inline-formula><mml:math id="inf18"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.45</mml:mn></mml:math></inline-formula> Mann-Whitney U-test) while PE/PPE genes showed higher levels of nucleotide diversity than both Essential (<inline-formula><mml:math id="inf19"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.022</mml:mn></mml:math></inline-formula> Mann-Whitney U-test) and Non-Essential genes (<inline-formula><mml:math id="inf20"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.013</mml:mn></mml:math></inline-formula> Mann-Whitney U-test) (<xref ref-type="fig" rid="fig6">Figure 6D</xref>).</p></sec><sec id="s2-9"><title>PE/PPE variation is independent of T-cell recognition</title><p>To test whether variation in Antigen or PE/PPE genes occurred in response to T-cell recognition, we separated each gene in these categories into (CD4<sup>+</sup> and CD8<sup>+</sup> T-cell) epitope and non-epitope concatenates and recalculated nucleotide diversity for these concatenates (<xref ref-type="fig" rid="fig6">Figure 6E–H</xref>). For both Antigen and PE/PPE genes (<xref ref-type="fig" rid="fig6">Figure 6F</xref> and <xref ref-type="fig" rid="fig6">Figure 6H</xref>), epitope concatenates were less diverse than non-epitope concatenates (<inline-formula><mml:math id="inf21"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.018</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf22"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.059</mml:mn></mml:math></inline-formula>, respectively, Mann-Whitney U-test). Only one in-host SNP was detected within an epitope-encoding region in the gene <italic>PPE18</italic> (<xref ref-type="fig" rid="fig6">Figure 6G</xref> and <xref ref-type="fig" rid="fig7s2">Figure 7—figure supplement 2</xref>, <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>). This suggests that T-cell recognition does not drive diversity in these regions. Looking within the three PE/PPE subfamilies (<xref ref-type="fig" rid="fig6">Figure 6I–J</xref>; <xref ref-type="bibr" rid="bib3">Brennan, 2017</xref>), the PPE genes appeared more diverse in-host than PE genes and PE-PGRS genes (<inline-formula><mml:math id="inf23"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.019</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf24"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>0.033</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> respectively, Mann-Whitney U-test).</p></sec><sec id="s2-10"><title>Identifying candidate pathoadaptive loci from genome-wide variation</title><p>To identify genes involved in pathogen adaptation (<xref ref-type="bibr" rid="bib47">Lieberman et al., 2011</xref>; <xref ref-type="bibr" rid="bib52">Marvig et al., 2015</xref>), we applied a test of mutational density (<xref ref-type="bibr" rid="bib26">Farhat et al., 2014</xref>; Materials and methods) by pooling variation across all 200 pairs of genomes and identifying those genes with more mutations than expected under a neutral model of evolution where variants are Poisson distributed across the genome (<xref ref-type="bibr" rid="bib26">Farhat et al., 2014</xref>; <xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>, Materials and methods). We also searched for evidence of convergent evolution, that is, genes or pathways where in-host SNPs developed in ≥ 2 patients (<xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>, <xref ref-type="supplementary-material" rid="supp17">Supplementary file 17</xref>). Seven known antibiotic resistance genes (<xref ref-type="bibr" rid="bib19">Didelot et al., 2016</xref>; <xref ref-type="bibr" rid="bib25">Farhat et al., 2013</xref>) had significant mutational density (<inline-formula><mml:math id="inf25"><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math></inline-formula>, Bonferroni correction) or were convergent across patients: <italic>rpoB</italic>, <italic>gyrA</italic>, <italic>katG</italic>, <italic>rpoC</italic>, <italic>embB, ethA</italic> and <italic>pncA</italic> (mutated in six, four, four, three, three, two, and one patient, respectively) (<xref ref-type="fig" rid="fig5">Figure 5B</xref> and <xref ref-type="fig" rid="fig5">Figure 5D</xref>). Single in-host SNPs occurred in eight additional known resistance loci including three intergenic regions, and in <italic>prpR,</italic> a gene recently implicated with drug tolerance (<xref ref-type="bibr" rid="bib36">Hicks et al., 2018</xref>; <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>).</p><p>Three genes with unknown function: Rv0139, Rv0895, and Rv1543 were convergent in two patients each, two of which (Rv0139, Rv1543) had significant mutational density (p&lt;2×10<sup>−5</sup>) and; three additional genes including <italic>PPE60</italic> displayed significant mutational density (p&lt;2×10<sup>−5</sup>) (<xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>). We found evidence for convergence in six pathways not known to result in antibiotic resistance. These pathways are involved with biotin biosynthesis (<italic>fadD23</italic>, <italic>fadD29</italic>, and <italic>fadD30</italic>), ribosomal large subunit proteins (<italic>rpmB1</italic>, <italic>rplE</italic>, and <italic>rplY</italic>), glycerolipid and glycerophospholipid metabolism (<italic>aldA</italic> and Rv2974c), ESAT-6 protein secretion (<italic>eccCa1</italic> and <italic>eccD1</italic>), coenzyme B12/cobalamin synthesis (<italic>cobH</italic> and <italic>cobK</italic>) and the uncharacterized pathway CBSS-164757.7.peg.5020 (<italic>fdxB</italic> and <italic>PPE18</italic>) (<xref ref-type="supplementary-material" rid="supp17">Supplementary file 17</xref>).</p></sec><sec id="s2-11"><title>In-host mutations display phylogenetic convergence across multiple global lineages</title><p>We reasoned that pathoadaptive mutations observed to sweep to fixation in-host and not compromise pathogen transmissibility are likely to arise independently within other patients and in separate geographic regions in a convergent manner (<xref ref-type="bibr" rid="bib25">Farhat et al., 2013</xref>). We screened a geographically diverse set of 20,352 sequenced clinical isolates belonging to global lineages 1–6 for mutations observed within host in which the alternate (mutant) allele swept over the course of sampling (141/174 in-host SNPs, Materials and methods, <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>, <xref ref-type="supplementary-material" rid="supp18">Supplementary file 18</xref>). Conservatively, a mutation was characterized as phylogenetically convergent if it was present in isolates from three or more global lineages but not fixed in any lineage (Materials and methods). We identified 26/141 in-host SNPs as phylogenetically convergent in our global sample of isolates (<xref ref-type="supplementary-material" rid="supp19">Supplementary file 19</xref>). <xref ref-type="fig" rid="fig7">Figure 7</xref> and <xref ref-type="fig" rid="fig7s1">Figure 7—figure supplements 1</xref>–<xref ref-type="fig" rid="fig7s2">2</xref> display the distribution of convergent alleles across the 20,353 isolates using t-Distributed Stochastic Neighbor Embedding (t-SNE) of the pairwise genetic distance matrix (Materials and methods). The convergent alleles included the PPE genes <italic>PPE18</italic> (1 site), <italic>PPE54</italic> (1 site) and <italic>PPE60</italic> (3 sites), as well as Rv0095c (2 sites) and Rv1944c (1 site) both conserved proteins of unknown function. In addition to several SNPs in loci associated with antibiotic resistance, <italic>gyrB</italic> (1 site), <italic>gyrA</italic> (2 sites), <italic>rpoB</italic> (4 sites), <italic>rpoC</italic> (3 sites), <italic>inhA</italic> (1 site), <italic>embB</italic> (3 sites), and <italic>gid</italic> (1 site).</p><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Mutations acquired in-host are phylogenetically convergent.</title><p>We constructed t-SNE plots from a pairwise SNP distance matrix for our global sample of 20,352 clinical isolates and 128,898 SNP sites (Materials and methods). (<bold>A</bold>) Labeling isolates by global lineage revealed that isolates cluster according to genetic similarity. Next, we labeled isolates by whether they carried a mutant allele that was also detected in-host. (<bold>B–F</bold>) Mutations in <italic>gyrA</italic>, <italic>rpoB</italic>, <italic>PPE18</italic>, <italic>PPE54</italic>, and <italic>PPE60</italic> were detected in-host (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>), occur in a global collection of isolates (<xref ref-type="supplementary-material" rid="supp18">Supplementary file 18</xref>) and are scattered across the tSNE plots, indicating that they belong to genetically different clusters of isolates (<xref ref-type="supplementary-material" rid="supp19">Supplementary file 19</xref>). Furthermore, all mutations with a signal of phylogenetic convergence were detected in isolates belonging to different clusters, confirming that theses mutations must have arisen independently in different genetic backgrounds (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplements 1</xref>–<xref ref-type="fig" rid="fig7s2">2</xref>). Each plot is labeled with the gene name each mutation occurs within, amino acid encoded by the reference allele, H37Rv codon position, and amino acid encoded by the mutant allele. N = number of isolates with mutant allele.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig7-v2.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>Mutations acquired in-host are phylogenetically convergent.</title><p>We constructed t-SNE plots from a pairwise SNP distance matrix for our global sample of 20,352 clinical isolates and 128,898 SNP sites (Materials and methods). Isolates are colored by global lineage in the first plot, the rest are colored by whether isolates had specific mutations in that were detected in-host (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). These mutations occur in a global collection of isolates (<xref ref-type="supplementary-material" rid="supp18">Supplementary file 18</xref>) and are scattered across the tSNE plots, indicating that they belong to genetically different clusters of isolates (<xref ref-type="supplementary-material" rid="supp19">Supplementary file 19</xref>) and have arisen independently in different genetic backgrounds. Each plot is labeled with the gene name each mutation occurs within, amino acid encoded by the reference allele, H37Rv codon position, and amino acid encoded by the mutant allele (for intergenic mutations - reference allele, H37Rv genome coordinate, and mutant allele). N = number of isolates with mutant allele.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig7-figsupp1-v2.tif"/></fig><fig id="fig7s2" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 2.</label><caption><title>Mutations acquired in-host are phylogenetically convergent.</title><p>We constructed t-SNE plots from a pairwise SNP distance matrix for our global sample of 20,352 clinical isolates and 128,898 SNP sites (Materials and methods). Isolates are colored by global lineage in the first plot, the rest are colored by whether isolates had specific mutations in that were detected in-host (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). These mutations occur in a global collection of isolates (<xref ref-type="supplementary-material" rid="supp18">Supplementary file 18</xref>) and are scattered across the tSNE plots, indicating that they belong to genetically different clusters of isolates (<xref ref-type="supplementary-material" rid="supp19">Supplementary file 19</xref>) and have arisen independently in different genetic backgrounds. Each plot is labeled with the gene name each mutation occurs within, amino acid encoded by the reference allele, H37Rv codon position, and amino acid encoded by the mutant allele (for intergenic mutations - reference allele, H37Rv genome coordinate, and mutant allele). N = number of isolates with mutant allele.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-fig7-figsupp2-v2.tif"/></fig></fig-group></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>In our Mtbc populations sequenced from active TB patients enriched for negative treatment outcomes, we find a wealth of dynamics in genetic loci associated with antibiotic resistance, including a high turnover of minor variants. Known factors that determine treatment outcome are complex and include severity of lung disease, cavitation and adherence to treatment among others (<xref ref-type="bibr" rid="bib40">Imperial et al., 2018</xref>). Additionally, resistance acquisition in the course of one infection is comparatively rare in most pathogenic bacteria (<xref ref-type="bibr" rid="bib51">Llewelyn et al., 2017</xref>). Here, we observe that 9% of patients with confirmed delayed culture conversion, failure and relapse amplify resistance over time. Our findings of a higher rate of resistance acquisition in patients with MDR at the outset and with time between sampling, emphasize the importance of appropriately tailoring treatment regimens as well as close surveillance for microbiological clearance and resistance acquisition by phenotypic or genotypic means. The observed high rate of resistance acquisition also emphasizes Mtbc’s biological adaptability and the long duration of drug pressure in vivo. In addition to clonal acquisition of resistance, we find that sequencing revealed a substantial proportion of mixed infection or reinfection (28% of samples collected ≥2 months apart). This high percentage suggests that patient treatment and control of disease transmission can be better guided if pathogen sequencing is routinely performed for cases with persistent positive cultures especially in high TB prevalence settings where reinfection is more likely. Reinfection can also introduce strains with a different antibiotic susceptibility profile requiring adjustment in the treatment regimen.</p><p>While prior studies have investigated the lowest resistance allele frequencies that can be detected in clinical sputum samples (<xref ref-type="bibr" rid="bib22">Dreyer et al., 2020</xref>; <xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>), there is little information on the clinical relevance of these low frequency variants. We provide a proof-of-concept analysis that minor AR alleles, occurring at a frequency 19%, can predict fixation of the variant with a specificity &gt;95% of mutations in-host, although we find the sensitivity of this threshold to be low. The low sensitivity is because the majority of alleles that sweep to fixation are actually not detectable at all at the first time point, suggesting that more frequent sampling may be needed. In the future, higher depth and more frequent sequencing can elucidate more clearly the role of minor AR allele detection in clinical management of TB treatment.</p><p>Various sources of noise contribute to allele frequency changes over time and challenge inference on bacterial composition in vivo. Here, we determined an appropriate threshold for identifying mutations in-host using average depth Mtbc WGS from cultured isolates and demonstrate the importance of including technical replicate WGS. While culturing sputa in vitro enriches Mtbc DNA for WGS it also creates experimental noise (<xref ref-type="bibr" rid="bib81">Vargas and Farhat, 2020</xref>) and can purge some of the genetic diversity present in the sputum sample (<xref ref-type="bibr" rid="bib57">Nimmo et al., 2019</xref>). The refinement of methods for DNA extraction directly from sputum (<xref ref-type="bibr" rid="bib84">Votintseva et al., 2017</xref>), may allow the calling of relevant changes in allele frequencies at lower thresholds in future work. This would permit the unbiased study of loci that may be under frequency-dependent selection, where changes in allele frequencies would unlikely change by as much as 70% as we used here.</p><p>We detected 174 alleles rising to near fixation in-host across our sample of 200 patients. The observed distribution of variants including the high rate of non-synonymous substitutions and the predominance of GC &gt; AT variants are consistent with the hypotheses of purifying pressure on synonymous variants and oxidative DNA damage, respectively (<xref ref-type="bibr" rid="bib29">Ford et al., 2011</xref>; <xref ref-type="bibr" rid="bib56">Namouchi et al., 2012</xref>) in Mtbc. This consistency adds validity to our variant calling approach. Overall, the observed diversity spared the CD4<sup>+</sup> and CD8<sup>+</sup> T cell epitope encoding regions of the genome providing further evidence that host adaptive immunity does not drive directional selection in Mtbc genomes now at short-time scales (<xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>; <xref ref-type="bibr" rid="bib16">Copin et al., 2014</xref>; <xref ref-type="bibr" rid="bib18">Coscolla et al., 2015</xref>). Diversity was concentrated in antibiotic resistance regions and strikingly also in PE/PPE genes (<xref ref-type="fig" rid="fig6">Figure 6D</xref>; <xref ref-type="bibr" rid="bib63">Phelan et al., 2016</xref>). Although previous studies have generally avoided reporting short-read variant calls in PE/PPE regions, we demonstrate using read simulation, visualization of Illumina read alignments and comparison with long-read sequencing data the accuracy of the SNPs captured in our study. We found PPE genes to be more diverse in-host than PE genes and detected a signal of positive selection acting on three genes belonging to the PPE sub-family (<xref ref-type="fig" rid="fig7">Figure 7</xref>). This indicates that PPE genes may be play an important role in the process of host-adaptation.</p><p>In addition to identifying in-host variation in 12 loci known to be involved in the acquisition of antibiotic resistance, we identified six genes and six pathways displaying diversity in-host and not known to be associated with antibiotic resistance (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>, <xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>–<xref ref-type="supplementary-material" rid="supp14">14</xref>, <xref ref-type="supplementary-material" rid="supp17">Supplementary file 17</xref>). For a subset, we demonstrate similar diversity has arisen independently in separate hosts and in strains with different genetic backgrounds suggesting positive selection (<xref ref-type="fig" rid="fig7">Figure 7</xref>). Evidence of directional selection in Mtbc genomes have thus far been largely restricted to adaptation to antibiotic treatment (<xref ref-type="bibr" rid="bib5">Brites and Gagneux, 2015</xref>; <xref ref-type="bibr" rid="bib19">Didelot et al., 2016</xref>; <xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>). The novel pathways showing in-host convergence may be important for interactions between host and pathogen arising from either metabolic or immune pressure. Mtbc is one of a few types of bacteria that possess the capacity for de novo coenzyme B12/cobalamin synthesis, and this pathway has been implicated in Mtbc survival in-host and Mtbc growth (<xref ref-type="bibr" rid="bib66">Rowley and Kendall, 2019</xref>). We identified four genetic variants that developed in three separate patients and in three consecutive genes from the same locus <italic>cobG,</italic> intergenic <italic>cobG-cobH, cobH</italic> and <italic>cobK</italic> (Rv2064-Rv2067). This observation contributes to mounting evidence on the importance of this pathway for in vivo Mtbc survival and may have implications for drug development (<xref ref-type="bibr" rid="bib33">Gopinath et al., 2013</xref>; <xref ref-type="bibr" rid="bib54">Minias et al., 2018</xref>). Biotin biosynthesis is also relatively unique to mycobacteria and plays an important role in Mtbc growth, infection and host survival during latency (<xref ref-type="bibr" rid="bib67">Salaemae et al., 2011</xref>). The other identified pathways include ESAT-6 protein secretion known to play a role in the modulation of host immune response by disrupting the phagosomal membrane (<xref ref-type="bibr" rid="bib11">Clemmensen et al., 2017</xref>).</p><p>The loci found to be phylogenetically convergent and not known to be associated with antibiotic resistance, include the genes Rv0095c, <italic>PPE18</italic>, <italic>PPE54</italic>, and <italic>PPE60</italic>. Consistent with the idea that positive selection is acting on alleles within these loci, we observe a reduction in diversity at the second time point for the patients in which drug-resistant alleles sweep to fixation and in which putative host-pathogen alleles sweep to fixation (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Although of unknown function, Rv0095c (SNP A85V) was recently associated with transmission success of an Mtbc cluster in Peru (<xref ref-type="bibr" rid="bib21">Dixit et al., 2019</xref>). Both <italic>PPE18</italic> and <italic>PPE60</italic> have been shown to interact with toll-like receptor 2 (TLR2) (<xref ref-type="bibr" rid="bib55">Nair et al., 2009</xref>; <xref ref-type="bibr" rid="bib73">Su et al., 2018</xref>). <italic>PPE18</italic> was the only gene to encode an epitope containing a SNP in-host; mutations in the epitope-encoding regions of this gene have previously been described in a set of geographically separated clinical isolates (<xref ref-type="bibr" rid="bib35">Hebert et al., 2007</xref>). Furthermore, <italic>PPE18</italic> codes for one of the antigens used in the construction of the M72/AS01E vaccine candidate (<xref ref-type="bibr" rid="bib75">Tait et al., 2019</xref>). Our results demonstrating that <italic>PPE18</italic> is under positive selection in the MTBC may have implications for the efficacy of this vaccine against genetically diverse Mtbc strains. <italic>PPE54</italic> has been implicated in Mtbc’s ability to arrest macrophage phagosomal maturation (phagosome-lysosome fusion) and thought to be vital for intracellular persistence (<xref ref-type="bibr" rid="bib6">Brodin et al., 2010</xref>). The mechanism by which <italic>PPE54</italic> accomplishes this is unknown, but Mtbc modification of phagosomal function is thought to be TLR2/TLR4-dependent (<xref ref-type="bibr" rid="bib64">Podinovskaia et al., 2013</xref>).</p><p>Mtbc is known to disrupt numerous <italic>innate</italic> immune mechanisms including phagosome maturation, apoptosis, autophagy as well as inhibition of MHC II expression through prolonged engagement with innate sensor toll-like receptor 2 (TLR2) among others (<xref ref-type="bibr" rid="bib24">Ernst, 2018</xref>). SNPs in human genes involved with innate-immune pathways have been implicated in-host susceptibility to TB (<xref ref-type="bibr" rid="bib1">Azad et al., 2012</xref>; <xref ref-type="bibr" rid="bib41">Kleinnijenhuis et al., 2011</xref>; <xref ref-type="bibr" rid="bib77">Tientcheu et al., 2017</xref>). Specifically, SNPs in TLR2 (thought to be the most important TLR in Mtbc recognition) (<xref ref-type="bibr" rid="bib77">Tientcheu et al., 2017</xref>) and TLR4 have been associated with susceptibility to TB disease (<xref ref-type="bibr" rid="bib1">Azad et al., 2012</xref>; <xref ref-type="bibr" rid="bib41">Kleinnijenhuis et al., 2011</xref>). Taken together, these observations and our results are consistent with ongoing co-evolution between humans and Mtbc with evidence for reciprocal adaptive changes, leaving a signature of selection in both humans and Mtbc populations (<xref ref-type="bibr" rid="bib5">Brites and Gagneux, 2015</xref>). Most co-evolution between Mtbc and humans, the main reciprocal adaptations between host and pathogen are thought to have occurred long ago and as a result of long-term host-pathogen interactions (<xref ref-type="bibr" rid="bib1">Azad et al., 2012</xref>; <xref ref-type="bibr" rid="bib5">Brites and Gagneux, 2015</xref>). Here, we observe these dynamics over the short evolutionary timescale of a single infection which has important implications for vaccine development (<xref ref-type="bibr" rid="bib3">Brennan, 2017</xref>).</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Sequence data</title><sec id="s4-1-1"><title>Longitudinal isolate pairs</title><p>This study included data for 614 clinical isolates of <italic>M. tuberculosis</italic> that were sampled from the sputum of 307 patients resulting in n = 307 longitudinal pairs. The sequencing data for 456 publicly available isolates was downloaded from Genbank (RRID:<ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_002760">SCR_002760; </ext-link><xref ref-type="bibr" rid="bib2">Benson et al., 2009</xref>), sequenced using Illumina chemistry to generate paired-end reads and came from previously published studies (T [<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>], C [<xref ref-type="bibr" rid="bib8">Casali et al., 2016</xref>], W [<xref ref-type="bibr" rid="bib85">Walker et al., 2013</xref>], B [<xref ref-type="bibr" rid="bib7">Bryant et al., 2013</xref>], G [<xref ref-type="bibr" rid="bib34">Guerra-Assunção et al., 2015</xref>], X [<xref ref-type="bibr" rid="bib91">Xu et al., 2018</xref>], H [<xref ref-type="bibr" rid="bib87">Witney et al., 2017</xref>], P [<xref ref-type="bibr" rid="bib28">Farhat et al., 2019</xref>]; <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>–<xref ref-type="supplementary-material" rid="supp2">2</xref>). We aggregated treatment from the source studies and added this metadata to <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref> for each longitudinal isolate. We include columns that indicate the timing of sampling of Mtbc relative to treatment, the treatment regimen administered and final patient outcome (and relevant details). Patient outcomes are defined as follows: <italic>Delayed culture conversion</italic> (sputum culture positive at baseline and 2 months treatment initiation with genomic analysis consistent with clonal infection), <italic>Failure or Relapse</italic> (sputum culture positive at baseline and 4.5 months treatment initiation with genomic analysis consistent with clonal infection), <italic>Failure or Relapse or Default</italic> (sputum culture positive at interval of 4.5 months with genomic analysis consistent with clonal infection, only partial treatment data is available) or N/A if date data was of low resolution, not available or no treatment data was available. We also determined <italic>Reinfection</italic> and <italic>Mixed infection</italic> based on the genomic analysis.</p></sec><sec id="s4-1-2"><title>Replicate isolate pairs</title><p>This study included three types of replicate isolate pairs. (S2 - Sequenced Twice) DNA pooled from a single Mtbc clinical isolate that had undergone in vitro expansion was sequenced in separate runs on an Illumina sequencing machine (m = 5). (C2 – Cultured and Sequenced Twice) Mtbc was cultured from a single frozen clinical sample at separate time points, then sequenced on an Illumina sequencing machine after DNA extraction from culture (m = 73). (P3) Three sputum samples were obtained from a single patient within a 24-hr period (<xref ref-type="bibr" rid="bib78">Trauner et al., 2017</xref>), cultured separately, underwent DNA extraction and then sequencing on an Illumina sequencing machine. For the purposes of this study, we compared these three isolates pairwise (m = 3).</p></sec><sec id="s4-1-3"><title>Global sequence data</title><p>We downloaded raw sequence data for 33,873 clinical isolates from the public domain (<xref ref-type="bibr" rid="bib2">Benson et al., 2009</xref>). Isolates had to meet the following quality control measures for inclusion in our study: (i) at least 90% of the reads had to be taxonomically classified as belonging to the <italic>M. tuberculosis</italic> complex after running the trimmed FASTQ files through Kraken (<xref ref-type="bibr" rid="bib88">Wood and Salzberg, 2014</xref>) and (ii) at least 95% of bases had to have coverage of at least 10x after mapping the processed reads to the H37Rv Reference Genome.</p></sec></sec><sec id="s4-2"><title>Epitope collection and analysis</title><p>CD4<sup>+</sup> T and CD8<sup>+</sup> T cell epitope sequences were downloaded from the Immune Epitope Database (RRID: <ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_006604">SCR_006604</ext-link>) (<xref ref-type="bibr" rid="bib83">Vita et al., 2015</xref>) on May 23rd, 2018 according to criteria described previously (<xref ref-type="bibr" rid="bib18">Coscolla et al., 2015</xref>) [linear peptides, <italic>M. tuberculosis</italic> complex (ID:77643, Mycobacterium complex), positive assays only, T cell assays, any MHC restriction, host: humans, any diseases, any reference type] yielding a set of 2031 epitope sequences (<xref ref-type="supplementary-material" rid="supp11">Supplementary file 11</xref>). We mapped each epitope sequence to the genes encoded by the H37Rv Reference Genome (<xref ref-type="bibr" rid="bib13">Cole et al., 1998</xref>) using BlastP with an e-value cutoff of 0.01 (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1A</xref>). We retained only epitope sequences that mapped to at least one region in H37Rv (due to sequence homology, some epitopes mapped to multiple regions) and whose BlastP peptide start/end coordinates matched those specified in IEDB (n = 1949,949 representing 1505 separate epitope entries in IEDB). We then filtered out any epitopes occurring in Mobile Genetic Elements which resulted in a final set of 1875 epitope sequences, representing 348 genes (antigens) used for downstream analysis. The distribution of peptide lengths for this final set of epitopes is given in <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1B</xref>. Since many of these epitope sequences overlap, we constructed non-redundant epitope concatenate sequences for each antigen (n = 348) gene (<xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>; <xref ref-type="bibr" rid="bib18">Coscolla et al., 2015</xref>; <xref ref-type="bibr" rid="bib72">Stucki et al., 2016</xref>). The regions of each antigen not encoding an epitope were concatenated into a non-epitope sequence for that gene.</p></sec><sec id="s4-3"><title>Gene sets</title><p>Every gene on H37Rv was classified into one of six non-redundant gene categories according to the following criteria: (i) genes identified as belonging to the PE/PPE family of genes unique to pathogenic mycobacteria, though to influence immunopathogenicity and characterized by conserved proline-glutamate (PE) and proline-proline-glutamate (PPE) motifs at the N protein termini (<xref ref-type="bibr" rid="bib4">Brennan and Delogu, 2002</xref>; <xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>; <xref ref-type="bibr" rid="bib63">Phelan et al., 2016</xref>) were classified as <italic>PE/PPE</italic> (n = 167), (ii) genes flagged as being associated with antibiotic resistance (<xref ref-type="bibr" rid="bib25">Farhat et al., 2013</xref>) were classified into the <italic>Antibiotic Resistance</italic> category (n = 28), (iii) genes encoding a CD4<sup>+</sup> or CD8<sup>+</sup> T-cell epitope (<xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>; <xref ref-type="bibr" rid="bib18">Coscolla et al., 2015</xref>; but not already classified as a PE/PPE or Antibiotic Resistance gene) were classified as an <italic>Antigen</italic> (n = 257), (iv) genes required for growth in vitro (<xref ref-type="bibr" rid="bib68">Sassetti et al., 2003</xref>) and in vivo (<xref ref-type="bibr" rid="bib69">Sassetti and Rubin, 2003</xref>) and not already placed into a category above were classified as <italic>Essential</italic> genes (n = 682), (v) genes flagged as transposases, integrases, phages, or insertion sequences were classified as <italic>Mobile Genetic Elements</italic> (<xref ref-type="bibr" rid="bib15">Comas et al., 2010</xref>) (n = 108), (vi) any remaining genes not already classified above were placed into the <italic>Non-Essential</italic> category (n = 2752) (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>).</p></sec><sec id="s4-4"><title>Illumina sequencing FastQ processing and mapping to H37Rv</title><p>The raw sequence reads from all sequenced isolates were trimmed with Prinseq (<xref ref-type="bibr" rid="bib70">Schmieder and Edwards, 2011</xref>) (settings: <monospace>-min_qual_mean 20</monospace>) (version 0.20.4) then aligned to the H37Rv Reference Genome (Genbank accession: NC_000962) with the BWA mem (<xref ref-type="bibr" rid="bib46">Li and Durbin, 2009</xref>) algorithm (settings: -M) (version 0.7.15). The resulting SAM files were then sorted (settings: <monospace>SORT_ORDER = coordinate</monospace>), converted to BAM format and processed for duplicate removal with Picard (<ext-link ext-link-type="uri" xlink:href="http://broadinstitute.github.io/picard/">http://broadinstitute.github.io/picard/</ext-link>) (version 2.8.0) (settings: <monospace>REMOVE_DUPLICATES = true, ASSUME_SORT_ORDER = coordinate</monospace>). The processed BAM files were then indexed with Samtools (<xref ref-type="bibr" rid="bib44">Li et al., 2009</xref>). We used Pilon (<xref ref-type="bibr" rid="bib86">Walker et al., 2014</xref>) on the resulting BAM files to call bases for all reference positions corresponding to H37Rv from pileup (settings: <monospace>--variant</monospace>).</p></sec><sec id="s4-5"><title>Empirical score for difficult-to-call regions</title><p>We extracted DNA from 15 Mtbc isolates (<xref ref-type="bibr" rid="bib23">Epperson and Strong, 2020</xref>), for which we had Illumina sequencing reads, to undergo PacBio sequencing (Appendix 2). Together, with public PacBio and Illumina sequencing data (<xref ref-type="bibr" rid="bib10">Chiner-Oms et al., 2019</xref>), we compiled 31 pairs of sequencing reads for comparison of variant calling between PacBio long-reads and Illumina short-reads (<xref ref-type="supplementary-material" rid="supp20">Supplementary file 20</xref>). Using the 31 isolates for which both Illumina and a complete PacBio assembly were available (Appendix 2), we evaluated the empirical base-pair recall (EBR) of all base-pair positions of the H37rv reference genome (Marin et al., in preparation). For each sample, the alignments (from minimap2) of each high confidence genome assembly to the H37Rv genome were used to infer the true nucleotide identity of each base pair position. To calculate the empirical base-pair recall, we calculated what % of the time our Illumina based variant calling pipeline, across 31 samples, confidently called the true nucleotide identity at a given genomic position. If Pilon variant calls did not produce a confident base call (<italic>Pass</italic>) for the position, it did not count as a correct base call. This yields a metric ranging from 0.0 to 1.0 for the consistency by which each base-pair is both confidently and correctly sequenced by our Illumina WGS-based variant calling pipeline for each position on the H37Rv reference genome. An H37Rv position with an EBR score of x% indicates that the base calls made from Illumina sequencing and mapping to H37Rv agreed with the base calls made from the PacBio de novo assemblies (Appendix 2) in x% of the Illumina-PacBio pairs. We masked difficult-to-call regions by dropping H37Rv positions with an EBR score below 0.8 as part of our variant calling procedure.</p></sec><sec id="s4-6"><title>Variant calling</title><sec id="s4-6-1"><title>Single-nucleotide polymorphism (SNP) calling</title><p>To prune out low-quality base calls that may have arisen due to sequencing or mapping error, we dropped any base calls that did not meet any of the following criteria (<xref ref-type="bibr" rid="bib17">Copin et al., 2016</xref>): (i) the call was flagged as either <italic>Pass</italic> or <italic>Ambiguous</italic> by Pilon, (ii) the reads aligning to that position supported at most two alleles (ensuring that 1 allele matched the reference allele if there were 2), (iii) the mean base quality at the locus was &gt; 20, (iv) the mean mapping quality at the locus was &gt; 30, (v) none of the reads aligning to the locus supported an insertion or deletion, (vi) a minimum coverage of 25 reads at the position, (vii) the EBR score for the position ≥0.80, and (viii) the position is not located in a mobile genetic element region of the reference genome. We then used the Pilon-generated (<xref ref-type="bibr" rid="bib86">Walker et al., 2014</xref>) VCF files to calculate the frequencies for both the reference and alternate alleles, using the <italic>INFO.QP</italic> field (which gives the proportion of reads supporting each base weighted by the base and mapping quality of the reads, <italic>BQ</italic> and <italic>MQ</italic> respectively, at the specific position) to determine the proportion of reads supporting each base for each locus of interest.</p></sec><sec id="s4-6-2"><title>Additional SNP filtering for isolate pairs</title><p>To call SNPs (and corresponding changes in allele frequencies) between pairs of isolates (Replicate and Longitudinal pairs), we required: (i) <italic>SNP Calling</italic> filters be met, (ii) the number of reads aligning to the position is below the 99th percentile for all of the calls made for that isolate, (iii) the call at that position passes all filters for each isolate in the pair, and (iv) SNPs in <italic>glpK</italic> were dropped as mutants arising in this gene are thought to be an artifact of in vitro expansion (<xref ref-type="bibr" rid="bib62">Pethe et al., 2010</xref>; <xref ref-type="bibr" rid="bib81">Vargas and Farhat, 2020</xref>); we detected four non-synonymous SNPs in <italic>glpK</italic> (ΔAF ≥ 25%) between three longitudinal pairs (mean ΔAF = 64%) and five non-synonymous SNPs in <italic>glpK</italic> (ΔAF ≥ 25%) among five replicate pairs (mean ΔAF = 45%).</p></sec><sec id="s4-6-3"><title>Additional SNP filtering for antibiotic resistance loci analysis</title><p>To call SNPs (and corresponding minor changes in allele frequencies) between pairs of isolates (Longitudinal Pairs), we required: (i) <italic>SNP Calling</italic> filters be met, (ii) <italic>Additional SNP Filtering for Isolate Pairs</italic> filters be met, (iii) <inline-formula><mml:math id="inf26"><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo> <mml:mi/><mml:msubsup><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mfenced><mml:mo>=</mml:mo> <mml:mi/><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">F</mml:mi> <mml:mi mathvariant="normal"/><mml:mo>≥</mml:mo><mml:mn>5</mml:mn><mml:mi>%</mml:mi></mml:math></inline-formula>, (iv) if <inline-formula><mml:math id="inf27"><mml:mn>5</mml:mn><mml:mi>%</mml:mi><mml:mo>≤</mml:mo><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">F</mml:mi><mml:mo>&lt;</mml:mo><mml:mn>20</mml:mn><mml:mi mathvariant="normal">%</mml:mi></mml:math></inline-formula>, then the SNP was only retained if each allele (across both isolates) with <inline-formula><mml:math id="inf28"><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">F</mml:mi> <mml:mi mathvariant="normal"/><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn><mml:mi>%</mml:mi></mml:math></inline-formula> was supported by at least five reads (ensuring that at least five reads supported each minor allele at lower values of <inline-formula><mml:math id="inf29"><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">F</mml:mi></mml:math></inline-formula>), (v) the SNP was classified as either intergenic or non-synonymous, (vi) the SNP was located in a gene, intergenic region or rRNA coding region associated with antibiotic resistance (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>).</p></sec><sec id="s4-6-4"><title>Additional SNP filtering for heterogenous SNPs</title><p>To call heterogenous SNPs in each isolate for a pair of longitudinal isolates, we required: (i) <italic>SNP Calling</italic> filters be met, (ii) the number of reads aligning to the position is below the 99th percentile for all of the calls made for that isolate, (iii) the alternate allele frequency for the SNP is such that <inline-formula><mml:math id="inf30"><mml:msup><mml:mrow><mml:mn>5</mml:mn><mml:mi>%</mml:mi><mml:mo>≤</mml:mo><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msup><mml:mo>≤</mml:mo><mml:mn>75</mml:mn><mml:mi>%</mml:mi></mml:math></inline-formula>, (iv) the position is not located in a mobile genetic element region of the reference genome, (v) the position is not located in a PE/PPE region of the reference genome.</p></sec><sec id="s4-6-5"><title>Additional SNP filtering for global isolates</title><p>To call alleles in our global and genetically diverse set of isolates, we required: (i) <italic>SNP Calling</italic> filters be met, with the modifications that (ii) the call was flagged as <italic>Pass</italic> by Pilon and (iii) 1 allele (either the reference or an alternate) was supported by at least 90% of the reads regardless of whether the other ≤10% of reads supported 1, 2, or 3 alleles.</p></sec></sec><sec id="s4-7"><title>Mixed lineage and contamination detection for longitudinal and replicate isolate pairs</title><sec id="s4-7-1"><title>Kraken</title><p>To filter out samples that may have been contaminated by foreign DNA during sample preparation, we ran the trimmed reads for each longitudinal and replicate isolate through Kraken2 (<xref ref-type="bibr" rid="bib88">Wood and Salzberg, 2014</xref>) against a database (<xref ref-type="bibr" rid="bib32">Goig et al., 2020</xref>) containing all of the sequences of bacteria, archaea, virus, protozoa, plasmids, and fungi in RefSeq (release 90) and the human genome (GRCh38). We calculated the proportion reads that were taxonomically classified under the <italic>Mycobacterium tuberculosis</italic> Complex (MTBC) for each isolate and implemented a threshold of 95%. An isolate pair was dropped if either isolate had less than 95% of reads aligning to MTBC.</p></sec><sec id="s4-7-2"><title>F2</title><p>To further reduce the effects of contamination, we aimed to identify samples that may have been patient to inter-lineage mixture samples resulting from of a co-infection (F2). We computed the F2 lineage-mixture metric for each longitudinal and replicate isolate (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). We wrote a custom script to carry out the same protocol for computing F2 as previously described (<xref ref-type="bibr" rid="bib90">Wyllie et al., 2018</xref>). Briefly, the method involves calculating the minor allele frequencies at lineage-defining SNPs (<xref ref-type="bibr" rid="bib14">Coll et al., 2014</xref>). From 64 sets of SNPs that define the deep branches of the MTBC (<xref ref-type="bibr" rid="bib14">Coll et al., 2014</xref>), we considered the 57 sets that contain more than 20 SNPs to obtain better estimates of minor variation (<xref ref-type="bibr" rid="bib14">Coll et al., 2014</xref>; <xref ref-type="bibr" rid="bib90">Wyllie et al., 2018</xref>). For each SNP set <inline-formula><mml:math id="inf31"><mml:mi>i</mml:mi></mml:math></inline-formula>, (i) we summed the total depth and (ii) the number of reads supporting the most abundant base (at each position) over all of the reference positions (SNPs) that met our mapping quality, base quality and insertion/deletion filters, which yields <inline-formula><mml:math id="inf32"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf33"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> respectively. Subtracting these two quantities yields the minor depth for SNP set <inline-formula><mml:math id="inf34"><mml:mi>i</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="inf35"><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The minor allele frequency estimate for SNP set <inline-formula><mml:math id="inf36"><mml:mi>i</mml:mi></mml:math></inline-formula> is then defined as <inline-formula><mml:math id="inf37"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>∕</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. Doing this for all 57 SNP sets gives <inline-formula><mml:math id="inf38"><mml:mfenced close="}" open="{" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo> <mml:mi/><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo> <mml:mi/><mml:mo>⋯</mml:mo><mml:mo>,</mml:mo> <mml:mi/><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>57</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:math></inline-formula>. We then sorted <inline-formula><mml:math id="inf39"><mml:mfenced close="}" open="{" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo> <mml:mi/><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo> <mml:mi/><mml:mo>⋯</mml:mo><mml:mo>,</mml:mo> <mml:mi/><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>57</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:math></inline-formula> in descending order and estimated the minor variant frequency for all of the reference positions (SNPs) corresponding to the top 2 sets (highest <inline-formula><mml:math id="inf40"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> values) which yields the F2 metric. Letting <inline-formula><mml:math id="inf41"><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:math></inline-formula> be the number of SNPs in the top two sets, then <inline-formula><mml:math id="inf42"><mml:mi>F</mml:mi><mml:mn>2</mml:mn><mml:mo>=</mml:mo> <mml:mi/><mml:mrow><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow> <mml:mi/></mml:mrow><mml:mo>/</mml:mo><mml:mrow> <mml:mi/><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. Isolate pairs were dropped if the F2 metric for either isolate passed the F2 threshold set for mixed lineage detection (<xref ref-type="fig" rid="fig1">Figure 1B</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>).</p></sec></sec><sec id="s4-8"><title>Pre-existing genotypic resistance</title><p>We determined pre-existing resistance for a patient (with a pair of longitudinal isolates) by scanning the first isolate for the detection of at least 1 of 177 SNPs predictive of resistance with AF ≥75% (from a minimal set of 238 variants [<xref ref-type="bibr" rid="bib27">Farhat et al., 2016</xref>]). Drug resistance was inferred from the whole genome sequencing data using a well validated set of 177 mutations at an allele frequency threshold (&gt;40%). Selection of these mutations and validation of this allele frequency threshold was previously described. This study made use of 1319 clinical Mtbc isolates with known drug resistance phenotypes. The data were randomly split into training and validation sets containing 67% and 33% of the isolates (respectively). The diagnostic set of mutations was determined using random forest predictive modeling in which a weighted model was run with serially smaller subsets of mutations to identify a minimal set of mutations to predict resistance to first- and second-line TB drugs. The resulting set of mutations predicted INH resistance with a sensitivity of 94% and specificity of 94% on the validation isolate set and predicted RIF resistance with a sensitivity of 93% and specificity of 95% on the validation isolate set. We excluded predictive indels and the <italic>gid</italic> E92D variant as the latter is likely a lineage marking variant that is not indicative of antibiotic resistance. We defined pre-existing multidrug resistance for a patient by scanning the first isolate collected for detection of at least 1 SNP predictive of Rifampicin resistance (14/178 predictive SNPs) and at least 1 SNP predictive of Isoniazid resistance (18/178 predictive SNPs). The genotypic resistance predictions for 13 antibiotics for all 614 longitudinal isolates from the 307 patients in our study can be found in <xref ref-type="supplementary-material" rid="supp21">Supplementary file 21</xref>.</p></sec><sec id="s4-9"><title>True and false positive rate analysis for heteroresistant mutations</title><p>To determine the predictive value of low-frequency heteroresistant alleles, we classified SNPs as fixed if the alternate allele frequency in the second isolate collected from the patient was at least 75% (alt AF<sub>2 </sub>≥ 75%). We first dropped SNPs for which alt AF<sub>1 </sub>≥ 75% and alt AF<sub>2 </sub>≥ 75% (high frequency mutant alleles in both isolates). We then set a threshold <inline-formula><mml:math id="inf43"><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:math></inline-formula> for the alternate allele frequency detected in the first isolate collected from the patient (alt AF<sub>1</sub>) and predicted whether an alternate allele would rise to a substantial proportion of the sample (alt AF<sub>2</sub> ≥ 75%) as follows:<disp-formula id="equ1"><mml:math id="m1"><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi> <mml:mi/><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi> <mml:mi/></mml:mrow></mml:msub><mml:mo>⟶</mml:mo><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi> <mml:mi/><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>75</mml:mn><mml:mi mathvariant="normal">%</mml:mi></mml:math></disp-formula><disp-formula id="equ2"><mml:math id="m2"><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi> <mml:mi/><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi> <mml:mi/></mml:mrow></mml:msub><mml:mo>⟶</mml:mo><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi> <mml:mi/><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:mn>75</mml:mn><mml:mi mathvariant="normal">%</mml:mi></mml:math></disp-formula></p><p>We classified every SNP as True Positive (TP), False Positive (FP), True Negative (TN) or False Negative (FN) according to:<disp-formula id="equ3"><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>:</mml:mo><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mtext> </mml:mtext></mml:mrow></mml:msub><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">&amp;</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:mn>75</mml:mn><mml:mrow><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equ4"><mml:math id="m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>:</mml:mo><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mtext> </mml:mtext></mml:mrow></mml:msub><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">&amp;</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>75</mml:mn><mml:mrow><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equ5"><mml:math id="m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>:</mml:mo><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mtext> </mml:mtext></mml:mrow></mml:msub><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">&amp;</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>75</mml:mn><mml:mrow><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equ6"><mml:math id="m6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>:</mml:mo><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mtext> </mml:mtext></mml:mrow></mml:msub><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">&amp;</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>A</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:mn>75</mml:mn><mml:mrow><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>True Positive Rates (TPR) and False Positive Rates (FPR) were calculated as:<disp-formula id="equ7"><mml:math id="m7"><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo> <mml:mi/><mml:mfrac><mml:mrow><mml:mo>#</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mo>#</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mo>#</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac> <mml:mi/> <mml:mi/> <mml:mi/> <mml:mi/><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>#</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mo>#</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mo>#</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>Finally, we made predictions for all SNPs and calculated the TPR and FPR for all values of <inline-formula><mml:math id="inf44"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:mi/><mml:mo>∈</mml:mo><mml:mfenced close="}" open="{" separators="|"><mml:mrow><mml:mn>0</mml:mn><mml:mi>%</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mi>%</mml:mi><mml:mo>,</mml:mo> <mml:mi/><mml:mn>2</mml:mn><mml:mi>%</mml:mi><mml:mo>,</mml:mo><mml:mo>⋯</mml:mo><mml:mo>,</mml:mo><mml:mn>98</mml:mn><mml:mi>%</mml:mi><mml:mo>,</mml:mo><mml:mn>99</mml:mn><mml:mi>%</mml:mi><mml:mo>,</mml:mo><mml:mn>100</mml:mn><mml:mi>%</mml:mi></mml:mrow></mml:mfenced></mml:math></inline-formula>.</p></sec><sec id="s4-10"><title>Mutation density test</title><p>The method to detect significant variation for a given locus amongst pairs of sequenced isolates has been described previously (<xref ref-type="bibr" rid="bib26">Farhat et al., 2014</xref>). Briefly, let <inline-formula><mml:math id="inf45"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="script">𝒩</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mtext> </mml:mtext><mml:mo>∼</mml:mo><mml:mtext> </mml:mtext><mml:mi>P</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>λ</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> be a random variable for the number of SNPs detected across all isolate pairs (for the in-host analysis this is the collection of longitudinal isolate pairs for all patients) for gene <inline-formula><mml:math id="inf46"><mml:mi>j</mml:mi></mml:math></inline-formula>. Let (i) <inline-formula><mml:math id="inf47"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo></mml:math></inline-formula> number of SNPs across all pairs for gene <inline-formula><mml:math id="inf48"><mml:mi>i</mml:mi></mml:math></inline-formula>, (ii) <inline-formula><mml:math id="inf49"><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> length of gene <inline-formula><mml:math id="inf50"><mml:mi>i</mml:mi></mml:math></inline-formula>, (iii) <inline-formula><mml:math id="inf51"><mml:mi>P</mml:mi><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> number of genome pairs and (iv) <inline-formula><mml:math id="inf52"><mml:mi>G</mml:mi><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> the number of genes across the genome being analyzed (all genes in the essential, non-essential, antigen, antibiotic resistant and family protein categories).</p><p>Then the length of the genome (concatenate of all genes being analyzed) is given by <inline-formula><mml:math id="inf53"><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mrow></mml:math></inline-formula> and the number of SNPs across all genes and genome pairs is given by <inline-formula><mml:math id="inf54"><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>. The null rate for <inline-formula><mml:math id="inf55"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="script">𝒩</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is given by the mean SNP distance between all pairs of isolates, weighted by the length of gene <inline-formula><mml:math id="inf56"><mml:mi>j</mml:mi></mml:math></inline-formula> as a fraction of the genome concatenate and number of isolate pairs:<disp-formula id="equ8"><mml:math id="m8"><mml:msub><mml:mrow><mml:mi>λ</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo> <mml:mi/><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced></mml:math></disp-formula></p><p>The p-value for gene <inline-formula><mml:math id="inf57"><mml:mi>j</mml:mi></mml:math></inline-formula> is then calculated as <inline-formula><mml:math id="inf58"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">r</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&gt;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">𝒩</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>. We tested 3,386 genes for mutational density and applied Bonferroni correction to determine a significance threshold. We determine a gene to have a significant amount of variation if the assigned p-value <inline-formula><mml:math id="inf59"><mml:mo>&lt;</mml:mo><mml:mfrac><mml:mrow><mml:mn>0.05</mml:mn></mml:mrow><mml:mrow><mml:mn>3,386</mml:mn></mml:mrow></mml:mfrac><mml:mo>≈</mml:mo><mml:mn>1.477</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>.</p></sec><sec id="s4-11"><title>Nucleotide diversity</title><p>We define the nucleotide diversity <inline-formula><mml:math id="inf60"><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:math></inline-formula> for a given gene <inline-formula><mml:math id="inf61"><mml:mi>g</mml:mi></mml:math></inline-formula> as follows: (i) let <inline-formula><mml:math id="inf62"><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> base-pair length of the gene, (ii) <inline-formula><mml:math id="inf63"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> number of in-host SNPs (independent of the change in allele frequency for each SNP) between the longitudinal isolates for patient <inline-formula><mml:math id="inf64"><mml:mi>i</mml:mi></mml:math></inline-formula> occurring on gene <inline-formula><mml:math id="inf65"><mml:mi>j</mml:mi></mml:math></inline-formula> and (iii) <inline-formula><mml:math id="inf66"><mml:mi>P</mml:mi><mml:mo>=</mml:mo> <mml:mi/></mml:math></inline-formula> number of patients. Then<disp-formula id="equ9"><mml:math id="m9"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>Correspondingly, let <inline-formula><mml:math id="inf67"><mml:mi>G</mml:mi></mml:math></inline-formula> be a category consisting of <inline-formula><mml:math id="inf68"><mml:mi>M</mml:mi></mml:math></inline-formula> genes, then the average nucleotide diversity for <inline-formula><mml:math id="inf69"><mml:mi>G</mml:mi> <mml:mi/></mml:math></inline-formula> is given by:<disp-formula id="equ10"><mml:math id="m10"><mml:msub><mml:mrow><mml:mi>π</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo> <mml:mi/><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mfenced close="|" open="|" separators="|"><mml:mrow><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced></mml:mrow></mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mfenced></mml:math></disp-formula></p></sec><sec id="s4-12"><title>SNP calling simulations in repetitive genomic regions</title><p>Certain repetitive regions of the <italic>Mycobacterium tuberculosis</italic> genome (ESX, PE/PPE loci) may give rise to false positive and false negative variant calls due to the mis-alignment of short-read sequencing data. To test the rate of false negative and false positive SNP calls in genes with <italic>in-host</italic> SNPs (<xref ref-type="fig" rid="fig5">Figure 5</xref>, <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>) we collected the set of non-redundant SNPs observed in these loci (<xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>). Next, we collected a set of publicly available reference genomes (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>) and introduced these mutations into the respective loci positions in the reference genomes. We then simulated short-read Illumina sequencing data of comparable quality to our sequencing data from these altered reference genomes. Using our variant-calling pipeline to call polymorphisms, we then estimated the number of true and false positive SNP calls for each gene, based off of how many introduced SNPs were called (true positives), how many introduced SNPs were not called (false negatives) and how many spurious SNPs were called (false positives). A schematic of our simulation methodology is given in <xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1, a</xref> detailed explanation is given in Appendix 1 and the results of our simulations (given in <xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>) confirm a low false-positive rate.</p></sec><sec id="s4-13"><title>Global lineage typing</title><p>We determined the global lineage of each longitudinal (<inline-formula><mml:math id="inf70"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>614</mml:mn></mml:math></inline-formula>) and global isolate (<inline-formula><mml:math id="inf71"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>32,033</mml:mn></mml:math></inline-formula>) using base calls from Pilon-generated VCF files and a 62-SNP lineage-defining diagnostic barcode from a previously published study (<xref ref-type="bibr" rid="bib14">Coll et al., 2014</xref>).</p></sec><sec id="s4-14"><title>Phylogenetic convergence analysis and t-SNE visualization</title><sec id="s4-14-1"><title>Construction of genotypes matrix</title><p>We detected SNP sites at 878,244 H37Rv reference positions (of which 61,918 SNPs were not biallelic) among our global sample of 33,873 isolates. After excluding SNP sites with rare minor alleles (sites in which alternate alleles were called in &lt;5 isolates) we retained SNPs at 146,874 positions. We constructed a 146,874 × 33,873 genotypes matrix (coded as 0:A, 1:C, 2:G, 3:T, 9:Missing) and filled in the matrix for the allele supported at each SNP site for each isolate according to the <italic>SNP Calling</italic> filters outlined above. If a base call at a specific reference position for an isolate did not meet the filter criteria that allele was coded as <italic>Missing</italic>.</p><p>Excluding 2348 SNP sites that had an EBR score &lt;0.80, another 1509 SNP sites located within mobile genetic element regions, and 3220 SNP sites in with missing calls in &gt;25% of isolates yielded a genotypes matrix with dimensions 139,797 × 33,873. Next, we excluded 1518 isolates with missing calls in &gt;25% of SNP sites yielding a genotypes matrix with dimensions 139,797 × 32,355. We used a previously published 62-SNP barcode (<xref ref-type="bibr" rid="bib14">Coll et al., 2014</xref>) to type the global lineage of each isolate in our sample. We further excluded 322 isolates that did not get assigned a global lineage, another 100 isolates that were used in our longitudinal analysis (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>), 152 isolates typed as <italic>Mycobacterium bovis</italic>, and 35 isolates typed as lineage 7. To improve computational efficiency and runtime, we randomly down sampled (using Python’s random.sample() function) our collection of lineage 2 and lineage 4 isolates to include 7000 isolates of each lineage in our sample. This excluded 1064 lineage 2 isolates and 10,330 lineage four isolates. These steps yielded a genotypes matrix with dimensions 139,797 × 20,352. Finally, we excluded 10,899 SNP sites from this filtered genotypes matrix in which the minor allele count = 0. The genotypes matrix used for downstream analysis had dimensions 128,898 × 20,352 representing 128,898 SNP sites across 20,352 isolates. The global lineage breakdown of the 20,352 isolates was: L1 = 2802, L2 = 7000, L3 = 3352, L4 = 7,000, L5 = 101, L6 = 97.</p></sec><sec id="s4-14-2"><title>Phylogenetic convergence test</title><p>We tested 141/174 in-host SNPs in which the alternate (mutant) allele frequency increased substantially across sampling (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>) for phylogenetic convergence. We scanned a set 20,352 global and genetically diverse isolates from our 128,898 × 20,352 genotypes matrix for these SNPs (<xref ref-type="supplementary-material" rid="supp18">Supplementary file 18</xref>). To determine phylogenetic convergence for a given SNP site, we required that the alternate allele (called against the H37Rv reference) be present but not fixed in at least three global lineages. More specifically, we required that the alternate allele be detected in at least one isolate within each lineage and not in &gt;95% of the isolates in that lineage. Twenty-six SNP sites across 12 genes and three intergenic regions were detected as having a signal of phylogenetic convergence (<xref ref-type="supplementary-material" rid="supp19">Supplementary file 19</xref>). </p></sec><sec id="s4-14-3"><title>t-SNE visualization</title><p>To construct the t-SNE plots that captured the genetic relatedness of the 20,352 isolates in our sample, we first constructed a pairwise SNP distance matrix. To efficiently compute this using our 128,898 × 20,352 genotypes matrix, we binarized the genotypes matrix and used sparse matrix multiplication implemented in Scipy to compute five 20,352 × 20,352 similarity matrices (<xref ref-type="bibr" rid="bib82">Virtanen et al., 2020</xref>). We constructed a similarity matrix for each nucleotide (<italic>A</italic>, <italic>C</italic>, <italic>G</italic>, <italic>T</italic>) where row <italic>i</italic>, column <italic>j</italic> of the similarity matrix for nucleotide <italic>x</italic> stored the number of <italic>x</italic>’s that isolate <italic>i</italic> and isolate <italic>j</italic> shared in common across all SNP sites. The fifth similarity matrix (<italic>N</italic>) stored the number of SNP sites in which neither isolate <italic>i</italic> and isolate <italic>j</italic> had a missing value. The pairwise SNP distance matrix (<italic>D</italic>) was then computed as <inline-formula><mml:math id="inf72"><mml:mi mathvariant="bold-italic">D</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">N</mml:mi><mml:mo>-</mml:mo><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">A</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">C</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">G</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">T</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula>. <bold><italic>D</italic></bold> had dimensions 20,352 × 20,352 where row <italic>i</italic>, column <italic>j</italic> stored the number of SNP sites in which isolate <italic>i</italic> and isolate <italic>j</italic> disagreed. We used <italic>D</italic> as input into a t-SNE algorithm implemented in Scikit-learn (<xref ref-type="bibr" rid="bib60">Pedregosa et al., 2011</xref>) (settings: perplexity = 175, n_components = 2, metric = ‘precomputed’, n_iter = 1000, learning_rate = 1500) to compute the embeddings for all 20,352 isolates in our sample. We used these embeddings to visualize the genetic relatedness of the isolates in two dimensions and colored isolates (points on the t-SNE plot) by lineage (<xref ref-type="fig" rid="fig7">Figure 7A</xref>). For visualizing specific mutations, isolates were colored according to whether or not the alternate (mutant) allele was called (<xref ref-type="fig" rid="fig7">Figure 7B–F</xref>, <xref ref-type="fig" rid="fig7s1">Figure 7—figure supplements 1</xref>–<xref ref-type="fig" rid="fig7s2">2</xref>).</p></sec></sec><sec id="s4-15"><title>Data analysis and variant annotation</title><p>Data analysis was performed using custom scripts run in Python and interfaced with iPython (<xref ref-type="bibr" rid="bib61">Perez and Granger, 2007</xref>). Statistical tests were run with Statsmodels (<xref ref-type="bibr" rid="bib71">Seabold and Perktold, 2010</xref>) and Figures were plotted using Matplotlib (<xref ref-type="bibr" rid="bib39">Hunter, 2007</xref>). Numpy (<xref ref-type="bibr" rid="bib79">van der Walt et al., 2011</xref>), Biopython (<xref ref-type="bibr" rid="bib12">Cock et al., 2009</xref>) and Pandas (<xref ref-type="bibr" rid="bib53">McKinney, 2010</xref>) were all used extensively in data cleaning and manipulation. Functional annotation of SNPs was done in Biopython (<xref ref-type="bibr" rid="bib12">Cock et al., 2009</xref>) using the H37Rv reference genome and the corresponding genome annotation. For every SNP called, we used the H37Rv reference position provided by Pilon (<xref ref-type="bibr" rid="bib86">Walker et al., 2014</xref>) generated VCF file to extract any overlapping CDS region and annotated SNPs accordingly. Each overlapping CDS regions was then translated into its corresponding peptide sequence with both the reference and alternate allele. SNPs in which the peptide sequences did not differ between alleles were labeled <italic>synonymous</italic>, SNPs in which the peptide sequences did differ were labeled <italic>non-synonymous</italic> and if there were no overlapping CDS regions for that reference position, then the SNP was labeled <italic>intergenic</italic>.</p></sec><sec id="s4-16"><title>Pathway definitions</title><p>We used SEED (<xref ref-type="bibr" rid="bib58">Overbeek et al., 2014</xref>) subsystem annotation to conduct pathway analysis and downloaded the subsystem classification for all features of <italic>Mycobacterium tuberculosis</italic> H37Rv (id: 83332.1) (<xref ref-type="supplementary-material" rid="supp15">Supplementary file 15</xref>). We mapped all of the annotated features from SEED to the annotation for H37Rv. Due to the slight inconsistency between the start and end chromosomal coordinates for features from SEED and our H37Rv annotation, we assigned a locus from H37Rv to a subsystem if both the start and end coordinates for this locus fell within a 20 base-pair window of the start and end coordinates for a feature in the SEED annotation (<xref ref-type="supplementary-material" rid="supp16">Supplementary file 16</xref>).</p></sec><sec id="s4-17"><title>Data and materials availability</title><p>All Mtbc sequencing data was collected from previously published studies and is publicly available. Individual accession numbers for the Mtbc genomes analyzed in this study can be found in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref> and information on which studies from which the data was generated can be found in the Materials and methods, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref> and <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>. All packages and software used in this study have been noted in the Materials and methods. Custom scripts written in python version 2.7.15 were used to conduct all analyses and interfaced via Jupyter Notebooks. Jupyter Notebooks and scripts written for data processing and analysis can be found in the following GitHub repository - <ext-link ext-link-type="uri" xlink:href="https://github.com/farhat-lab/in-host-Mtbc-dynamics">https://github.com/farhat-lab/in-host-Mtbc-dynamics</ext-link>; <xref ref-type="bibr" rid="bib80">Vargas, 2021</xref> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:abc75682dade292117f5414a4c2e39b255734bf8;origin=https://github.com/farhat-lab/in-host-Mtbc-dynamics;visit=swh:1:snp:448c7f8292ae3c0918fbedd68678b381f0fd99cb;anchor=swh:1:rev:36e27011b5cfaed00521a38652fe2dc853832f25/">swh:1:rev:36e27011b5cfaed00521a38652fe2dc853832f25</ext-link>).</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We thank the members of the Farhat lab for helpful discussions and comments on the research project and manuscript. We thank S Fortune, N Hicks and D Warner for helpful suggestions on the manuscript. We thank A Narayan for helpful suggestions on constructing t-SNE visualizations for phylogenetic convergence. RVJ was supported by the National Science Foundation Graduate Research Fellowship under Grant No. DGE1745303. MF was supported by NIH/BD2K K01 ES026835 and NIH NIAID R01 AI55765. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health. Portions of this research were conducted on the O2 High Performance Compute Cluster, supported by the Research Computing Group, at Harvard Medical School.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Resources, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Data curation, Bioinformatics Support</p></fn><fn fn-type="con" id="con3"><p>Data curation, Bioinformatics Support</p></fn><fn fn-type="con" id="con4"><p>Resources, Cultured Mtb isolates and performed DNA extraction in preparation for PacBio sequencing</p></fn><fn fn-type="con" id="con5"><p>Resources, Prepared libraries and performed PacBio sequencing runs</p></fn><fn fn-type="con" id="con6"><p>Resources, Prepared libraries and performed PacBio sequencing runs</p></fn><fn fn-type="con" id="con7"><p>Resources, Cultured Mtb isolates and performed DNA extraction in preparation for PacBio sequencing</p></fn><fn fn-type="con" id="con8"><p>Resources, Cultured Mtb isolates and performed DNA extraction in preparation for PacBio sequencing</p></fn><fn fn-type="con" id="con9"><p>Resources, Cultured Mtb isolates and performed DNA extraction in preparation for PacBio sequencing</p></fn><fn fn-type="con" id="con10"><p>Conceptualization, Resources, Data curation, Formal analysis, Supervision, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>A table containing details for the eight studies; the sources for the longitudinal isolate pairs.</title><p>Information includes: (1) reference for each source study, (2) number of patients included in this study, (3) a description of the sample collection, (4) timing of when sputum samples were collected relative to treatment initiation/cessation (if available).</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp1-v2.xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>A table containing details for all replicate and longitudinal isolates before Kraken, F2, or pairwise SNP filtering.</title><p>Includes aggregated patient treatment from the source studies and metadata for each longitudinal isolate. We include columns that indicate the timing of sampling of Mtbc relative to treatment, the treatment regimen administered and final patient outcome (and relevant details). Patient outcomes are defined as follows: <italic>Delayed culture conversion</italic> (sputum culture positive at baseline and 2 months treatment initiation with genomic analysis consistent with clonal infection), <italic>Failure or Relapse</italic> (sputum culture positive at baseline and 4.5 months treatment initiation with genomic analysis consistent with clonal infection), <italic>Failure or Relapse or Default</italic> (sputum culture positive at interval of 4.5 months with genomic analysis consistent with clonal infection, only partial treatment data is available) or N/A if date data was of low resolution, not available or no treatment data was available. We also determined <italic>Reinfection</italic> and <italic>Mixed infection</italic> based on the genomic analysis. Legend for antibiotics: H = isoniazid, R = rifampicin, Rp = rifapentine, E = ethambutol, Z = pyrazinamide, C = capreomycin, S = cycloserine, P = para-aminosalicylic acid, Et = ethionamide, K = kanamycin, A = Amikacin, L = levofloxacin, Cf = ciprofloxacin, M = moxifloxacin, Sm = streptomycin.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp2-v2.xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>A table containing details for all <inline-formula><mml:math id="inf73"><mml:mfenced separators="|"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>400</mml:mn></mml:mrow></mml:mfenced></mml:math></inline-formula> longitudinal isolates used for in-host analysis after filtering for contaminated and mixed isolate pairs.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp3-v2.xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>A table with the gene categories assigned to each H37Rv locus tag.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp4-v2.xlsx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>A table containing a list of genomic regions (with H37Rv coordinates) associated with antibiotic resistance.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp5-v2.xlsx"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>A table containing all SNPs (with ΔAF ≥ 5%) in loci associated with antibiotic resistance (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) across our sample of 200 longitudinal isolate pairs.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp6-v2.xlsx"/></supplementary-material><supplementary-material id="supp7"><label>Supplementary file 7.</label><caption><title>A table containing all pre-existing antibiotic resistant SNPs detected in the first isolate collected from each patient with collection dates ≥2 months apart (178/200).</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp7-v2.xlsx"/></supplementary-material><supplementary-material id="supp8"><label>Supplementary file 8.</label><caption><title>A table containing information for all 174 in-host SNPs detected across all longitudinal isolate pairs.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp8-v2.xlsx"/></supplementary-material><supplementary-material id="supp9"><label>Supplementary file 9.</label><caption><title>A table with details for the 54 publicly available completed (reference) genomes used in our simulations.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp9-v2.xlsx"/></supplementary-material><supplementary-material id="supp10"><label>Supplementary file 10.</label><caption><title>A table with the non-redundant <italic>in-host</italic> SNPs identified within genes and used for SNP calling simulations.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp10-v2.xlsx"/></supplementary-material><supplementary-material id="supp11"><label>Supplementary file 11.</label><caption><title>A table containing all of the epitopes downloaded from IEDB on May 23, 2018.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61805-supp11-v2.csv"/></supplementary-material><supplementary-material id="supp12"><label>Supplementary file 12.</label><caption><title>A table containing the epitopes belonging to <italic>PPE18</italic> where an in-host SNP was detected.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp12-v2.xlsx"/></supplementary-material><supplementary-material id="supp13"><label>Supplementary file 13.</label><caption><title>A table of all genes identified as <italic>dense</italic>, along with assigned gene category and p-value from mutation density test.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp13-v2.xlsx"/></supplementary-material><supplementary-material id="supp14"><label>Supplementary file 14.</label><caption><title>A table of all genes identified as <italic>convergent</italic>, along with assigned gene category and the number of patients with an in-host SNP in each gene.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp14-v2.xlsx"/></supplementary-material><supplementary-material id="supp15"><label>Supplementary file 15.</label><caption><title>A table containing the downloaded SEED annotation for H37Rv.</title></caption><media mime-subtype="tab-separated-values" mimetype="text" xlink:href="elife-61805-supp15-v2.tsv"/></supplementary-material><supplementary-material id="supp16"><label>Supplementary file 16.</label><caption><title>A table containing the list of H37Rv locus tags corresponding to each subsystem classified by SEED.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61805-supp16-v2.csv"/></supplementary-material><supplementary-material id="supp17"><label>Supplementary file 17.</label><caption><title>A table containing the pathways and (corresponding in-host SNPs) displaying evidence of parallel evolution.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp17-v2.xlsx"/></supplementary-material><supplementary-material id="supp18"><label>Supplementary file 18.</label><caption><title>A table with details for all SNP calls made in a global collection of 20,352 publicly available isolates after screening for in-host SNPs (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>).</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp18-v2.xlsx"/></supplementary-material><supplementary-material id="supp19"><label>Supplementary file 19.</label><caption><title>A table with details for in-host SNPs (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>) that displayed a signature of phylogenetic convergence after screening a global collection of 20,352 publicly available isolates (Materials and methods).</title><p>The number of isolates with each unique mutation (broken down by global lineage) is given.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp19-v2.xlsx"/></supplementary-material><supplementary-material id="supp20"><label>Supplementary file 20.</label><caption><title>A table containing details for isolates that underwent Illumina and PacBio sequencing.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp20-v2.xlsx"/></supplementary-material><supplementary-material id="supp21"><label>Supplementary file 21.</label><caption><title>A table containing the genotypic resistance predictions for 13 antibiotics for all 614 longitudinal isolates from the 307 patients in our study (S: susceptible, R: resistant).</title><p>Legend for antibiotics: INH = isoniazid, RIF = rifampicin, EMB = ethambutol, PZA = pyrazinamide, CAP = capreomycin, PAS = para-aminosalicylic acid, ETH = ethionamide, KAN = kanamycin, AMK = Amikacin, LEVO = levofloxacin, CIP = ciprofloxacin, OFLX = ofloxacin, STR = streptomycin.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-61805-supp21-v2.xlsx"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="pdf" mimetype="application" xlink:href="elife-61805-transrepform-v2.pdf"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>All Mtbc sequencing data was collected from previously published studies and is publicly available. Individual accession numbers for the Mtbc genomes analyzed in this study can be found in Supplementary File 2 and information on which studies from which the data was generated can be found in the Methods, Figure 1 - figure supplement 1 and Supplementary File 1. All packages and software used in this study have been noted in the Methods. Custom scripts written in python version 2.7.15 were used to conduct all analyses and interfaced via Jupyter Notebooks. Jupyter Notebooks and scripts written for data processing and analysis can be found in the following GitHub repository - <ext-link ext-link-type="uri" xlink:href="https://github.com/farhat-lab/in-host-Mtbc-dynamics">https://github.com/farhat-lab/in-host-Mtbc-dynamics</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:rev:36e27011b5cfaed00521a38652fe2dc853832f25/">https://archive.softwareheritage.org/swh:1:rev:36e27011b5cfaed00521a38652fe2dc853832f25/</ext-link>).</p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Azad</surname> <given-names>AK</given-names></name><name><surname>Sadee</surname> <given-names>W</given-names></name><name><surname>Schlesinger</surname> <given-names>LS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Innate immune gene polymorphisms in tuberculosis</article-title><source>Infection and Immunity</source><volume>80</volume><fpage>3343</fpage><lpage>3359</lpage><pub-id pub-id-type="doi">10.1128/IAI.00443-12</pub-id><pub-id pub-id-type="pmid">22825450</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benson</surname> <given-names>DA</given-names></name><name><surname>Karsch-Mizrachi</surname> <given-names>I</given-names></name><name><surname>Lipman</surname> <given-names>DJ</given-names></name><name><surname>Ostell</surname> <given-names>J</given-names></name><name><surname>Sayers</surname> <given-names>EW</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>GenBank</article-title><source>Nucleic Acids Research</source><volume>37</volume><fpage>D26</fpage><lpage>D31</lpage><pub-id pub-id-type="doi">10.1093/nar/gkn723</pub-id><pub-id pub-id-type="pmid">18940867</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brennan</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The enigmatic PE/PPE multigene family of mycobacteria and tuberculosis vaccination</article-title><source>Infection and Immunity</source><volume>85</volume><elocation-id>e00969–16</elocation-id><pub-id pub-id-type="doi">10.1128/IAI.00969-16</pub-id><pub-id pub-id-type="pmid">28348055</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brennan</surname> <given-names>MJ</given-names></name><name><surname>Delogu</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>The PE multigene family: a 'molecular mantra' for mycobacteria</article-title><source>Trends in Microbiology</source><volume>10</volume><fpage>246</fpage><lpage>249</lpage><pub-id pub-id-type="doi">10.1016/S0966-842X(02)02335-1</pub-id><pub-id pub-id-type="pmid">11973159</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brites</surname> <given-names>D</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Co-evolution of Mycobacterium tuberculosis and <italic>Homo sapiens</italic></article-title><source>Immunological Reviews</source><volume>264</volume><fpage>6</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1111/imr.12264</pub-id><pub-id pub-id-type="pmid">25703549</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brodin</surname> <given-names>P</given-names></name><name><surname>Poquet</surname> <given-names>Y</given-names></name><name><surname>Levillain</surname> <given-names>F</given-names></name><name><surname>Peguillet</surname> <given-names>I</given-names></name><name><surname>Larrouy-Maumus</surname> <given-names>G</given-names></name><name><surname>Gilleron</surname> <given-names>M</given-names></name><name><surname>Ewann</surname> <given-names>F</given-names></name><name><surname>Christophe</surname> <given-names>T</given-names></name><name><surname>Fenistein</surname> <given-names>D</given-names></name><name><surname>Jang</surname> <given-names>J</given-names></name><name><surname>Jang</surname> <given-names>MS</given-names></name><name><surname>Park</surname> <given-names>SJ</given-names></name><name><surname>Rauzier</surname> <given-names>J</given-names></name><name><surname>Carralot</surname> <given-names>JP</given-names></name><name><surname>Shrimpton</surname> <given-names>R</given-names></name><name><surname>Genovesio</surname> <given-names>A</given-names></name><name><surname>Gonzalo-Asensio</surname> <given-names>JA</given-names></name><name><surname>Puzo</surname> <given-names>G</given-names></name><name><surname>Martin</surname> <given-names>C</given-names></name><name><surname>Brosch</surname> <given-names>R</given-names></name><name><surname>Stewart</surname> <given-names>GR</given-names></name><name><surname>Gicquel</surname> <given-names>B</given-names></name><name><surname>Neyrolles</surname> <given-names>O</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>High content phenotypic cell-based visual screen identifies Mycobacterium tuberculosis acyltrehalose-containing glycolipids involved in Phagosome remodeling</article-title><source>PLOS Pathogens</source><volume>6</volume><elocation-id>e1001100</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1001100</pub-id><pub-id pub-id-type="pmid">20844580</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bryant</surname> <given-names>JM</given-names></name><name><surname>Harris</surname> <given-names>SR</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Dawson</surname> <given-names>R</given-names></name><name><surname>Diacon</surname> <given-names>AH</given-names></name><name><surname>van Helden</surname> <given-names>P</given-names></name><name><surname>Pym</surname> <given-names>A</given-names></name><name><surname>Mahayiddin</surname> <given-names>AA</given-names></name><name><surname>Chuchottaworn</surname> <given-names>C</given-names></name><name><surname>Sanne</surname> <given-names>IM</given-names></name><name><surname>Louw</surname> <given-names>C</given-names></name><name><surname>Boeree</surname> <given-names>MJ</given-names></name><name><surname>Hoelscher</surname> <given-names>M</given-names></name><name><surname>McHugh</surname> <given-names>TD</given-names></name><name><surname>Bateson</surname> <given-names>AL</given-names></name><name><surname>Hunt</surname> <given-names>RD</given-names></name><name><surname>Mwaigwisya</surname> <given-names>S</given-names></name><name><surname>Wright</surname> <given-names>L</given-names></name><name><surname>Gillespie</surname> <given-names>SH</given-names></name><name><surname>Bentley</surname> <given-names>SD</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Whole-genome sequencing to establish relapse or re-infection with Mycobacterium tuberculosis: a retrospective observational study</article-title><source>The Lancet Respiratory Medicine</source><volume>1</volume><fpage>786</fpage><lpage>792</lpage><pub-id pub-id-type="doi">10.1016/S2213-2600(13)70231-5</pub-id><pub-id pub-id-type="pmid">24461758</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Casali</surname> <given-names>N</given-names></name><name><surname>Broda</surname> <given-names>A</given-names></name><name><surname>Harris</surname> <given-names>SR</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Brown</surname> <given-names>T</given-names></name><name><surname>Drobniewski</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Whole genome sequence analysis of a large Isoniazid-Resistant tuberculosis outbreak in London: a retrospective observational study</article-title><source>PLOS Medicine</source><volume>13</volume><elocation-id>e1002137</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pmed.1002137</pub-id><pub-id pub-id-type="pmid">27701423</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chin</surname> <given-names>CS</given-names></name><name><surname>Alexander</surname> <given-names>DH</given-names></name><name><surname>Marks</surname> <given-names>P</given-names></name><name><surname>Klammer</surname> <given-names>AA</given-names></name><name><surname>Drake</surname> <given-names>J</given-names></name><name><surname>Heiner</surname> <given-names>C</given-names></name><name><surname>Clum</surname> <given-names>A</given-names></name><name><surname>Copeland</surname> <given-names>A</given-names></name><name><surname>Huddleston</surname> <given-names>J</given-names></name><name><surname>Eichler</surname> <given-names>EE</given-names></name><name><surname>Turner</surname> <given-names>SW</given-names></name><name><surname>Korlach</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Nonhybrid, finished microbial genome assemblies from long-read SMRT sequencing data</article-title><source>Nature Methods</source><volume>10</volume><fpage>563</fpage><lpage>569</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2474</pub-id><pub-id pub-id-type="pmid">23644548</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chiner-Oms</surname> <given-names>Á</given-names></name><name><surname>Berney</surname> <given-names>M</given-names></name><name><surname>Boinett</surname> <given-names>C</given-names></name><name><surname>González-Candelas</surname> <given-names>F</given-names></name><name><surname>Young</surname> <given-names>DB</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name><name><surname>Jacobs</surname> <given-names>WR</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Cortes</surname> <given-names>T</given-names></name><name><surname>Comas</surname> <given-names>I</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Genome-wide mutational biases fuel transcriptional diversity in the Mycobacterium tuberculosis complex</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>3994</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-11948-6</pub-id><pub-id pub-id-type="pmid">31488832</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Clemmensen</surname> <given-names>HS</given-names></name><name><surname>Knudsen</surname> <given-names>NPH</given-names></name><name><surname>Rasmussen</surname> <given-names>EM</given-names></name><name><surname>Winkler</surname> <given-names>J</given-names></name><name><surname>Rosenkrands</surname> <given-names>I</given-names></name><name><surname>Ahmad</surname> <given-names>A</given-names></name><name><surname>Lillebaek</surname> <given-names>T</given-names></name><name><surname>Sherman</surname> <given-names>DR</given-names></name><name><surname>Andersen</surname> <given-names>PL</given-names></name><name><surname>Aagaard</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>An attenuated Mycobacterium tuberculosis clinical strain with a defect in ESX-1 secretion induces minimal host immune responses and pathology</article-title><source>Scientific Reports</source><volume>7</volume><elocation-id>46666</elocation-id><pub-id pub-id-type="doi">10.1038/srep46666</pub-id><pub-id pub-id-type="pmid">28436493</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cock</surname> <given-names>PJ</given-names></name><name><surname>Antao</surname> <given-names>T</given-names></name><name><surname>Chang</surname> <given-names>JT</given-names></name><name><surname>Chapman</surname> <given-names>BA</given-names></name><name><surname>Cox</surname> <given-names>CJ</given-names></name><name><surname>Dalke</surname> <given-names>A</given-names></name><name><surname>Friedberg</surname> <given-names>I</given-names></name><name><surname>Hamelryck</surname> <given-names>T</given-names></name><name><surname>Kauff</surname> <given-names>F</given-names></name><name><surname>Wilczynski</surname> <given-names>B</given-names></name><name><surname>de Hoon</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Biopython: freely available Python tools for computational molecular biology and bioinformatics</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1422</fpage><lpage>1423</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp163</pub-id><pub-id pub-id-type="pmid">19304878</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cole</surname> <given-names>ST</given-names></name><name><surname>Brosch</surname> <given-names>R</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Garnier</surname> <given-names>T</given-names></name><name><surname>Churcher</surname> <given-names>C</given-names></name><name><surname>Harris</surname> <given-names>D</given-names></name><name><surname>Gordon</surname> <given-names>SV</given-names></name><name><surname>Eiglmeier</surname> <given-names>K</given-names></name><name><surname>Gas</surname> <given-names>S</given-names></name><name><surname>Barry</surname> <given-names>CE</given-names></name><name><surname>Tekaia</surname> <given-names>F</given-names></name><name><surname>Badcock</surname> <given-names>K</given-names></name><name><surname>Basham</surname> <given-names>D</given-names></name><name><surname>Brown</surname> <given-names>D</given-names></name><name><surname>Chillingworth</surname> <given-names>T</given-names></name><name><surname>Connor</surname> <given-names>R</given-names></name><name><surname>Davies</surname> <given-names>R</given-names></name><name><surname>Devlin</surname> <given-names>K</given-names></name><name><surname>Feltwell</surname> <given-names>T</given-names></name><name><surname>Gentles</surname> <given-names>S</given-names></name><name><surname>Hamlin</surname> <given-names>N</given-names></name><name><surname>Holroyd</surname> <given-names>S</given-names></name><name><surname>Hornsby</surname> <given-names>T</given-names></name><name><surname>Jagels</surname> <given-names>K</given-names></name><name><surname>Krogh</surname> <given-names>A</given-names></name><name><surname>McLean</surname> <given-names>J</given-names></name><name><surname>Moule</surname> <given-names>S</given-names></name><name><surname>Murphy</surname> <given-names>L</given-names></name><name><surname>Oliver</surname> <given-names>K</given-names></name><name><surname>Osborne</surname> <given-names>J</given-names></name><name><surname>Quail</surname> <given-names>MA</given-names></name><name><surname>Rajandream</surname> <given-names>MA</given-names></name><name><surname>Rogers</surname> <given-names>J</given-names></name><name><surname>Rutter</surname> <given-names>S</given-names></name><name><surname>Seeger</surname> <given-names>K</given-names></name><name><surname>Skelton</surname> <given-names>J</given-names></name><name><surname>Squares</surname> <given-names>R</given-names></name><name><surname>Squares</surname> <given-names>S</given-names></name><name><surname>Sulston</surname> <given-names>JE</given-names></name><name><surname>Taylor</surname> <given-names>K</given-names></name><name><surname>Whitehead</surname> <given-names>S</given-names></name><name><surname>Barrell</surname> <given-names>BG</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Deciphering the biology of Mycobacterium tuberculosis from the complete genome sequence</article-title><source>Nature</source><volume>393</volume><elocation-id>537</elocation-id><pub-id pub-id-type="doi">10.1038/31159</pub-id><pub-id pub-id-type="pmid">9634230</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coll</surname> <given-names>F</given-names></name><name><surname>McNerney</surname> <given-names>R</given-names></name><name><surname>Guerra-Assunção</surname> <given-names>JA</given-names></name><name><surname>Glynn</surname> <given-names>JR</given-names></name><name><surname>Perdigão</surname> <given-names>J</given-names></name><name><surname>Viveiros</surname> <given-names>M</given-names></name><name><surname>Portugal</surname> <given-names>I</given-names></name><name><surname>Pain</surname> <given-names>A</given-names></name><name><surname>Martin</surname> <given-names>N</given-names></name><name><surname>Clark</surname> <given-names>TG</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A robust SNP barcode for typing Mycobacterium tuberculosis complex strains</article-title><source>Nature Communications</source><volume>5</volume><elocation-id>4812</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms5812</pub-id><pub-id pub-id-type="pmid">25176035</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comas</surname> <given-names>I</given-names></name><name><surname>Chakravartti</surname> <given-names>J</given-names></name><name><surname>Small</surname> <given-names>PM</given-names></name><name><surname>Galagan</surname> <given-names>J</given-names></name><name><surname>Niemann</surname> <given-names>S</given-names></name><name><surname>Kremer</surname> <given-names>K</given-names></name><name><surname>Ernst</surname> <given-names>JD</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Human T cell epitopes of Mycobacterium tuberculosis are evolutionarily hyperconserved</article-title><source>Nature Genetics</source><volume>42</volume><fpage>498</fpage><lpage>503</lpage><pub-id pub-id-type="doi">10.1038/ng.590</pub-id><pub-id pub-id-type="pmid">20495566</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Copin</surname> <given-names>R</given-names></name><name><surname>Coscollá</surname> <given-names>M</given-names></name><name><surname>Seiffert</surname> <given-names>SN</given-names></name><name><surname>Bothamley</surname> <given-names>G</given-names></name><name><surname>Sutherland</surname> <given-names>J</given-names></name><name><surname>Mbayo</surname> <given-names>G</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name><name><surname>Ernst</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Sequence diversity in the pe_pgrs genes of Mycobacterium tuberculosis is independent of human T cell recognition</article-title><source>mBio</source><volume>5</volume><elocation-id>e00960-13</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.00960-13</pub-id><pub-id pub-id-type="pmid">24425732</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Copin</surname> <given-names>R</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Louie</surname> <given-names>E</given-names></name><name><surname>Escuyer</surname> <given-names>V</given-names></name><name><surname>Coscolla</surname> <given-names>M</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name><name><surname>Palmer</surname> <given-names>GH</given-names></name><name><surname>Ernst</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Within host evolution selects for a dominant genotype of Mycobacterium tuberculosis while T cells increase pathogen genetic diversity</article-title><source>PLOS Pathogens</source><volume>12</volume><elocation-id>e1006111</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1006111</pub-id><pub-id pub-id-type="pmid">27973588</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coscolla</surname> <given-names>M</given-names></name><name><surname>Copin</surname> <given-names>R</given-names></name><name><surname>Sutherland</surname> <given-names>J</given-names></name><name><surname>Gehre</surname> <given-names>F</given-names></name><name><surname>de Jong</surname> <given-names>B</given-names></name><name><surname>Owolabi</surname> <given-names>O</given-names></name><name><surname>Mbayo</surname> <given-names>G</given-names></name><name><surname>Giardina</surname> <given-names>F</given-names></name><name><surname>Ernst</surname> <given-names>JD</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>M. tuberculosis T cell epitope analysis reveals paucity of antigenic variation and identifies rare variable TB antigens</article-title><source>Cell Host &amp; Microbe</source><volume>18</volume><fpage>538</fpage><lpage>548</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2015.10.008</pub-id><pub-id pub-id-type="pmid">26607161</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Didelot</surname> <given-names>X</given-names></name><name><surname>Walker</surname> <given-names>AS</given-names></name><name><surname>Peto</surname> <given-names>TE</given-names></name><name><surname>Crook</surname> <given-names>DW</given-names></name><name><surname>Wilson</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Within-host evolution of bacterial pathogens</article-title><source>Nature Reviews Microbiology</source><volume>14</volume><fpage>150</fpage><lpage>162</lpage><pub-id pub-id-type="doi">10.1038/nrmicro.2015.13</pub-id><pub-id pub-id-type="pmid">26806595</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dillon</surname> <given-names>MM</given-names></name><name><surname>Sung</surname> <given-names>W</given-names></name><name><surname>Lynch</surname> <given-names>M</given-names></name><name><surname>Cooper</surname> <given-names>VS</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The rate and molecular spectrum of spontaneous mutations in the GC-Rich multichromosome genome of Burkholderia cenocepacia</article-title><source>Genetics</source><volume>200</volume><fpage>935</fpage><lpage>946</lpage><pub-id pub-id-type="doi">10.1534/genetics.115.176834</pub-id><pub-id pub-id-type="pmid">25971664</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dixit</surname> <given-names>A</given-names></name><name><surname>Freschi</surname> <given-names>L</given-names></name><name><surname>Vargas</surname> <given-names>R</given-names></name><name><surname>Calderon</surname> <given-names>R</given-names></name><name><surname>Sacchettini</surname> <given-names>J</given-names></name><name><surname>Drobniewski</surname> <given-names>F</given-names></name><name><surname>Galea</surname> <given-names>JT</given-names></name><name><surname>Contreras</surname> <given-names>C</given-names></name><name><surname>Yataco</surname> <given-names>R</given-names></name><name><surname>Zhang</surname> <given-names>Z</given-names></name><name><surname>Lecca</surname> <given-names>L</given-names></name><name><surname>Kolokotronis</surname> <given-names>SO</given-names></name><name><surname>Mathema</surname> <given-names>B</given-names></name><name><surname>Farhat</surname> <given-names>MR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Whole genome sequencing identifies bacterial factors affecting transmission of multidrug-resistant tuberculosis in a high-prevalence setting</article-title><source>Scientific Reports</source><volume>9</volume><elocation-id>5602</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-019-41967-8</pub-id><pub-id pub-id-type="pmid">30944370</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dreyer</surname> <given-names>V</given-names></name><name><surname>Utpatel</surname> <given-names>C</given-names></name><name><surname>Kohl</surname> <given-names>TA</given-names></name><name><surname>Barilar</surname> <given-names>I</given-names></name><name><surname>Gröschel</surname> <given-names>MI</given-names></name><name><surname>Feuerriegel</surname> <given-names>S</given-names></name><name><surname>Niemann</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Detection of low-frequency resistance-mediating SNPs in next-generation sequencing data of Mycobacterium tuberculosis complex strains with binoSNP</article-title><source>Scientific Reports</source><volume>10</volume><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.1038/s41598-020-64708-8</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Epperson</surname> <given-names>LE</given-names></name><name><surname>Strong</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A scalable, efficient, and safe method to prepare high quality DNA from mycobacteria and other challenging cells</article-title><source>Journal of Clinical Tuberculosis and Other Mycobacterial Diseases</source><volume>19</volume><elocation-id>100150</elocation-id><pub-id pub-id-type="doi">10.1016/j.jctube.2020.100150</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ernst</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mechanisms of M. tuberculosis Immune Evasion as Challenges to TB Vaccine Design</article-title><source>Cell Host &amp; Microbe</source><volume>24</volume><fpage>34</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2018.06.004</pub-id><pub-id pub-id-type="pmid">30001523</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farhat</surname> <given-names>MR</given-names></name><name><surname>Shapiro</surname> <given-names>BJ</given-names></name><name><surname>Kieser</surname> <given-names>KJ</given-names></name><name><surname>Sultana</surname> <given-names>R</given-names></name><name><surname>Jacobson</surname> <given-names>KR</given-names></name><name><surname>Victor</surname> <given-names>TC</given-names></name><name><surname>Warren</surname> <given-names>RM</given-names></name><name><surname>Streicher</surname> <given-names>EM</given-names></name><name><surname>Calver</surname> <given-names>A</given-names></name><name><surname>Sloutsky</surname> <given-names>A</given-names></name><name><surname>Kaur</surname> <given-names>D</given-names></name><name><surname>Posey</surname> <given-names>JE</given-names></name><name><surname>Plikaytis</surname> <given-names>B</given-names></name><name><surname>Oggioni</surname> <given-names>MR</given-names></name><name><surname>Gardy</surname> <given-names>JL</given-names></name><name><surname>Johnston</surname> <given-names>JC</given-names></name><name><surname>Rodrigues</surname> <given-names>M</given-names></name><name><surname>Tang</surname> <given-names>PK</given-names></name><name><surname>Kato-Maeda</surname> <given-names>M</given-names></name><name><surname>Borowsky</surname> <given-names>ML</given-names></name><name><surname>Muddukrishna</surname> <given-names>B</given-names></name><name><surname>Kreiswirth</surname> <given-names>BN</given-names></name><name><surname>Kurepina</surname> <given-names>N</given-names></name><name><surname>Galagan</surname> <given-names>J</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name><name><surname>Birren</surname> <given-names>B</given-names></name><name><surname>Rubin</surname> <given-names>EJ</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Sabeti</surname> <given-names>PC</given-names></name><name><surname>Murray</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Genomic analysis identifies targets of convergent positive selection in drug-resistant Mycobacterium tuberculosis</article-title><source>Nature Genetics</source><volume>45</volume><fpage>1183</fpage><lpage>1189</lpage><pub-id pub-id-type="doi">10.1038/ng.2747</pub-id><pub-id pub-id-type="pmid">23995135</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farhat</surname> <given-names>MR</given-names></name><name><surname>Shapiro</surname> <given-names>BJ</given-names></name><name><surname>Sheppard</surname> <given-names>SK</given-names></name><name><surname>Colijn</surname> <given-names>C</given-names></name><name><surname>Murray</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A phylogeny-based sampling strategy and power calculator informs genome-wide associations study design for microbial pathogens</article-title><source>Genome Medicine</source><volume>6</volume><elocation-id>101</elocation-id><pub-id pub-id-type="doi">10.1186/s13073-014-0101-7</pub-id><pub-id pub-id-type="pmid">25484920</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farhat</surname> <given-names>MR</given-names></name><name><surname>Sultana</surname> <given-names>R</given-names></name><name><surname>Iartchouk</surname> <given-names>O</given-names></name><name><surname>Bozeman</surname> <given-names>S</given-names></name><name><surname>Galagan</surname> <given-names>J</given-names></name><name><surname>Sisk</surname> <given-names>P</given-names></name><name><surname>Stolte</surname> <given-names>C</given-names></name><name><surname>Nebenzahl-Guimaraes</surname> <given-names>H</given-names></name><name><surname>Jacobson</surname> <given-names>K</given-names></name><name><surname>Sloutsky</surname> <given-names>A</given-names></name><name><surname>Kaur</surname> <given-names>D</given-names></name><name><surname>Posey</surname> <given-names>J</given-names></name><name><surname>Kreiswirth</surname> <given-names>BN</given-names></name><name><surname>Kurepina</surname> <given-names>N</given-names></name><name><surname>Rigouts</surname> <given-names>L</given-names></name><name><surname>Streicher</surname> <given-names>EM</given-names></name><name><surname>Victor</surname> <given-names>TC</given-names></name><name><surname>Warren</surname> <given-names>RM</given-names></name><name><surname>van Soolingen</surname> <given-names>D</given-names></name><name><surname>Murray</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genetic determinants of drug resistance in Mycobacterium tuberculosis and their diagnostic value</article-title><source>American Journal of Respiratory and Critical Care Medicine</source><volume>194</volume><fpage>621</fpage><lpage>630</lpage><pub-id pub-id-type="doi">10.1164/rccm.201510-2091OC</pub-id><pub-id pub-id-type="pmid">26910495</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farhat</surname> <given-names>MR</given-names></name><name><surname>Freschi</surname> <given-names>L</given-names></name><name><surname>Calderon</surname> <given-names>R</given-names></name><name><surname>Ioerger</surname> <given-names>T</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name><name><surname>Meehan</surname> <given-names>CJ</given-names></name><name><surname>de Jong</surname> <given-names>B</given-names></name><name><surname>Rigouts</surname> <given-names>L</given-names></name><name><surname>Sloutsky</surname> <given-names>A</given-names></name><name><surname>Kaur</surname> <given-names>D</given-names></name><name><surname>Sunyaev</surname> <given-names>S</given-names></name><name><surname>van Soolingen</surname> <given-names>D</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name><name><surname>Sacchettini</surname> <given-names>J</given-names></name><name><surname>Murray</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>GWAS for quantitative resistance phenotypes in Mycobacterium tuberculosis reveals resistance genes and regulatory regions</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>2128</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-10110-6</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ford</surname> <given-names>CB</given-names></name><name><surname>Lin</surname> <given-names>PL</given-names></name><name><surname>Chase</surname> <given-names>MR</given-names></name><name><surname>Shah</surname> <given-names>RR</given-names></name><name><surname>Iartchouk</surname> <given-names>O</given-names></name><name><surname>Galagan</surname> <given-names>J</given-names></name><name><surname>Mohaideen</surname> <given-names>N</given-names></name><name><surname>Ioerger</surname> <given-names>TR</given-names></name><name><surname>Sacchettini</surname> <given-names>JC</given-names></name><name><surname>Lipsitch</surname> <given-names>M</given-names></name><name><surname>Flynn</surname> <given-names>JL</given-names></name><name><surname>Fortune</surname> <given-names>SM</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Use of whole genome sequencing to estimate the mutation rate of Mycobacterium tuberculosis during latent infection</article-title><source>Nature Genetics</source><volume>43</volume><fpage>482</fpage><lpage>486</lpage><pub-id pub-id-type="doi">10.1038/ng.811</pub-id><pub-id pub-id-type="pmid">21516081</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ford</surname> <given-names>C</given-names></name><name><surname>Yusim</surname> <given-names>K</given-names></name><name><surname>Ioerger</surname> <given-names>T</given-names></name><name><surname>Feng</surname> <given-names>S</given-names></name><name><surname>Chase</surname> <given-names>M</given-names></name><name><surname>Greene</surname> <given-names>M</given-names></name><name><surname>Korber</surname> <given-names>B</given-names></name><name><surname>Fortune</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Mycobacterium tuberculosis--heterogeneity revealed through whole genome sequencing</article-title><source>Tuberculosis</source><volume>92</volume><fpage>194</fpage><lpage>201</lpage><pub-id pub-id-type="doi">10.1016/j.tube.2011.11.003</pub-id><pub-id pub-id-type="pmid">22218163</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gagneux</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Ecology and evolution of <italic>Mycobacterium tuberculosis</italic></article-title><source>Nature Reviews Microbiology</source><volume>16</volume><fpage>202</fpage><lpage>213</lpage><pub-id pub-id-type="doi">10.1038/nrmicro.2018.8</pub-id><pub-id pub-id-type="pmid">29456241</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goig</surname> <given-names>GA</given-names></name><name><surname>Blanco</surname> <given-names>S</given-names></name><name><surname>Garcia-Basteiro</surname> <given-names>AL</given-names></name><name><surname>Comas</surname> <given-names>I</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Contaminant DNA in bacterial sequencing experiments is a major source of false genetic variability</article-title><source>BMC Biology</source><volume>18</volume><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1186/s12915-020-0748-z</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gopinath</surname> <given-names>K</given-names></name><name><surname>Moosa</surname> <given-names>A</given-names></name><name><surname>Mizrahi</surname> <given-names>V</given-names></name><name><surname>Warner</surname> <given-names>DF</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Vitamin B(12) metabolism in Mycobacterium tuberculosis</article-title><source>Future Microbiology</source><volume>8</volume><fpage>1405</fpage><lpage>1418</lpage><pub-id pub-id-type="doi">10.2217/fmb.13.113</pub-id><pub-id pub-id-type="pmid">24199800</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guerra-Assunção</surname> <given-names>JA</given-names></name><name><surname>Houben</surname> <given-names>RM</given-names></name><name><surname>Crampin</surname> <given-names>AC</given-names></name><name><surname>Mzembe</surname> <given-names>T</given-names></name><name><surname>Mallard</surname> <given-names>K</given-names></name><name><surname>Coll</surname> <given-names>F</given-names></name><name><surname>Khan</surname> <given-names>P</given-names></name><name><surname>Banda</surname> <given-names>L</given-names></name><name><surname>Chiwaya</surname> <given-names>A</given-names></name><name><surname>Pereira</surname> <given-names>RP</given-names></name><name><surname>McNerney</surname> <given-names>R</given-names></name><name><surname>Harris</surname> <given-names>D</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Clark</surname> <given-names>TG</given-names></name><name><surname>Glynn</surname> <given-names>JR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Recurrence due to relapse or reinfection with Mycobacterium tuberculosis: a whole-genome sequencing approach in a large, population-based cohort with a high HIV infection prevalence and active follow-up</article-title><source>Journal of Infectious Diseases</source><volume>211</volume><fpage>1154</fpage><lpage>1163</lpage><pub-id pub-id-type="doi">10.1093/infdis/jiu574</pub-id><pub-id pub-id-type="pmid">25336729</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hebert</surname> <given-names>AM</given-names></name><name><surname>Talarico</surname> <given-names>S</given-names></name><name><surname>Yang</surname> <given-names>D</given-names></name><name><surname>Durmaz</surname> <given-names>R</given-names></name><name><surname>Marrs</surname> <given-names>CF</given-names></name><name><surname>Zhang</surname> <given-names>L</given-names></name><name><surname>Foxman</surname> <given-names>B</given-names></name><name><surname>Yang</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>DNA polymorphisms in the pepA and PPE18 genes among clinical strains of Mycobacterium tuberculosis: implications for vaccine efficacy</article-title><source>Infection and Immunity</source><volume>75</volume><fpage>5798</fpage><lpage>5805</lpage><pub-id pub-id-type="doi">10.1128/IAI.00335-07</pub-id><pub-id pub-id-type="pmid">17893137</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hicks</surname> <given-names>ND</given-names></name><name><surname>Yang</surname> <given-names>J</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Zhao</surname> <given-names>B</given-names></name><name><surname>Grad</surname> <given-names>YH</given-names></name><name><surname>Liu</surname> <given-names>L</given-names></name><name><surname>Ou</surname> <given-names>X</given-names></name><name><surname>Chang</surname> <given-names>Z</given-names></name><name><surname>Xia</surname> <given-names>H</given-names></name><name><surname>Zhou</surname> <given-names>Y</given-names></name><name><surname>Wang</surname> <given-names>S</given-names></name><name><surname>Dong</surname> <given-names>J</given-names></name><name><surname>Sun</surname> <given-names>L</given-names></name><name><surname>Zhu</surname> <given-names>Y</given-names></name><name><surname>Zhao</surname> <given-names>Y</given-names></name><name><surname>Jin</surname> <given-names>Q</given-names></name><name><surname>Fortune</surname> <given-names>SM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Clinically prevalent mutations in Mycobacterium tuberculosis alter propionate metabolism and mediate multidrug tolerance</article-title><source>Nature Microbiology</source><volume>3</volume><fpage>1032</fpage><lpage>1042</lpage><pub-id pub-id-type="doi">10.1038/s41564-018-0218-3</pub-id><pub-id pub-id-type="pmid">30082724</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>W</given-names></name><name><surname>Li</surname> <given-names>L</given-names></name><name><surname>Myers</surname> <given-names>JR</given-names></name><name><surname>Marth</surname> <given-names>GT</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>ART: a next-generation sequencing read simulator</article-title><source>Bioinformatics</source><volume>28</volume><fpage>593</fpage><lpage>594</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr708</pub-id><pub-id pub-id-type="pmid">22199392</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunt</surname> <given-names>M</given-names></name><name><surname>Silva</surname> <given-names>ND</given-names></name><name><surname>Otto</surname> <given-names>TD</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Keane</surname> <given-names>JA</given-names></name><name><surname>Harris</surname> <given-names>SR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Circlator: automated circularization of genome assemblies using long sequencing reads</article-title><source>Genome Biology</source><volume>16</volume><elocation-id>294</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-015-0849-0</pub-id><pub-id pub-id-type="pmid">26714481</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Matplotlib: a 2D graphics environment</article-title><source>Computing in Science &amp; Engineering</source><volume>9</volume><fpage>90</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Imperial</surname> <given-names>MZ</given-names></name><name><surname>Nahid</surname> <given-names>P</given-names></name><name><surname>Phillips</surname> <given-names>PPJ</given-names></name><name><surname>Davies</surname> <given-names>GR</given-names></name><name><surname>Fielding</surname> <given-names>K</given-names></name><name><surname>Hanna</surname> <given-names>D</given-names></name><name><surname>Hermann</surname> <given-names>D</given-names></name><name><surname>Wallis</surname> <given-names>RS</given-names></name><name><surname>Johnson</surname> <given-names>JL</given-names></name><name><surname>Lienhardt</surname> <given-names>C</given-names></name><name><surname>Savic</surname> <given-names>RM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A patient-level pooled analysis of treatment-shortening regimens for drug-susceptible pulmonary tuberculosis</article-title><source>Nature Medicine</source><volume>24</volume><fpage>1708</fpage><lpage>1715</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0224-2</pub-id><pub-id pub-id-type="pmid">30397355</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kleinnijenhuis</surname> <given-names>J</given-names></name><name><surname>Oosting</surname> <given-names>M</given-names></name><name><surname>Joosten</surname> <given-names>LA</given-names></name><name><surname>Netea</surname> <given-names>MG</given-names></name><name><surname>Van Crevel</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Innate immune recognition of Mycobacterium tuberculosis</article-title><source>Clinical &amp; Developmental Immunology</source><volume>2011</volume><elocation-id>405310</elocation-id><pub-id pub-id-type="doi">10.1155/2011/405310</pub-id><pub-id pub-id-type="pmid">21603213</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kolmogorov</surname> <given-names>M</given-names></name><name><surname>Yuan</surname> <given-names>J</given-names></name><name><surname>Lin</surname> <given-names>Y</given-names></name><name><surname>Pevzner</surname> <given-names>PA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Assembly of long, error-prone reads using repeat graphs</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>540</fpage><lpage>546</lpage><pub-id pub-id-type="doi">10.1038/s41587-019-0072-8</pub-id><pub-id pub-id-type="pmid">30936562</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kurtz</surname> <given-names>S</given-names></name><name><surname>Phillippy</surname> <given-names>A</given-names></name><name><surname>Delcher</surname> <given-names>AL</given-names></name><name><surname>Smoot</surname> <given-names>M</given-names></name><name><surname>Shumway</surname> <given-names>M</given-names></name><name><surname>Antonescu</surname> <given-names>C</given-names></name><name><surname>Salzberg</surname> <given-names>SL</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Versatile and open software for comparing large genomes</article-title><source>Genome Biology</source><volume>5</volume><elocation-id>R12</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2004-5-2-r12</pub-id><pub-id pub-id-type="pmid">14759262</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Handsaker</surname> <given-names>B</given-names></name><name><surname>Wysoker</surname> <given-names>A</given-names></name><name><surname>Fennell</surname> <given-names>T</given-names></name><name><surname>Ruan</surname> <given-names>J</given-names></name><name><surname>Homer</surname> <given-names>N</given-names></name><name><surname>Marth</surname> <given-names>G</given-names></name><name><surname>Abecasis</surname> <given-names>G</given-names></name><name><surname>Durbin</surname> <given-names>R</given-names></name><collab>1000 Genome Project Data Processing Subgroup</collab></person-group><year iso-8601-date="2009">2009</year><article-title>The sequence alignment/Map format and SAMtools</article-title><source>Bioinformatics</source><volume>25</volume><fpage>2078</fpage><lpage>2079</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id><pub-id pub-id-type="pmid">19505943</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Minimap2: pairwise alignment for nucleotide sequences</article-title><source>Bioinformatics</source><volume>34</volume><fpage>3094</fpage><lpage>3100</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty191</pub-id><pub-id pub-id-type="pmid">29750242</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Durbin</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Fast and accurate short read alignment with Burrows-Wheeler transform</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1754</fpage><lpage>1760</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id><pub-id pub-id-type="pmid">19451168</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lieberman</surname> <given-names>TD</given-names></name><name><surname>Michel</surname> <given-names>J-B</given-names></name><name><surname>Aingaran</surname> <given-names>M</given-names></name><name><surname>Potter-Bynoe</surname> <given-names>G</given-names></name><name><surname>Roux</surname> <given-names>D</given-names></name><name><surname>Davis</surname> <given-names>MR</given-names></name><name><surname>Skurnik</surname> <given-names>D</given-names></name><name><surname>Leiby</surname> <given-names>N</given-names></name><name><surname>LiPuma</surname> <given-names>JJ</given-names></name><name><surname>Goldberg</surname> <given-names>JB</given-names></name><name><surname>McAdam</surname> <given-names>AJ</given-names></name><name><surname>Priebe</surname> <given-names>GP</given-names></name><name><surname>Kishony</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Parallel bacterial evolution within multiple patients identifies candidate pathogenicity genes</article-title><source>Nature Genetics</source><volume>43</volume><fpage>1275</fpage><lpage>1280</lpage><pub-id pub-id-type="doi">10.1038/ng.997</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lieberman</surname> <given-names>TD</given-names></name><name><surname>Flett</surname> <given-names>KB</given-names></name><name><surname>Yelin</surname> <given-names>I</given-names></name><name><surname>Martin</surname> <given-names>TR</given-names></name><name><surname>McAdam</surname> <given-names>AJ</given-names></name><name><surname>Priebe</surname> <given-names>GP</given-names></name><name><surname>Kishony</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genetic variation of a bacterial pathogen within individuals with cystic fibrosis provides a record of selective pressures</article-title><source>Nature Genetics</source><volume>46</volume><fpage>82</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1038/ng.2848</pub-id><pub-id pub-id-type="pmid">24316980</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lieberman</surname> <given-names>TD</given-names></name><name><surname>Wilson</surname> <given-names>D</given-names></name><name><surname>Misra</surname> <given-names>R</given-names></name><name><surname>Xiong</surname> <given-names>LL</given-names></name><name><surname>Moodley</surname> <given-names>P</given-names></name><name><surname>Cohen</surname> <given-names>T</given-names></name><name><surname>Kishony</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genomic diversity in autopsy samples reveals within-host dissemination of HIV-associated Mycobacterium tuberculosis</article-title><source>Nature Medicine</source><volume>22</volume><fpage>1470</fpage><lpage>1474</lpage><pub-id pub-id-type="doi">10.1038/nm.4205</pub-id><pub-id pub-id-type="pmid">27798613</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>PL</given-names></name><name><surname>Ford</surname> <given-names>CB</given-names></name><name><surname>Coleman</surname> <given-names>MT</given-names></name><name><surname>Myers</surname> <given-names>AJ</given-names></name><name><surname>Gawande</surname> <given-names>R</given-names></name><name><surname>Ioerger</surname> <given-names>T</given-names></name><name><surname>Sacchettini</surname> <given-names>J</given-names></name><name><surname>Fortune</surname> <given-names>SM</given-names></name><name><surname>Flynn</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Sterilization of granulomas is common in active and latent tuberculosis despite within-host variability in bacterial killing</article-title><source>Nature Medicine</source><volume>20</volume><fpage>75</fpage><lpage>79</lpage><pub-id pub-id-type="doi">10.1038/nm.3412</pub-id><pub-id pub-id-type="pmid">24336248</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Llewelyn</surname> <given-names>MJ</given-names></name><name><surname>Fitzpatrick</surname> <given-names>JM</given-names></name><name><surname>Darwin</surname> <given-names>E</given-names></name><name><surname>Tonkin-Crine</surname> <given-names>S</given-names></name><name><surname>Gorton</surname> <given-names>C</given-names></name><name><surname>Paul</surname> <given-names>J</given-names></name><name><surname>Peto</surname> <given-names>TEA</given-names></name><name><surname>Yardley</surname> <given-names>L</given-names></name><name><surname>Hopkins</surname> <given-names>S</given-names></name><name><surname>Walker</surname> <given-names>AS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The antibiotic course has had its day</article-title><source>BMJ</source><volume>358</volume><elocation-id>j3418</elocation-id><pub-id pub-id-type="doi">10.1136/bmj.j3418</pub-id><pub-id pub-id-type="pmid">28747365</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Marvig</surname> <given-names>RL</given-names></name><name><surname>Sommer</surname> <given-names>LM</given-names></name><name><surname>Molin</surname> <given-names>S</given-names></name><name><surname>Johansen</surname> <given-names>HK</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Convergent evolution and adaptation of <italic>Pseudomonas aeruginosa</italic> within patients with cystic fibrosis</article-title><source>Nature Genetics</source><volume>47</volume><fpage>57</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1038/ng.3148</pub-id><pub-id pub-id-type="pmid">25401299</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>McKinney</surname> <given-names>W</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Data structures for statistical computing in Python</article-title><conf-name>Proceedings of the 9th Python in Science Conference</conf-name><fpage>51</fpage><lpage>56</lpage></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minias</surname> <given-names>A</given-names></name><name><surname>Minias</surname> <given-names>P</given-names></name><name><surname>Czubat</surname> <given-names>B</given-names></name><name><surname>Dziadek</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Purifying selective pressure suggests the functionality of a vitamin B12 biosynthesis pathway in a global population of Mycobacterium tuberculosis</article-title><source>Genome Biology and Evolution</source><volume>10</volume><fpage>2326</fpage><lpage>2337</lpage><pub-id pub-id-type="doi">10.1093/gbe/evy153</pub-id><pub-id pub-id-type="pmid">30060031</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nair</surname> <given-names>S</given-names></name><name><surname>Ramaswamy</surname> <given-names>PA</given-names></name><name><surname>Ghosh</surname> <given-names>S</given-names></name><name><surname>Joshi</surname> <given-names>DC</given-names></name><name><surname>Pathak</surname> <given-names>N</given-names></name><name><surname>Siddiqui</surname> <given-names>I</given-names></name><name><surname>Sharma</surname> <given-names>P</given-names></name><name><surname>Hasnain</surname> <given-names>SE</given-names></name><name><surname>Mande</surname> <given-names>SC</given-names></name><name><surname>Mukhopadhyay</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The PPE18 of <italic>Mycobacterium tuberculosis</italic> Interacts with TLR2 and Activates IL-10 Induction in Macrophage</article-title><source>J Immunol.</source><volume>183</volume><fpage>6269</fpage><lpage>6281</lpage><pub-id pub-id-type="doi">10.4049/jimmunol.0901367</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Namouchi</surname> <given-names>A</given-names></name><name><surname>Didelot</surname> <given-names>X</given-names></name><name><surname>Schöck</surname> <given-names>U</given-names></name><name><surname>Gicquel</surname> <given-names>B</given-names></name><name><surname>Rocha</surname> <given-names>EP</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>After the bottleneck: genome-wide diversification of the Mycobacterium tuberculosis complex by mutation, recombination, and natural selection</article-title><source>Genome Research</source><volume>22</volume><fpage>721</fpage><lpage>734</lpage><pub-id pub-id-type="doi">10.1101/gr.129544.111</pub-id><pub-id pub-id-type="pmid">22377718</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nimmo</surname> <given-names>C</given-names></name><name><surname>Shaw</surname> <given-names>LP</given-names></name><name><surname>Doyle</surname> <given-names>R</given-names></name><name><surname>Williams</surname> <given-names>R</given-names></name><name><surname>Brien</surname> <given-names>K</given-names></name><name><surname>Burgess</surname> <given-names>C</given-names></name><name><surname>Breuer</surname> <given-names>J</given-names></name><name><surname>Balloux</surname> <given-names>F</given-names></name><name><surname>Pym</surname> <given-names>AS</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Whole genome sequencing Mycobacterium tuberculosis directly from sputum identifies more genetic diversity than sequencing from culture</article-title><source>BMC Genomics</source><volume>20</volume><elocation-id>389</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-019-5782-2</pub-id><pub-id pub-id-type="pmid">31109296</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Overbeek</surname> <given-names>R</given-names></name><name><surname>Olson</surname> <given-names>R</given-names></name><name><surname>Pusch</surname> <given-names>GD</given-names></name><name><surname>Olsen</surname> <given-names>GJ</given-names></name><name><surname>Davis</surname> <given-names>JJ</given-names></name><name><surname>Disz</surname> <given-names>T</given-names></name><name><surname>Edwards</surname> <given-names>RA</given-names></name><name><surname>Gerdes</surname> <given-names>S</given-names></name><name><surname>Parrello</surname> <given-names>B</given-names></name><name><surname>Shukla</surname> <given-names>M</given-names></name><name><surname>Vonstein</surname> <given-names>V</given-names></name><name><surname>Wattam</surname> <given-names>AR</given-names></name><name><surname>Xia</surname> <given-names>F</given-names></name><name><surname>Stevens</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The SEED and the rapid annotation of microbial genomes using subsystems technology (RAST)</article-title><source>Nucleic Acids Research</source><volume>42</volume><fpage>D206</fpage><lpage>D214</lpage><pub-id pub-id-type="doi">10.1093/nar/gkt1226</pub-id><pub-id pub-id-type="pmid">24293654</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pai</surname> <given-names>M</given-names></name><name><surname>Behr</surname> <given-names>MA</given-names></name><name><surname>Dowdy</surname> <given-names>D</given-names></name><name><surname>Dheda</surname> <given-names>K</given-names></name><name><surname>Divangahi</surname> <given-names>M</given-names></name><name><surname>Boehme</surname> <given-names>CC</given-names></name><name><surname>Ginsberg</surname> <given-names>A</given-names></name><name><surname>Swaminathan</surname> <given-names>S</given-names></name><name><surname>Spigelman</surname> <given-names>M</given-names></name><name><surname>Getahun</surname> <given-names>H</given-names></name><name><surname>Menzies</surname> <given-names>D</given-names></name><name><surname>Raviglione</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Tuberculosis</article-title><source>Nature Reviews Disease Primers</source><volume>2</volume><elocation-id>16076</elocation-id><pub-id pub-id-type="doi">10.1038/nrdp.2016.76</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F</given-names></name><name><surname>Varoquaux</surname> <given-names>G</given-names></name><name><surname>Gramfort</surname> <given-names>A</given-names></name><name><surname>Michel</surname> <given-names>V</given-names></name><name><surname>Thirion</surname> <given-names>B</given-names></name><name><surname>Grisel</surname> <given-names>O</given-names></name><name><surname>Blondel</surname> <given-names>M</given-names></name><name><surname>Prettenhofer</surname> <given-names>P</given-names></name><name><surname>Weiss</surname> <given-names>R</given-names></name><name><surname>Dubourg</surname> <given-names>V</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Scikit-learn: machine learning in Python</article-title><source>J Mach Learn Res</source><volume>12</volume><fpage>2825</fpage><lpage>2830</lpage></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Perez</surname> <given-names>F</given-names></name><name><surname>Granger</surname> <given-names>BE</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>IPython: a system for interactive scientific computing</article-title><source>Computing in Science &amp; Engineering</source><volume>9</volume><fpage>21</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2007.53</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pethe</surname> <given-names>K</given-names></name><name><surname>Sequeira</surname> <given-names>PC</given-names></name><name><surname>Agarwalla</surname> <given-names>S</given-names></name><name><surname>Rhee</surname> <given-names>K</given-names></name><name><surname>Kuhen</surname> <given-names>K</given-names></name><name><surname>Phong</surname> <given-names>WY</given-names></name><name><surname>Patel</surname> <given-names>V</given-names></name><name><surname>Beer</surname> <given-names>D</given-names></name><name><surname>Walker</surname> <given-names>JR</given-names></name><name><surname>Duraiswamy</surname> <given-names>J</given-names></name><name><surname>Jiricek</surname> <given-names>J</given-names></name><name><surname>Keller</surname> <given-names>TH</given-names></name><name><surname>Chatterjee</surname> <given-names>A</given-names></name><name><surname>Tan</surname> <given-names>MP</given-names></name><name><surname>Ujjini</surname> <given-names>M</given-names></name><name><surname>Rao</surname> <given-names>SP</given-names></name><name><surname>Camacho</surname> <given-names>L</given-names></name><name><surname>Bifani</surname> <given-names>P</given-names></name><name><surname>Mak</surname> <given-names>PA</given-names></name><name><surname>Ma</surname> <given-names>I</given-names></name><name><surname>Barnes</surname> <given-names>SW</given-names></name><name><surname>Chen</surname> <given-names>Z</given-names></name><name><surname>Plouffe</surname> <given-names>D</given-names></name><name><surname>Thayalan</surname> <given-names>P</given-names></name><name><surname>Ng</surname> <given-names>SH</given-names></name><name><surname>Au</surname> <given-names>M</given-names></name><name><surname>Lee</surname> <given-names>BH</given-names></name><name><surname>Tan</surname> <given-names>BH</given-names></name><name><surname>Ravindran</surname> <given-names>S</given-names></name><name><surname>Nanjundappa</surname> <given-names>M</given-names></name><name><surname>Lin</surname> <given-names>X</given-names></name><name><surname>Goh</surname> <given-names>A</given-names></name><name><surname>Lakshminarayana</surname> <given-names>SB</given-names></name><name><surname>Shoen</surname> <given-names>C</given-names></name><name><surname>Cynamon</surname> <given-names>M</given-names></name><name><surname>Kreiswirth</surname> <given-names>B</given-names></name><name><surname>Dartois</surname> <given-names>V</given-names></name><name><surname>Peters</surname> <given-names>EC</given-names></name><name><surname>Glynne</surname> <given-names>R</given-names></name><name><surname>Brenner</surname> <given-names>S</given-names></name><name><surname>Dick</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A chemical genetic screen in Mycobacterium tuberculosis identifies carbon-source-dependent growth inhibitors devoid of in vivo efficacy</article-title><source>Nature Communications</source><volume>1</volume><elocation-id>57</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms1060</pub-id><pub-id pub-id-type="pmid">20975714</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Phelan</surname> <given-names>JE</given-names></name><name><surname>Coll</surname> <given-names>F</given-names></name><name><surname>Bergval</surname> <given-names>I</given-names></name><name><surname>Anthony</surname> <given-names>RM</given-names></name><name><surname>Warren</surname> <given-names>R</given-names></name><name><surname>Sampson</surname> <given-names>SL</given-names></name><name><surname>Gey van Pittius</surname> <given-names>NC</given-names></name><name><surname>Glynn</surname> <given-names>JR</given-names></name><name><surname>Crampin</surname> <given-names>AC</given-names></name><name><surname>Alves</surname> <given-names>A</given-names></name><name><surname>Bessa</surname> <given-names>TB</given-names></name><name><surname>Campino</surname> <given-names>S</given-names></name><name><surname>Dheda</surname> <given-names>K</given-names></name><name><surname>Grandjean</surname> <given-names>L</given-names></name><name><surname>Hasan</surname> <given-names>R</given-names></name><name><surname>Hasan</surname> <given-names>Z</given-names></name><name><surname>Miranda</surname> <given-names>A</given-names></name><name><surname>Moore</surname> <given-names>D</given-names></name><name><surname>Panaiotov</surname> <given-names>S</given-names></name><name><surname>Perdigao</surname> <given-names>J</given-names></name><name><surname>Portugal</surname> <given-names>I</given-names></name><name><surname>Sheen</surname> <given-names>P</given-names></name><name><surname>de Oliveira Sousa</surname> <given-names>E</given-names></name><name><surname>Streicher</surname> <given-names>EM</given-names></name><name><surname>van Helden</surname> <given-names>PD</given-names></name><name><surname>Viveiros</surname> <given-names>M</given-names></name><name><surname>Hibberd</surname> <given-names>ML</given-names></name><name><surname>Pain</surname> <given-names>A</given-names></name><name><surname>McNerney</surname> <given-names>R</given-names></name><name><surname>Clark</surname> <given-names>TG</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Recombination in pe/ppe genes contributes to genetic variation in Mycobacterium tuberculosis lineages</article-title><source>BMC Genomics</source><volume>17</volume><elocation-id>151</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-016-2467-y</pub-id><pub-id pub-id-type="pmid">26923687</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Podinovskaia</surname> <given-names>M</given-names></name><name><surname>Lee</surname> <given-names>W</given-names></name><name><surname>Caldwell</surname> <given-names>S</given-names></name><name><surname>Russell</surname> <given-names>DG</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Infection of macrophages with Mycobacterium tuberculosis induces global modifications to phagosomal function</article-title><source>Cellular Microbiology</source><volume>15</volume><fpage>843</fpage><lpage>859</lpage><pub-id pub-id-type="doi">10.1111/cmi.12092</pub-id><pub-id pub-id-type="pmid">23253353</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rhoads</surname> <given-names>A</given-names></name><name><surname>Au</surname> <given-names>KF</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>PacBio sequencing and its applications</article-title><source>Genomics, Proteomics &amp; Bioinformatics</source><volume>13</volume><fpage>278</fpage><lpage>289</lpage><pub-id pub-id-type="doi">10.1016/j.gpb.2015.08.002</pub-id><pub-id pub-id-type="pmid">26542840</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rowley</surname> <given-names>CA</given-names></name><name><surname>Kendall</surname> <given-names>MM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>To B12 or not to B12: five questions on the role of cobalamin in host-microbial interactions</article-title><source>PLOS Pathogens</source><volume>15</volume><elocation-id>e1007479</elocation-id><pub-id pub-id-type="doi">10.1371/journal.ppat.1007479</pub-id><pub-id pub-id-type="pmid">30605490</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Salaemae</surname> <given-names>W</given-names></name><name><surname>Azhar</surname> <given-names>A</given-names></name><name><surname>Booker</surname> <given-names>GW</given-names></name><name><surname>Polyak</surname> <given-names>SW</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Biotin biosynthesis in Mycobacterium tuberculosis: physiology, biochemistry and molecular intervention</article-title><source>Protein &amp; Cell</source><volume>2</volume><fpage>691</fpage><lpage>695</lpage><pub-id pub-id-type="doi">10.1007/s13238-011-1100-8</pub-id><pub-id pub-id-type="pmid">21976058</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sassetti</surname> <given-names>CM</given-names></name><name><surname>Boyd</surname> <given-names>DH</given-names></name><name><surname>Rubin</surname> <given-names>EJ</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Genes required for mycobacterial growth defined by high density mutagenesis</article-title><source>Molecular Microbiology</source><volume>48</volume><fpage>77</fpage><lpage>84</lpage><pub-id pub-id-type="doi">10.1046/j.1365-2958.2003.03425.x</pub-id><pub-id pub-id-type="pmid">12657046</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sassetti</surname> <given-names>CM</given-names></name><name><surname>Rubin</surname> <given-names>EJ</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Genetic requirements for mycobacterial survival during infection</article-title><source>PNAS</source><volume>100</volume><fpage>12989</fpage><lpage>12994</lpage><pub-id pub-id-type="doi">10.1073/pnas.2134250100</pub-id><pub-id pub-id-type="pmid">14569030</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmieder</surname> <given-names>R</given-names></name><name><surname>Edwards</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Quality control and preprocessing of metagenomic datasets</article-title><source>Bioinformatics</source><volume>27</volume><fpage>863</fpage><lpage>864</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr026</pub-id><pub-id pub-id-type="pmid">21278185</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Seabold</surname> <given-names>S</given-names></name><name><surname>Perktold</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Econometric and statistical modeling with Python</article-title><conf-name>Proceedings of He 9th Python in Science Conference Statsmodels</conf-name></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stucki</surname> <given-names>D</given-names></name><name><surname>Brites</surname> <given-names>D</given-names></name><name><surname>Jeljeli</surname> <given-names>L</given-names></name><name><surname>Coscolla</surname> <given-names>M</given-names></name><name><surname>Liu</surname> <given-names>Q</given-names></name><name><surname>Trauner</surname> <given-names>A</given-names></name><name><surname>Fenner</surname> <given-names>L</given-names></name><name><surname>Rutaihwa</surname> <given-names>L</given-names></name><name><surname>Borrell</surname> <given-names>S</given-names></name><name><surname>Luo</surname> <given-names>T</given-names></name><name><surname>Gao</surname> <given-names>Q</given-names></name><name><surname>Kato-Maeda</surname> <given-names>M</given-names></name><name><surname>Ballif</surname> <given-names>M</given-names></name><name><surname>Egger</surname> <given-names>M</given-names></name><name><surname>Macedo</surname> <given-names>R</given-names></name><name><surname>Mardassi</surname> <given-names>H</given-names></name><name><surname>Moreno</surname> <given-names>M</given-names></name><name><surname>Tudo Vilanova</surname> <given-names>G</given-names></name><name><surname>Fyfe</surname> <given-names>J</given-names></name><name><surname>Globan</surname> <given-names>M</given-names></name><name><surname>Thomas</surname> <given-names>J</given-names></name><name><surname>Jamieson</surname> <given-names>F</given-names></name><name><surname>Guthrie</surname> <given-names>JL</given-names></name><name><surname>Asante-Poku</surname> <given-names>A</given-names></name><name><surname>Yeboah-Manu</surname> <given-names>D</given-names></name><name><surname>Wampande</surname> <given-names>E</given-names></name><name><surname>Ssengooba</surname> <given-names>W</given-names></name><name><surname>Joloba</surname> <given-names>M</given-names></name><name><surname>Henry Boom</surname> <given-names>W</given-names></name><name><surname>Basu</surname> <given-names>I</given-names></name><name><surname>Bower</surname> <given-names>J</given-names></name><name><surname>Saraiva</surname> <given-names>M</given-names></name><name><surname>Vaconcellos</surname> <given-names>SEG</given-names></name><name><surname>Suffys</surname> <given-names>P</given-names></name><name><surname>Koch</surname> <given-names>A</given-names></name><name><surname>Wilkinson</surname> <given-names>R</given-names></name><name><surname>Gail-Bekker</surname> <given-names>L</given-names></name><name><surname>Malla</surname> <given-names>B</given-names></name><name><surname>Ley</surname> <given-names>SD</given-names></name><name><surname>Beck</surname> <given-names>HP</given-names></name><name><surname>de Jong</surname> <given-names>BC</given-names></name><name><surname>Toit</surname> <given-names>K</given-names></name><name><surname>Sanchez-Padilla</surname> <given-names>E</given-names></name><name><surname>Bonnet</surname> <given-names>M</given-names></name><name><surname>Gil-Brusola</surname> <given-names>A</given-names></name><name><surname>Frank</surname> <given-names>M</given-names></name><name><surname>Penlap Beng</surname> <given-names>VN</given-names></name><name><surname>Eisenach</surname> <given-names>K</given-names></name><name><surname>Alani</surname> <given-names>I</given-names></name><name><surname>Wangui Ndung'u</surname> <given-names>P</given-names></name><name><surname>Revathi</surname> <given-names>G</given-names></name><name><surname>Gehre</surname> <given-names>F</given-names></name><name><surname>Akter</surname> <given-names>S</given-names></name><name><surname>Ntoumi</surname> <given-names>F</given-names></name><name><surname>Stewart-Isherwood</surname> <given-names>L</given-names></name><name><surname>Ntinginya</surname> <given-names>NE</given-names></name><name><surname>Rachow</surname> <given-names>A</given-names></name><name><surname>Hoelscher</surname> <given-names>M</given-names></name><name><surname>Cirillo</surname> <given-names>DM</given-names></name><name><surname>Skenders</surname> <given-names>G</given-names></name><name><surname>Hoffner</surname> <given-names>S</given-names></name><name><surname>Bakonyte</surname> <given-names>D</given-names></name><name><surname>Stakenas</surname> <given-names>P</given-names></name><name><surname>Diel</surname> <given-names>R</given-names></name><name><surname>Crudu</surname> <given-names>V</given-names></name><name><surname>Moldovan</surname> <given-names>O</given-names></name><name><surname>Al-Hajoj</surname> <given-names>S</given-names></name><name><surname>Otero</surname> <given-names>L</given-names></name><name><surname>Barletta</surname> <given-names>F</given-names></name><name><surname>Jane Carter</surname> <given-names>E</given-names></name><name><surname>Diero</surname> <given-names>L</given-names></name><name><surname>Supply</surname> <given-names>P</given-names></name><name><surname>Comas</surname> <given-names>I</given-names></name><name><surname>Niemann</surname> <given-names>S</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mycobacterium tuberculosis lineage 4 comprises globally distributed and geographically restricted sublineages</article-title><source>Nature Genetics</source><volume>48</volume><fpage>1535</fpage><lpage>1543</lpage><pub-id pub-id-type="doi">10.1038/ng.3704</pub-id><pub-id pub-id-type="pmid">27798628</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>H</given-names></name><name><surname>Zhang</surname> <given-names>Z</given-names></name><name><surname>Liu</surname> <given-names>Z</given-names></name><name><surname>Peng</surname> <given-names>B</given-names></name><name><surname>Kong</surname> <given-names>C</given-names></name><name><surname>Wang</surname> <given-names>H</given-names></name><name><surname>Zhang</surname> <given-names>Z</given-names></name><name><surname>Xu</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mycobacterium tuberculosis PPE60 antigen drives Th1/Th17 responses via Toll-like receptor 2–dependent maturation of dendritic cells</article-title><source>Journal of Biological Chemistry</source><volume>293</volume><fpage>10287</fpage><lpage>10302</lpage><pub-id pub-id-type="doi">10.1074/jbc.RA118.001696</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>G</given-names></name><name><surname>Luo</surname> <given-names>T</given-names></name><name><surname>Yang</surname> <given-names>C</given-names></name><name><surname>Dong</surname> <given-names>X</given-names></name><name><surname>Li</surname> <given-names>J</given-names></name><name><surname>Zhu</surname> <given-names>Y</given-names></name><name><surname>Zheng</surname> <given-names>H</given-names></name><name><surname>Tian</surname> <given-names>W</given-names></name><name><surname>Wang</surname> <given-names>S</given-names></name><name><surname>Barry</surname> <given-names>CE</given-names></name><name><surname>Mei</surname> <given-names>J</given-names></name><name><surname>Gao</surname> <given-names>Q</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Dynamic population changes in Mycobacterium tuberculosis during acquisition and fixation of drug resistance in patients</article-title><source>The Journal of Infectious Diseases</source><volume>206</volume><fpage>1724</fpage><lpage>1733</lpage><pub-id pub-id-type="doi">10.1093/infdis/jis601</pub-id><pub-id pub-id-type="pmid">22984115</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tait</surname> <given-names>DR</given-names></name><name><surname>Hatherill</surname> <given-names>M</given-names></name><name><surname>Van Der Meeren</surname> <given-names>O</given-names></name><name><surname>Ginsberg</surname> <given-names>AM</given-names></name><name><surname>Van Brakel</surname> <given-names>E</given-names></name><name><surname>Salaun</surname> <given-names>B</given-names></name><name><surname>Scriba</surname> <given-names>TJ</given-names></name><name><surname>Akite</surname> <given-names>EJ</given-names></name><name><surname>Ayles</surname> <given-names>HM</given-names></name><name><surname>Bollaerts</surname> <given-names>A</given-names></name><name><surname>Demoitié</surname> <given-names>MA</given-names></name><name><surname>Diacon</surname> <given-names>A</given-names></name><name><surname>Evans</surname> <given-names>TG</given-names></name><name><surname>Gillard</surname> <given-names>P</given-names></name><name><surname>Hellström</surname> <given-names>E</given-names></name><name><surname>Innes</surname> <given-names>JC</given-names></name><name><surname>Lempicki</surname> <given-names>M</given-names></name><name><surname>Malahleha</surname> <given-names>M</given-names></name><name><surname>Martinson</surname> <given-names>N</given-names></name><name><surname>Mesia Vela</surname> <given-names>D</given-names></name><name><surname>Muyoyeta</surname> <given-names>M</given-names></name><name><surname>Nduba</surname> <given-names>V</given-names></name><name><surname>Pascal</surname> <given-names>TG</given-names></name><name><surname>Tameris</surname> <given-names>M</given-names></name><name><surname>Thienemann</surname> <given-names>F</given-names></name><name><surname>Wilkinson</surname> <given-names>RJ</given-names></name><name><surname>Roman</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Final analysis of a trial of M72/AS01<sub>E</sub>Vaccin<sub>e</sub> to prevent tuberculosis</article-title><source>New England Journal of Medicine</source><volume>381</volume><fpage>2429</fpage><lpage>2439</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa1909953</pub-id><pub-id pub-id-type="pmid">31661198</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thorvaldsdóttir</surname> <given-names>H</given-names></name><name><surname>Robinson</surname> <given-names>JT</given-names></name><name><surname>Mesirov</surname> <given-names>JP</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Integrative genomics viewer (IGV): high-performance genomics data visualization and exploration</article-title><source>Briefings in Bioinformatics</source><volume>14</volume><fpage>178</fpage><lpage>192</lpage><pub-id pub-id-type="doi">10.1093/bib/bbs017</pub-id><pub-id pub-id-type="pmid">22517427</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tientcheu</surname> <given-names>LD</given-names></name><name><surname>Koch</surname> <given-names>A</given-names></name><name><surname>Ndengane</surname> <given-names>M</given-names></name><name><surname>Andoseh</surname> <given-names>G</given-names></name><name><surname>Kampmann</surname> <given-names>B</given-names></name><name><surname>Wilkinson</surname> <given-names>RJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Immunological consequences of strain variation within the Mycobacterium tuberculosis complex</article-title><source>European Journal of Immunology</source><volume>47</volume><fpage>432</fpage><lpage>445</lpage><pub-id pub-id-type="doi">10.1002/eji.201646562</pub-id><pub-id pub-id-type="pmid">28150302</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Trauner</surname> <given-names>A</given-names></name><name><surname>Liu</surname> <given-names>Q</given-names></name><name><surname>Via</surname> <given-names>LE</given-names></name><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Ruan</surname> <given-names>X</given-names></name><name><surname>Liang</surname> <given-names>L</given-names></name><name><surname>Shi</surname> <given-names>H</given-names></name><name><surname>Chen</surname> <given-names>Y</given-names></name><name><surname>Wang</surname> <given-names>Z</given-names></name><name><surname>Liang</surname> <given-names>R</given-names></name><name><surname>Zhang</surname> <given-names>W</given-names></name><name><surname>Wei</surname> <given-names>W</given-names></name><name><surname>Gao</surname> <given-names>J</given-names></name><name><surname>Sun</surname> <given-names>G</given-names></name><name><surname>Brites</surname> <given-names>D</given-names></name><name><surname>England</surname> <given-names>K</given-names></name><name><surname>Zhang</surname> <given-names>G</given-names></name><name><surname>Gagneux</surname> <given-names>S</given-names></name><name><surname>Barry</surname> <given-names>CE</given-names></name><name><surname>Gao</surname> <given-names>Q</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The within-host population dynamics of Mycobacterium tuberculosis vary with treatment efficacy</article-title><source>Genome Biology</source><volume>18</volume><elocation-id>71</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-017-1196-0</pub-id><pub-id pub-id-type="pmid">28424085</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van der Walt</surname> <given-names>S</given-names></name><name><surname>Colbert</surname> <given-names>SC</given-names></name><name><surname>Varoquaux</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The NumPy array: a structure for efficient numerical computation</article-title><source>Computing in Science &amp; Engineering</source><volume>13</volume><fpage>22</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2011.37</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Vargas</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>in-host-Mtbc-dynamics</data-title><source>Github</source><version designator="36e2701">36e2701</version><ext-link ext-link-type="uri" xlink:href="https://github.com/farhat-lab/in-host-Mtbc-dynamics">https://github.com/farhat-lab/in-host-Mtbc-dynamics</ext-link></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vargas</surname> <given-names>R</given-names></name><name><surname>Farhat</surname> <given-names>MR</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Antibiotic treatment and selection for <italic>glpK</italic> mutations in patients with active tuberculosis disease</article-title><source>PNAS</source><volume>117</volume><fpage>3910</fpage><lpage>3912</lpage><pub-id pub-id-type="doi">10.1073/pnas.1920788117</pub-id><pub-id pub-id-type="pmid">32075922</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname> <given-names>P</given-names></name><name><surname>Gommers</surname> <given-names>R</given-names></name><name><surname>Oliphant</surname> <given-names>TE</given-names></name><name><surname>Haberland</surname> <given-names>M</given-names></name><name><surname>Reddy</surname> <given-names>T</given-names></name><name><surname>Cournapeau</surname> <given-names>D</given-names></name><name><surname>Burovski</surname> <given-names>E</given-names></name><name><surname>Peterson</surname> <given-names>P</given-names></name><name><surname>Weckesser</surname> <given-names>W</given-names></name><name><surname>Bright</surname> <given-names>J</given-names></name><name><surname>van der Walt</surname> <given-names>SJ</given-names></name><name><surname>Brett</surname> <given-names>M</given-names></name><name><surname>Wilson</surname> <given-names>J</given-names></name><name><surname>Millman</surname> <given-names>KJ</given-names></name><name><surname>Mayorov</surname> <given-names>N</given-names></name><name><surname>Nelson</surname> <given-names>ARJ</given-names></name><name><surname>Jones</surname> <given-names>E</given-names></name><name><surname>Kern</surname> <given-names>R</given-names></name><name><surname>Larson</surname> <given-names>E</given-names></name><name><surname>Carey</surname> <given-names>CJ</given-names></name><name><surname>Polat</surname> <given-names>I</given-names></name><name><surname>Feng</surname> <given-names>Y</given-names></name><name><surname>Moore</surname> <given-names>EW</given-names></name><name><surname>VanderPlas</surname> <given-names>J</given-names></name><name><surname>Laxalde</surname> <given-names>D</given-names></name><name><surname>Perktold</surname> <given-names>J</given-names></name><name><surname>Cimrman</surname> <given-names>R</given-names></name><name><surname>Henriksen</surname> <given-names>I</given-names></name><name><surname>Quintero</surname> <given-names>EA</given-names></name><name><surname>Harris</surname> <given-names>CR</given-names></name><name><surname>Archibald</surname> <given-names>AM</given-names></name><name><surname>Ribeiro</surname> <given-names>AH</given-names></name><name><surname>Pedregosa</surname> <given-names>F</given-names></name><name><surname>van Mulbregt</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SciPy 1.0: fundamental algorithms for scientific computing in Python</article-title><source>Nature Methods</source><volume>17</volume><fpage>261</fpage><lpage>272</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0686-2</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vita</surname> <given-names>R</given-names></name><name><surname>Overton</surname> <given-names>JA</given-names></name><name><surname>Greenbaum</surname> <given-names>JA</given-names></name><name><surname>Ponomarenko</surname> <given-names>J</given-names></name><name><surname>Clark</surname> <given-names>JD</given-names></name><name><surname>Cantrell</surname> <given-names>JR</given-names></name><name><surname>Wheeler</surname> <given-names>DK</given-names></name><name><surname>Gabbard</surname> <given-names>JL</given-names></name><name><surname>Hix</surname> <given-names>D</given-names></name><name><surname>Sette</surname> <given-names>A</given-names></name><name><surname>Peters</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The immune epitope database (IEDB) 3.0</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>D405</fpage><lpage>D412</lpage><pub-id pub-id-type="doi">10.1093/nar/gku938</pub-id><pub-id pub-id-type="pmid">25300482</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Votintseva</surname> <given-names>AA</given-names></name><name><surname>Bradley</surname> <given-names>P</given-names></name><name><surname>Pankhurst</surname> <given-names>L</given-names></name><name><surname>Del Ojo Elias</surname> <given-names>C</given-names></name><name><surname>Loose</surname> <given-names>M</given-names></name><name><surname>Nilgiriwala</surname> <given-names>K</given-names></name><name><surname>Chatterjee</surname> <given-names>A</given-names></name><name><surname>Smith</surname> <given-names>EG</given-names></name><name><surname>Sanderson</surname> <given-names>N</given-names></name><name><surname>Walker</surname> <given-names>TM</given-names></name><name><surname>Morgan</surname> <given-names>MR</given-names></name><name><surname>Wyllie</surname> <given-names>DH</given-names></name><name><surname>Walker</surname> <given-names>AS</given-names></name><name><surname>Peto</surname> <given-names>TEA</given-names></name><name><surname>Crook</surname> <given-names>DW</given-names></name><name><surname>Iqbal</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Same-Day diagnostic and surveillance data for tuberculosis via Whole-Genome sequencing of direct respiratory samples</article-title><source>Journal of Clinical Microbiology</source><volume>55</volume><fpage>1285</fpage><lpage>1298</lpage><pub-id pub-id-type="doi">10.1128/JCM.02483-16</pub-id><pub-id pub-id-type="pmid">28275074</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname> <given-names>TM</given-names></name><name><surname>Ip</surname> <given-names>CL</given-names></name><name><surname>Harrell</surname> <given-names>RH</given-names></name><name><surname>Evans</surname> <given-names>JT</given-names></name><name><surname>Kapatai</surname> <given-names>G</given-names></name><name><surname>Dedicoat</surname> <given-names>MJ</given-names></name><name><surname>Eyre</surname> <given-names>DW</given-names></name><name><surname>Wilson</surname> <given-names>DJ</given-names></name><name><surname>Hawkey</surname> <given-names>PM</given-names></name><name><surname>Crook</surname> <given-names>DW</given-names></name><name><surname>Parkhill</surname> <given-names>J</given-names></name><name><surname>Harris</surname> <given-names>D</given-names></name><name><surname>Walker</surname> <given-names>AS</given-names></name><name><surname>Bowden</surname> <given-names>R</given-names></name><name><surname>Monk</surname> <given-names>P</given-names></name><name><surname>Smith</surname> <given-names>EG</given-names></name><name><surname>Peto</surname> <given-names>TE</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Whole-genome sequencing to delineate Mycobacterium tuberculosis outbreaks: a retrospective observational study</article-title><source>The Lancet Infectious Diseases</source><volume>13</volume><fpage>137</fpage><lpage>146</lpage><pub-id pub-id-type="doi">10.1016/S1473-3099(12)70277-3</pub-id><pub-id pub-id-type="pmid">23158499</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname> <given-names>BJ</given-names></name><name><surname>Abeel</surname> <given-names>T</given-names></name><name><surname>Shea</surname> <given-names>T</given-names></name><name><surname>Priest</surname> <given-names>M</given-names></name><name><surname>Abouelliel</surname> <given-names>A</given-names></name><name><surname>Sakthikumar</surname> <given-names>S</given-names></name><name><surname>Cuomo</surname> <given-names>CA</given-names></name><name><surname>Zeng</surname> <given-names>Q</given-names></name><name><surname>Wortman</surname> <given-names>J</given-names></name><name><surname>Young</surname> <given-names>SK</given-names></name><name><surname>Earl</surname> <given-names>AM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Pilon: an integrated tool for comprehensive microbial variant detection and genome assembly improvement</article-title><source>PLOS ONE</source><volume>9</volume><elocation-id>e112963</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0112963</pub-id><pub-id pub-id-type="pmid">25409509</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Witney</surname> <given-names>AA</given-names></name><name><surname>Bateson</surname> <given-names>AL</given-names></name><name><surname>Jindani</surname> <given-names>A</given-names></name><name><surname>Phillips</surname> <given-names>PP</given-names></name><name><surname>Coleman</surname> <given-names>D</given-names></name><name><surname>Stoker</surname> <given-names>NG</given-names></name><name><surname>Butcher</surname> <given-names>PD</given-names></name><name><surname>McHugh</surname> <given-names>TD</given-names></name><collab>RIFAQUIN Study Team</collab></person-group><year iso-8601-date="2017">2017</year><article-title>Use of whole-genome sequencing to distinguish relapse from reinfection in a completed tuberculosis clinical trial</article-title><source>BMC Medicine</source><volume>15</volume><elocation-id>71</elocation-id><pub-id pub-id-type="doi">10.1186/s12916-017-0834-4</pub-id><pub-id pub-id-type="pmid">28351427</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wood</surname> <given-names>DE</given-names></name><name><surname>Salzberg</surname> <given-names>SL</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Kraken: ultrafast metagenomic sequence classification using exact alignments</article-title><source>Genome Biology</source><volume>15</volume><elocation-id>R46</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2014-15-3-r46</pub-id><pub-id pub-id-type="pmid">24580807</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="report"><person-group person-group-type="author"><collab>World Health Organization</collab></person-group><year iso-8601-date="2018">2018</year><source>Global Tuberculosis Report 2018</source><publisher-name>World Health Organization</publisher-name></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wyllie</surname> <given-names>DH</given-names></name><name><surname>Robinson</surname> <given-names>E</given-names></name><name><surname>Peto</surname> <given-names>T</given-names></name><name><surname>Crook</surname> <given-names>DW</given-names></name><name><surname>Ajileye</surname> <given-names>A</given-names></name><name><surname>Rathod</surname> <given-names>P</given-names></name><name><surname>Allen</surname> <given-names>R</given-names></name><name><surname>Jarrett</surname> <given-names>L</given-names></name><name><surname>Smith</surname> <given-names>EG</given-names></name><name><surname>Walker</surname> <given-names>AS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Identifying mixed <italic>Mycobacterium tuberculosis</italic> Infection and Laboratory Cross-Contamination during Mycobacterial Sequencing Programs</article-title><source>Journal of Clinical Microbiology</source><volume>56</volume><elocation-id>e00923-18</elocation-id><pub-id pub-id-type="doi">10.1128/JCM.00923-18</pub-id><pub-id pub-id-type="pmid">30209183</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Y</given-names></name><name><surname>Liu</surname> <given-names>F</given-names></name><name><surname>Chen</surname> <given-names>S</given-names></name><name><surname>Wu</surname> <given-names>J</given-names></name><name><surname>Hu</surname> <given-names>Y</given-names></name><name><surname>Zhu</surname> <given-names>B</given-names></name><name><surname>Sun</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>In vivo evolution of drug-resistant Mycobacterium tuberculosis in patients during long-term treatment</article-title><source>BMC Genomics</source><volume>19</volume><elocation-id>640</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-018-5010-5</pub-id><pub-id pub-id-type="pmid">30157763</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D</given-names></name><name><surname>Gomez</surname> <given-names>JE</given-names></name><name><surname>Chien</surname> <given-names>JY</given-names></name><name><surname>Haseley</surname> <given-names>N</given-names></name><name><surname>Desjardins</surname> <given-names>CA</given-names></name><name><surname>Earl</surname> <given-names>AM</given-names></name><name><surname>Hsueh</surname> <given-names>PR</given-names></name><name><surname>Hung</surname> <given-names>DT</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genomic analysis of the evolution of fluoroquinolone resistance in Mycobacterium tuberculosis prior to tuberculosis diagnosis</article-title><source>Antimicrobial Agents and Chemotherapy</source><volume>60</volume><fpage>6600</fpage><lpage>6608</lpage><pub-id pub-id-type="doi">10.1128/AAC.00664-16</pub-id><pub-id pub-id-type="pmid">27572408</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><boxed-text><sec id="s8" sec-type="appendix"><title>SNP calling simulations</title><sec id="s8-1"><title>Reference genome collection</title><p>We downloaded 60 reference genomes (RefGenome) (i.e. completely assembled <italic>Mycobacterium tuberculosis</italic> genomes) from NCBI (RRID:<ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_002760">SCR_002760</ext-link>) (Genbank accession IDs can be found in <xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>). We limited our collection to genomes for which there were corresponding annotation files.</p></sec><sec id="s8-2"><title>Mapping CDS regions from reference genomes to H37Rv</title><p>Since the regions of interest were repetitive loci that have many homologies elsewhere in the genome, we were unable to use traditional alignment methods to map the genes of interest from H37Rv to the other RefGenomes. Instead, we made use of the clonal structure of the Mtbc genome to construct gene mappings from H37Rv to the RefGenomes as follows (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1A</xref>):</p><list list-type="order"><list-item><p>For each gene <inline-formula><mml:math id="inf74"><mml:mi>g</mml:mi></mml:math></inline-formula> annotated in H37Rv, collect the set of gene lengths 5 genes upstream and 5 genes downstream of <inline-formula><mml:math id="inf75"><mml:mi>g</mml:mi></mml:math></inline-formula> from H37Rv. Compare the set of 11 H37Rv gene lengths to every set of 11 consecutive gene neighborhoods on the RefGenome and assign a score based off of the intersection of each pair of sets.</p></list-item><list-item><p>Look at the gene neighborhood(s) with the top score after scanning the RefGenome and pairwise globally align (<xref ref-type="bibr" rid="bib12">Cock et al., 2009</xref>) <inline-formula><mml:math id="inf76"><mml:mi>g</mml:mi></mml:math></inline-formula> to every gene in the top scoring neighborhood using the following criteria: (i) identical characters are given 2 points, (ii) 1 point is deducted for each non-identical character, (iii) 2 points are deducted for opening a gap, (iv) 2 points are deducted for extending a gap.</p></list-item><list-item><p>Take the top scoring alignment <inline-formula><mml:math id="inf77"><mml:mi>r</mml:mi></mml:math></inline-formula> and assign a mapping from H37Rv gene <inline-formula><mml:math id="inf78"><mml:mi>g</mml:mi></mml:math></inline-formula> to RefGenome gene <inline-formula><mml:math id="inf79"><mml:mi>r</mml:mi></mml:math></inline-formula> if (i) the pairwise alignment score is <inline-formula><mml:math id="inf80"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and (ii) the base pair length of <inline-formula><mml:math id="inf81"><mml:mi>g</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf82"><mml:mi>r</mml:mi></mml:math></inline-formula> are equivalent (the latter ensures correct placement of mutations in downstream analysis). If either of these criteria is not met, then we do not assign a mapping from <inline-formula><mml:math id="inf83"><mml:mi>g</mml:mi></mml:math></inline-formula> to any CDS region on that RefGenome.</p></list-item></list></sec><sec id="s8-3"><title>Filtering low-quality mapped reference genomes</title><p>To assess the quality of the mappings from H37Rv to the set of RefGenomes, we compared the reference position start coordinates of each assigned mapping between each RefGenome and H37Rv. Again, making use of Mtbc clonality, we reasoned that the genomic structure of each pair of genomes is similar (if each RefGenome is indexed to start at the first gene on H37Rv <italic>Rv0001</italic>, then well mapped RefGenomes will have mapped genes that are located within a neighborhood of the coordinates from H37Rv). To test this (for each RefGenome), we took the absolute difference between the start coordinates for all of the mapped genes between the RefGenome and H37Rv. We then averaged these differences across all gene mappings between both genomes.</p><p>This measures the conservation (of the ordering) of the mapped genes between each pair of genomes (H37Rv and RefGenome) and gives an indication of how successful the mappings were on a global scale. We downloaded and mapped genes for 60 Genome Assemblies from GenBank (<xref ref-type="bibr" rid="bib2">Benson et al., 2009</xref>) and assessed the quality of each set of mappings using the measure described above (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1B-C</xref>). We excluded 6 RefGenomes on the basis of sporadic gene mappings against H37Rv which was determined by looking at the distribution of the mapping measure for all 60 assemblies. We kept the remaining 54 genomes for use in the simulations (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>).</p></sec><sec id="s8-4"><title>Altering RefGenomes at SNP test sites</title><p>We make use of the set of the (non-redundant) observed in-host SNPs across all genes (<xref ref-type="fig" rid="fig5">Figure 5D</xref>, <xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>). We alter each RefGenome by introducing mutations (that correspond to the aforementioned SNPs) into the genes successfully mapped to H37Rv, ensuring that the new bases differ from the corresponding base positions on H37Rv. Since successful mappings require that the mapped genes be the same length, the mutations are introduced into the same site on the RefGenome with respect to the gene specific coordinates (i.e. a gene <inline-formula><mml:math id="inf84"><mml:mi>n</mml:mi></mml:math></inline-formula> bp long will have coordinates <inline-formula><mml:math id="inf85"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:mo>⋯</mml:mo><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> from <inline-formula><mml:math id="inf86"><mml:msup><mml:mrow><mml:mn>5</mml:mn></mml:mrow><mml:mrow><mml:mi>'</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo><mml:mn>3</mml:mn><mml:mi>'</mml:mi></mml:math></inline-formula>). We store information pertaining to which bases were altered for each RefGenome <inline-formula><mml:math id="inf87"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi>β</mml:mi></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>. No simulations are run for genes on RefGenomes that are not successfully mapped to H37Rv.</p></sec><sec id="s8-5"><title>Simulating reads from complete Genomes</title><p>To validate our SNP calling methodology using the set of RefGenomes, we used ART (<xref ref-type="bibr" rid="bib37">Huang et al., 2012</xref>) to simulate short-read sequencing data altered versions of the RefGenomes (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1B</xref>). Since the aim of our simulations was to study the quality of our variant calls on our real data, we simulated data for each (altered) RefGenome that was of comparable quality to our real sequencing data: Illumina HiSeq 1000, read length of 100 bp, mean coverage of 80x, paired end reads, 200 bp mean size of DNA fragments, 25 bp standard deviation of DNA fragment size (settings: -ss HS10 -l 100 f 80 p -m 200 s 25).</p></sec><sec id="s8-6"><title>Mapping simulated reads to H37Rv and calling SNPs</title><p>Next we mapped the pool of simulated reads from the altered RefGenomes against the H37Rv reference genome and called SNPs according to most of the same procedures and WGS filters outlined in Materials and methods. However, in this instance we called SNPs at reference positions that supported an alternate allele and required that calls were flagged as <italic>Pass</italic> by Pilon (where the alternate allele frequency was ≥75% and no <italic>Ambiguous</italic>, <italic>Low Coverage</italic>, or <italic>Deletion</italic> flags were present at that position). For each RefGenome, this yielded the set of SNPs (between the altered RefGenome and H37Rv) called by our pipeline <inline-formula><mml:math id="inf88"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="bold">B</mml:mi></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1B</xref>).</p></sec><sec id="s8-7"><title>Calling SNPs with MUMmer</title><p>We used Mummer3 (<xref ref-type="bibr" rid="bib43">Kurtz et al., 2004</xref>) to call SNPs between H37Rv and each (unaltered) RefGenome. We aligned each pair of genomes and called SNPs between the alignments using the following commands:</p><list list-type="order"><list-item><p><monospace>nucmer -mum H37Rv.fasta RefGenome.fasta</monospace></p> </list-item><list-item><p><monospace>delta-filter -r -q H37Rv_RefGenome.delta &gt; H37Rv_RefGenome.filter</monospace></p> </list-item><list-item><p><monospace>show-snps -Clr -T H37Rv_RefGenome.filter &gt; H37Rv_RefGenome.snps</monospace></p> </list-item></list><p>The resulting SNP calls yielded the set of SNPs between each of the unmodified (unaltered) RefGenomes and H37Rv <inline-formula><mml:math id="inf89"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="bold">A</mml:mi></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1B</xref>).</p></sec><sec id="s8-8"><title>True and false positive SNP call analysis</title><p>To calculate the number of <italic>true positives</italic> and <italic>false positives</italic> with regard to our SNP calling pipeline for each gene <inline-formula><mml:math id="inf90"><mml:mi>g</mml:mi></mml:math></inline-formula> of interest (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>), we define the following sets of H37Rv coordinates for each RefGenome:</p><list list-type="bullet"><list-item><p><inline-formula><mml:math id="inf91"><mml:mi mathvariant="bold-italic">β</mml:mi></mml:math></inline-formula> - SNPs introduced into (altered) RefGenome</p></list-item><list-item><p><inline-formula><mml:math id="inf92"><mml:mi mathvariant="bold-italic">A</mml:mi></mml:math></inline-formula> - SNPs called between (unaltered) RefGenome &amp; H37Rv</p></list-item><list-item><p><inline-formula><mml:math id="inf93"><mml:mi mathvariant="bold-italic">B</mml:mi></mml:math></inline-formula> - SNPs called between (altered) RefGenome &amp; H37Rv</p></list-item><list-item><p><inline-formula><mml:math id="inf94"><mml:mi mathvariant="bold-italic">C</mml:mi></mml:math></inline-formula> - all reference positions (or coordinates) on H37Rv</p></list-item></list><p>The set of coordinates where an alternate allele was introduced into the RefGenome and called by the pipeline (true positive SNPs for gene <inline-formula><mml:math id="inf95"><mml:mi>g</mml:mi></mml:math></inline-formula>) is given by:<disp-formula id="equ11"><mml:math id="m11"><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>∖</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>∩</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>β</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>∖</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:math></disp-formula>where we normalize by SNP set <inline-formula><mml:math id="inf96"><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to make sure we're only accounting for test SNPs in our computations. The set of coordinates where an alternate allele was note introduced and called by the pipeline (false positive SNPs for gene <inline-formula><mml:math id="inf97"><mml:mi>g</mml:mi></mml:math></inline-formula>) is given by:<disp-formula id="equ12"><mml:math id="m12"><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>∖</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>∩</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>∖</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p><p>The set of coordinates where an alternate allele was introduced but was not called by the pipeline (false negative SNPs for gene <inline-formula><mml:math id="inf98"><mml:mi>g</mml:mi></mml:math></inline-formula>) is given by:<disp-formula id="equ13"><mml:math id="m13"><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>β</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>∖</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>∖</mml:mo><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p><p>The results of our simulations (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>) indicate that the number of true positive calls is consistent with the number of known SNPs across all genes and simulations. Perhaps more importantly, our results also suggest that false positive calls are rarely made for any SNP in our sample. Thus, while we may not have called all of the existing variation between paired isolates (false negative calls), it is unlikely that we called non-existing variation between any pair of isolates (false positives). That is, false-positive SNPs are rarely called, even in repetitive loci such as the PE/PPE gene family, supporting our decision to keep all SNP calls for downstream analysis.</p><fig id="app1fig1" position="float"><label>Appendix 1—figure 1.</label><caption><title>Overview of simulation methodology.</title><p>To test the accuracy of calling SNPs in repetitive regions with our workflow, we introduced mutations into complete <italic>Mycobacterium tuberculosis</italic> genomes (Reference Genomes), simulated reads from those genomes and assessed the accuracy recalling the mutations from the simulated reads while not introducing spurious mutations. (<bold>A</bold>) We used a sliding window of gene lengths along with a local alignment algorithm to map genes from the H37Rv reference genome to the set Reference Genomes. (<bold>C</bold>) We discarded Reference Genomes that mapped poorly (gene-to-gene) to the H37Rv reference genome (green-RefGenomes kept for simulations, red-discarded RefGenomes). (<bold>B</bold>) A schematic of our simulation methodology from Reference Genome collection to obtaining SNP sets <inline-formula><mml:math id="inf99"><mml:mi mathvariant="bold-italic">A</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="inf100"><mml:mi mathvariant="bold-italic">B</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf101"><mml:mi mathvariant="bold-italic">β</mml:mi></mml:math></inline-formula> which are used in our calculations of true positive and false positive calls for each gene.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-app1-fig1-v2.tif"/></fig><fig id="app1fig2" position="float"><label>Appendix 1—figure 2.</label><caption><title>Simulations indicate that we can accurately recall most introduced SNPs while rarely making spurious SNP calls.</title><p>We tested the number of true and false positives for each gene with detectable in-host SNPs (<xref ref-type="fig" rid="fig5">Figure 5D</xref>). For each gene we collected a set of non-redundant <italic>in-host</italic> SNPs (genomic positions at which these SNPs were called) observed across all patients (<xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>), the number of SNPs collected for each gene is given in (<bold>D</bold>). We then introduced these mutations into 54 complete genomes (RefGenomes) (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>) and simulated reads after introducing the respective mutations. Only genes that were mapped from H37Rv to a given RefGenome were part of the simulation for that RefGenome. (<bold>C</bold>) The number of successful mappings for each gene (i.e. the number of times each gene was part of a simulation). This is also the number of times true and false positive estimates were calculated for each gene (one estimate / simulation). (<bold>A</bold>) False positive calls were rarely made across all genes and simulation runs indicating the rarity of false positive SNP calls (calling a mutation that wasn't introduced) made by our pipeline for observed in-host SNPs, even in repetitive regions. (<bold>B</bold>) The number of true positive calls across all genes (across most simulation runs) closely matched the number of introduced SNPs for each gene indicating the rarity of False Negative SNP calls (not calling a mutation that was introduced). We note that no true or false positive estimates for <italic>Rv0192A</italic> were computed since this gene did not map to H37Rv for any of the 54 Reference Genomes used for the simulations.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-app1-fig2-v2.tif"/></fig></sec></sec></boxed-text></app><app id="appendix-2"><title>Appendix 2</title><boxed-text><sec id="s9" sec-type="appendix"><title>PacBio assembly vs. Illumina mapping SNP calling</title><sec id="s9-1"><title>DNA extraction and PacBio sequencing of Mtbc isolates</title><p>DNA extraction was performed according to a published protocol (<xref ref-type="bibr" rid="bib23">Epperson and Strong, 2020</xref>). Approximately 1 µg of high molecular weight genomic DNA was used as input for SMRTbell preparation, according to the manufacturer’s specifications (SMRTbell Template Preparation Kit 1.0, Pacific Biosciences, <ext-link ext-link-type="uri" xlink:href="https://www.pacb.com/wp-content/uploads/2015/09/Procedure-Checklist-20-kb-Template-Preparation-Using-BluePippin-Size-Selection.pdf">https://www.pacb.com/wp-content/uploads/2015/09/Procedure-Checklist-20-kb-Template-Preparation-Using-BluePippin-Size-Selection.pdf</ext-link>). Briefly, HMW gDNA was sheared to 20 kb using the Covaris g-tube at 4500 rpm. Following shearing, gDNA underwent DNA damage repair, ligation to SMRTbell adaptors and exonuclease treatment to remove any unligated gDNA. At least 500 ng final SMRTbell library per sample was cleaned with AMPure PB beads and 3-50 kb fragments were size selected using the BluePippin system on 0.75% agarose cassettes and S1 ladder, as specified by the manufacturer (Sage Science). Size selected SMRTbell libraries were annealed to sequencing primer and bound to the P6 polymerase prior to loading on the RSII sequencing system (Pacific Biosciences). Sequencing was performed using C4 chemistry and 240-min movies. Following data collection, raw data was converted into subreads for subsequent analysis using the RS_Subreads.1 pipeline within SMRTPortal (version 2.3), the web-based bioinformatics suite for analysis of RSII data.</p></sec><sec id="s9-2"><title>PacBio de novo assembly, genome polishing, and variant calling</title><p>PacBio and Illumina sequencing data was available for 34 clinical Mtbc isolates (<xref ref-type="bibr" rid="bib10">Chiner-Oms et al., 2019</xref>). We used Flye (<xref ref-type="bibr" rid="bib42">Kolmogorov et al., 2019</xref>) to de novo assemble the raw PacBio subreads from these 34 isolates (settings: <monospace>--pacbio-raw</monospace> <monospace>--genome-size</monospace> 5 m) (version 2.5). If Flye identified the presence of a circular contig, Circlator (<xref ref-type="bibr" rid="bib38">Hunt et al., 2015</xref>) was used to set the start each assembly at the DnaA locus. PacBio’s bax2bam function (settings: --subread) was used to convert PacBio legacy BAX files to BAM format. We ran PacBio’s implementation of Minimap2 (<xref ref-type="bibr" rid="bib45">Li, 2018</xref>) (pbmm2) to map and sort raw PacBio subreads to the de novo assembly. We iteratively polished the assembly three times by running the Quiver algorithm (<xref ref-type="bibr" rid="bib9">Chin et al., 2013</xref>) and used Samtools (<xref ref-type="bibr" rid="bib44">Li et al., 2009</xref>) to index the fasta files from the resulting assemblies. Thirty-one of our 34 samples assembled into a single circular contig (<xref ref-type="supplementary-material" rid="supp20">Supplementary file 20</xref>). We excluded three isolates that did not have a single circular assembly from downstream analysis. To call SNPs relative to the H37Rv reference, we used Minimap2 (<xref ref-type="bibr" rid="bib45">Li, 2018</xref>) to align each PacBio assembly to the H37Rv reference sequence. We used the <italic>paftools.js call</italic> utility included with Minimap2 to generate variant calls from each assembly to reference alignment.</p></sec></sec></boxed-text></app><app id="appendix-3"><title>Appendix 3</title><boxed-text><sec id="s10" sec-type="appendix"><title>Antibiotic resistance analyses for confirmed failure and relapse patients</title><p>We repeated our analyses on the allele frequency dynamics within antibiotic resistance (AR) loci and rates of resistance amplification using a subset of 121/200 patients with confirmed treatment failure or relapse (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>) (corresponding to the following sections - Results: In-host pathogen dynamics in antibiotic resistance loci, Allele frequency &gt;19% predicts subsequent fixation of resistance variants, Determinants of antibiotic resistance acquisition and microbiological treatment failure). For all 200 cases the order of sampling was available, but for 195/200 (119/121) confirmed failure patients) we also had the exact dates of sampling which were required for some analyses. We found that the analysis conclusions were unchanged between both the 121 subset and the full 200 patient sample.</p></sec><sec id="s11" sec-type="appendix"><title>In-host pathogen dynamics in antibiotic resistance loci</title><p>We detected 1401 (compared to 1939 using 200 patients, <xref ref-type="fig" rid="fig2">Figure 2B</xref>) non-synonymous and intergenic SNPs in 36 antibiotic resistance loci (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) that change in allele frequency by at least 5% between the first and second time points across our sample of 121 patients (<xref ref-type="fig" rid="app3fig1">Appendix 3—figure 1A</xref>). Of these SNPs, 1292 were non-synonymous, 61 were intergenic and 48 occurred with <italic>rrs</italic> (compared to 1774 non-synonymous, 91 intergenic, and 74 from the sample of 200 patients). We then determined the lowest AR frequency that can accurately predict the development of fixed resistance alleles later in time. After discarding 14 (20 using 200 patients) SNPs that were fixed (allele frequency &gt;75%) at both times points, we studied the allele frequency trajectories of 1387 (1,919 using 200 patients) AR SNPs.</p></sec><sec id="s12" sec-type="appendix"><title>Allele frequency &gt;19% predicts subsequent fixation of resistance variants</title><p>We calculated the allele frequency AF1 (in the first isolate collected from each patient) that predicted subsequent fixation of resistance variants and allowed a maximum False Positive Rate of 5%. Using the full set of 200 patients we calculated an optimal threshold of AF1*=19% with an associated sensitivity of 27.0% and a specificity of 95.8% (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). Ten mutant alleles across 14 isolates from seven patients had a frequency between 19% and 75% at the first time point and rose to fixation at the second time point (mean change in allele frequencies = 41%). Using the subset of 121 patients, we calculated a threshold of AF1*=17% with an associated sensitivity of 41.7% and a specificity of 95.3% (<xref ref-type="fig" rid="app3fig1">Appendix 3—figure 1B</xref>). Five mutant alleles across 10 isolates from four patients had a frequency between 17% and 75% at the first time point and rose to fixation at the second time point (mean change in allele frequencies = 48%).</p></sec><sec id="s13" sec-type="appendix"><title>Determinants of antibiotic resistance acquisition and microbiological treatment failure</title><p>Analyzing the 119 patients with clonal infection and known isolate sampling dates (195 using the full set of patients), we aimed to identify overall rates of resistance acquisition by focusing on AR SNPs with moderate to high changes in allele frequency &gt;= 40%. Using the set of 195 patients with clonal infection and sampling date, we detected 38 AR SNPs. We found that AR acquisition was more likely as the time between sampling increased, with the OR of AR acquisition being 1.023 per 30 day increment (95% CI 1.002, 1.045, p=0.035 Logistic Regression). Using the set of 119 patients with clonal infection and sampling date, we detected 13 AR SNPs. While AR acquisition was associated with the time between sampling, the association was not significant for this smaller sample with the OR acquisition being 1.017 per 30 day increment (95% CI 0.98, 1.055, p=0.375 Logistic Regression).</p><p>We then examined the relationship between pre-existing resistance and new AR acquisition in the subset of 195 patients that had samples collected &gt;= 2 months apart consistent with persistent or relapsed infection for that duration (n = 178) and separately analyzed 119 patients with confirmed failure or relapse. We defined pre-existing resistance as &gt;= 1 fixed AR SNP in the first isolate collected. Using the set of 178 patients, we found 259 pre-existing AR SNPs with 41% (73/178) of failure patients harboring resistance to any drug at first sampling (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). The majority of this resistance was MDR (<xref ref-type="fig" rid="fig3">Figure 3C</xref>) (multidrug resistance to at least isoniazid and rifamycin) 64% (47/73) and new resistance acquisition occurred mostly in patients with pre-existing resistance 20/27 (74%) (OR = 5.28, p=2.2×10-4 Fisher’s exact test) or pre-existing MDR (OR = 3.85, p=3.4×10-3 Fisher’s exact test). Using the set of 119 confirmed failure patients, we found 136 pre-existing AR SNPs with 30% (36/119) of failure patients harboring resistance to any drug at first sampling (<xref ref-type="fig" rid="app3fig2">Appendix 3—figure 2A</xref>). The majority of this resistance was MDR (<xref ref-type="fig" rid="app3fig2">Appendix 3—figure 2B</xref>) 67% (24/36) and new resistance acquisition occurred mostly in patients with pre-existing resistance 7/11 (64%) (OR = 4.77, p=1.8×10-2 Fisher’s exact test) or pre-existing MDR (OR = 3.9, p=0.04 Fisher’s exact test).</p><fig id="app3fig1" position="float"><label>Appendix 3—figure 1.</label><caption><title>Allele frequency dynamics within antibiotic resistance loci in 121/200 confirmed failure/relapse patients.</title><p>(<bold>A</bold>) The allele frequency trajectories for SNPs that occur in patients over the course of infection can be used to study the prediction of further antibiotic resistance using the frequency of alternate alleles detected in the longitudinal isolates collected from patients. (<bold>B</bold>) Plot of true positive rate (TPR) and false positive rate (FPR) for detecting eventual fixation of a resistance allele (<inline-formula><mml:math id="inf102"><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>≥</mml:mo> <mml:mi/></mml:math></inline-formula> 75%) as a function of initial allele frequency (<inline-formula><mml:math id="inf103"><mml:msub><mml:mrow><mml:mi>A</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-app3-fig1-v2.tif"/></fig><fig id="app3fig2" position="float"><label>Appendix 3—figure 2.</label><caption><title>Pre-existing resistance is associated with resistance amplification in 121/200 confirmed failure/relapse patients.</title><p>Among patients who fail treatment, (<bold>A</bold>) patients with pre-existing mutations that confer antibiotic resistance and (<bold>B</bold>) those that have pre-existing MDR are more likely to acquire antibiotic resistance mutations throughout the course of infection.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-app3-fig2-v2.tif"/></fig><fig id="app3fig3" position="float"><label>Appendix 3—figure 3.</label><caption><title>In-host SNP counts vs. time between isolate collection (119/121 confirmed failure/relapse patients with dates for both isolates shown).</title><p>Regressing the number of SNPs per patients on the timing between isolate collection, we found SNPs to accumulate at an average rate of 0.64 SNPs per genome per year (<inline-formula><mml:math id="inf104"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>7.24</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>9</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61805-app3-fig3-v2.tif"/></fig></sec></boxed-text></app></app-group></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.61805.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Kana</surname><given-names>Bavesh D</given-names></name><role>Reviewing Editor</role><aff><institution>University of the Witwatersrand</institution><country>South Africa</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>Your manuscript provides interesting and novel insight on the level of within host genetic diversity displayed by <italic>Mycobacterium tuberculosis</italic> during colonisation of the human lung. Furthermore, the data on association of how the presence of these variants affect treatment outcome and the potential for emergence of drug resistance provide important insight, illustrating how pathogen genetics can be used to guide tuberculosis treatment.</p><p><bold>Decision letter after peer review:</bold></p><p>[Editors’ note: the authors submitted for reconsideration following the decision after peer review. What follows is the decision letter after the first round of review.]</p><p>Thank you for submitting your work entitled &quot;In-host population dynamics of <italic>M. tuberculosis</italic> during treatment failure&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by a Senior Editor, a Reviewing Editor, and three reviewers. The reviewers have opted to remain anonymous.</p><p>Our decision has been reached after consultation between the reviewers. Based on these discussions and the individual reviews below, we regret to inform you that your work will not be considered further for publication in <italic>eLife</italic>.</p><p>Reviewers considered the methodological approach in your study interesting and novel. However, there were several major concerns raised about select components of the analysis and appending conclusions. Further, the results did not appear to contribute a substantive advance over what is already known.</p><p><italic>Reviewer #1:</italic></p><p>The study aims to provide insights in the evolution of <italic>M. tuberculosis</italic> in the host with a particular focus on resistance development. The authors can show, that pre-existing mutations predict failure and can be fixed in follow up samples. That is per se not a new finding, but the data are comprehensive and provide a detailed view into in host dynamics in patients with failing regimens.</p><p>What I found problematic in the study is that the authors work with only two samples per patient, thus, not really allowing the analyses of the longitudinal evolutionary dynamics in TB patients. Accordingly, the title should be modified to reflect more the work the authors performed.</p><p>Also, the selection of the resistance variants during therapy has been shown several times in previous papers and is the natural process for resistance development. So, the information presented is not really new.</p><p>What I find interesting is the calculation of the &quot;Lowest&quot; low frequency level that can be detected. However, the paragraph is not really conclusive and the authors should refer to recently published papers e.g. Dreyer et al., 2020.</p><p>Overall, the paper is well written, however, more data esp. patient data need to be provided to allow a clear cut interpretation of the data presented.</p><p>Essential revisions</p><p>Introduction – I would refer to <italic>M. tuberculosis</italic> complex</p><p>Results</p><p>Subsection “Identifying clonal Mtb populations in-host” – where are the data concerning MTB lineage shown?</p><p>Subsection “In-host pathogen dynamics in antibiotic resistance loci” – how can you differentiate contaminations from real low frequency variants if you have only the two datapoints? What statistical method did you use to distinguish sequencing errors from real SNPs?</p><p>Subsection “Allele frequency &gt;19% predicts subsequent fixation of resistance variants” – this paragraph is not really clear to me – what was the lowest frequency then that can be detected and results in treatment failure? Please look at Dreyer et al., 2020.</p><p>Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – don't understand this sentence – If the patients have treatment failure, they are treated all the time – or?</p><p>Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – the finding that pre-existing resistance mutations are leading to a higher chance of failure and resistance development is not surprising. What needs to be provided here are data on the effectivity of the given regimen. Meaning, one needs to know with how many active drugs a patient was treated when the low level resistance mutation was present. That at the ends most likely defines the chance of resistance development. This information thus is crucial for the interpretation of this paragraph.</p><p>Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – the argument may be valid, however, without data on the treatment regimen and the resistance data at a given timepoint, this is hard to made. This information needs to be provided.</p><p>Subsection “Antibiotic resistance and PE/PPE genes vary while antigens remain conserved” – variants in PE/PPE genes – in the opinion of this reviewer, a valid data analysis with regard to SNPs is impossible in repeat regions. Simply as you can not really allocate repeat block to a particular region in the reference genome. In addition, what are you doing with regions not present in the reference genome? If the authors want to keep this analyses in, they need to include a different way of confirmation e.g. Sanger sequencing of the regions from the strains of interest.</p><p>Subsection “Identifying candidate pathoadaptive loci from genome-wide variation” – resistance related genes need to be excluded here – or? That these have convergent evolution is not a surprise due to the selection pressure. This also induced the diversity. To really look into that, you would need to look into pan-susceptible isolates. What are the resistance types here?</p><p>Subsection “In-host mutations display phylogenetic convergence across multiple global lineages” – How do you distinguish site of likely sequencing error from real convergent evolution sites?</p><p>Discussion</p><p>– Please distinguish what is your finding and what is reported by others.</p><p>– “such patients”? What patients are you referring to? In your paper, you are not detailing patient characteristics. So, conclusions are difficult. In addition, I would be really cautious with the statement &quot;relatively high percentage&quot; develop resistance. To assay that, you would need to have a well-defined clinical trial in which you have clear cut in depth data for the patients characteristics incl. treatment data for the course of the treatment in relation to fixed and emerging resistances. At best with more than two serial isolates (see also comments above).</p><p>– Please reflect a bit more about the status of the published research here. What have others reported and detected? There have been several case reports already looking into that and other studies using new bioinformatic tools are available also.</p><p><italic>Reviewer #2:</italic></p><p>The manuscript by Vargas et al., presents a re-analysis of longitudinal samples for 307 TB patients. The analysis allows the authors to infer patterns of genetic diversity within a patient over time and the impact of different selective forces including treatment and host-pathogen interactions. The manuscript documents very well the analysis done as well as the different steps to reach the conclusions. However, in my opinion it lacks clarity and to some point novelty as it identified mostly known targets of drug resistance or targets of H-P difficult to validate. The methodology is however novel and relevant showing that in-host variation can be robustly analyzed in MTBC. I have some comments I would like the authors to address:</p><p>One major problem is that the authors lack associated clinical, demographic and epidemiological information about the cases under study. This is a major limitations as the authors can only look at the impact of bacterial genetic background (as infer from the WGS analysis) but not at relevant factor that we know or we suspect can influence levels of genetic diversity: HIV status, diabetes, treatment regimen, clinical adherence, nature of the lesions….etc. The authors correctly pool all the cases as failure (delayed culture conversion, treatment failure, relapse) but the reasons behind can be many and can impact the interpretation of some results. For example, H-P interaction loci will likely be linked to HIV status.</p><p>Subsection “Identifying clonal Mtb populations in-host”. The 7 SNPs threshold to discard a clonal infection seems misleading. Such an amount of variants could be compatible with clonal diversification in some cases with longitudinal samples taken several months apart and under antibiotic treatment pressure. From Figure 1C we can see the first reinfection case would have a 19-SNP difference, which seems compatible with the presence of two genotypes. But how would you classify a patient with a 10-SNP difference if you had it?</p><p>Figure 3. Related with the above question. One there is selection for a DR allele you expect a sweep of a particular clone, in practice this should translate in a decrease of diversity from Sample 1 to sample 2. Can you see this effect in Figure 3? Can you mark values for cases in which selection of DR is involved? Even more, in other patients where selection for H-P interaction is likely going on, can you see the effect? This will reinforce the idea that indeed those loci are involved in positive selection</p><p>Subsection “In-host pathogen dynamics in antibiotic resistance loci”. The 5% change threshold seems low if the variant is already at low frequency in the first sample. Of course, this is heavily influenced by read depth, but e.g. a change from 4% to 9% may be contributed by as little as 3 reads at 60X. The value could be adjusted dynamically according to the read depth stat for every sample.</p><p>Subsection “In-host pathogen dynamics in antibiotic resistance loci”. Wouldn't we expect a slightly higher mutation rate in this type of patients enriched with resistant strains and who failed treatment? On one hand because DR varaints are being fixed but also because It is expected to have some kind of hitchhiking effect during positive selection of DR variants in a clonal population. How the rate varies patient by patient?</p><p>Subsection “Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions”. Do the frequencies of SNPs in PE/PPE genes correlate in PacBio vs Illumina results? Also, are they mostly fixed or variable? A scatter plot maybe good here. This has implications to discuss about adaptation to host and the rate at which that would happen. In general, there is little information about the pacbio analysis and if it validates not just PE/PPE variation but other variation described for the patients sequenced with both technologies.</p><p>Discussion. Any reference to compare the presented value given the varied sources of the samples?</p><p>Discussion. If we talk about reinfections, not only is routinely performed sequencing advisable as the authors suggest, but also on subsequent samples from the patient to identify the second strain and adjust the treatment if needed.</p><p>Discussion. The low sensitivity of the 19% frequency threshold could be explained by the fact that random mutations appearing along a genotype that has acquired a drug resistance SNP (or any other variant that increases its fitness) will get fixated even if they have no phenotypic effect. How many synonymous mutations fall in this category? If you narrow down only to nonsynonymous will you increase sensitivity?</p><p>Subsection “In-host mutations display phylogenetic convergence across multiple global lineages”. In-host diversity. The authors analyze the gene-bygen diversity in-host. However, selection in-host does not necessarily reflect epidemiological success. The authors have the chance to look at it by comparing diversity within the host versus diversity between hosts (not just with the serial sample dataset but comparing to the reference collection of global isolates). Is there a correlation between in-host diversity vs between-host diversity?</p><p><italic>Reviewer #3:</italic></p><p>In this manuscript, the authors compile a significant body of work analyzing in-host population dynamics of <italic>Mycobacterium tuberculosis</italic>. The authors appropriately make use of publicly available data to compile a large dataset of paired samples in the same study participants over time, and the laboratory, bioinformatic, and statistical methods employed were well-designed to answer the questions of this manuscript. Overall, the analysis of 200 study participants from 8 studies has several important findings, including the low frequency of new resistance mutations in these participants, the importance of heteroresistance, in which minority variants representing {greater than or equal to}19% of reads predicted fixation in future samples, the significant contribution of prior resistance to development of new resistance, the greater role of drug resistance-associated mutations developing during drug treatment rather than new epitope-related mutations, confirmation of the development of new mutations in samples representing globally diverse lineages, and confirmation of a mutation rate within these samples that matches that of previous studies. Their findings suggest an important contribution of WGS to the prediction of treatment failure due to the potential superiority of WGS over phenotypic and rapid testing methods to identify heteroresistance at the start of treatment, which could presumably reduce the risk of treatment failure by indicating the need to adjust treatment regimens early.</p><p>While this manuscript represents a significant contribution to the field, a few considerations ought to be addressed. First, the use of public data, while laudable and appropriate for the aims of this study, introduces significant heterogeneity in the timing of sample collection and specific treatment regimens received. This does not, on its own, negatively affect the work of the manuscript, but the extent to which the authors combine these varied treatments and sample collection time points into a discussion of treatment failure and the relative contributions of resistance-associated mutations vs selection from the host's innate immune system requires further discussion. Supplementary file 1 summarized the heterogeneity of treatment regimens and sample collection time points to the extent that they are available. This indicates that some samples were collected before, during, and after treatment. Similarly, participants received diverse combinations of isoniazid, rifampin, rifapentine, pyrazinamide, ethambutol, streptomycin, and moxifloxacin with randomization of participants within each study introducing significant heterogeneity of selection pressure and time frames across samples in this study. The authors note a mutation frequency similar to the range derived in the absence of drug pressure (subsection “Characteristics of mutations in-host”) and refer in the Discussion section to inadequate therapy, which is hard to interpret with such heterogeneity. Due to the impact of baseline resistance (defined as allelic frequency &gt;75%) and MDR disease on the development of new mutations during treatment, greater discussion of individual sample timing and duration/type of drug pressure would be helpful, as would be discussion of the allelic frequency threshold used to define prior resistance. Figure 6 confirms the significant impact of drug pressure as the primary driver of these mutations, rather than mutations in epitope encoding genes, so the extent to which mutations in epitope encoding genes are specifically varying in response to host activity (or not varying) is not clearly related to the host response from the data as presented. Similarly, the authors note that mutations in drug resistance-associated loci is common and occurs across distinct Mtb lineages. While the finding of phylogenetic convergence is important, an alternative framing of the finding would be to consider these sites to vary in the presence of drug pressure independent of lineage.</p><p>Of related concern, the authors defined study participants as having met criteria for failed therapy due to culture positivity after only 2 months of treatment. While patients receiving appropriate treatment ideally develop early culture conversion, many sources would require a longer time frame than 2 months to assign treatment failure, particularly if these participants were receiving experimental study regimens. Due to the heterogeneity introduced by the sample inclusion strategy, the conclusions of the manuscript with respect to negative treatment outcomes should be presented with greater weight placed on time between samples and time since starting therapy. Reframing the findings of the study as changes that occur during treatment, rather than changes occurring in the setting of failure, could also improve the support for the conclusions of the manuscript. For example, the authors identify a strong correlation between the SNP diversity of each participant's first sample and second sample and conclude that this demonstrates ineffective therapy (subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure”). How did this correlate among those participants who were not thought to have failed therapy? Did this vary with time to culture conversion among those converted later (that is, differentiating between those with &quot;appropriate culture conversion&quot;, &quot;delayed culture conversion,&quot; and true &quot;failure&quot;). A comparison between participants with eventual success and those who either never converted or who changed regimens due to emerging resistance would help differentiate this issue. Alternatively, a comparison between the SNP diversity among the 44 participants excluded due to different strains might better support the conclusion in the discussion that the sustained diversity identified is due to the absence of effective therapy.</p><p>Finally, the Introduction is a bit confusing, combining discussions of the impact of host immunity, drug pressure, and microbiological and sequencing biases in selection of bacterial subpopulations. The result is that the reader is left confused about how to frame their interpretation of the study findings. This may be improved by a simpler introduction of the problem of unknown in-host variation over time and a summary of experiments that explain which bacterial factors should be considered at baseline to help predict future changes in Mtb isolates as well as the fixation of mutations in different genomic loci over time.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.61805.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><p>[Editors’ note: The authors appealed the original decision. What follows is the authors’ response to the first round of review.]</p><disp-quote content-type="editor-comment"><p>Reviewer #1:</p><p>The study aims to provide insights in the evolution of M. tuberculosis in the host with a particular focus on resistance development. The authors can show, that pre-existing mutations predict failure and can be fixed in follow up samples. That is per se not a new finding, but the data are comprehensive and provide a detailed view into in host dynamics in patients with failing regimens.</p><p>(1.1) What I found problematic in the study is that the authors work with only two samples per patient, thus, not really allowing the analyses of the longitudinal evolutionary dynamics in TB patients. Accordingly, the title should be modified to reflect more the work the authors performed.</p></disp-quote><p>We thank the reviewer for raising these two points. We agree that other smaller studies have attempted to study longitudinal isolates, largely with the focus of distinguishing re-infection with a new <italic>M. Tuberculosis</italic> (Mtb) strain from true treatment failure <italic>i.e.</italic> continued infection with the same strain. We want to emphasize that no other study to our knowledge has investigated Mtb in-host evolution at the scale of this study. Given the low rate of TB failure and relapse after treatment of drug susceptible TB, usually &lt;10%, and moderate rate for drug resistant TB, ~30-40%, as well as Mtb’s slow rate of evolution, it is necessary to pool data from many studies to explore this question with sufficient statistical power. This is precisely what we do in this study. Although we do not study multiple longitudinal samples per patient, and focus our analysis on two samples per patient, we believe this constitutes a significant advance of knowledge. This is especially the case as we largely focus on large allele frequency changes between the first and second time point and do not study transient or very low frequency variation that may require sampling at more than two time points. The two samples we study are longitudinally collected from each patient with dates of collection documented for nearly all isolates. In the Results section (Genome-wide in-host diversity), we describe how we used technical replicates to approximate the noise that arises from sampling and how a large change in allele frequencies between sampling is likely indicative of a mutant allele getting purged or sweeping to fixation in-host. This benchmarking constitutes an important methodological advance, in our opinion, for studying pathogen in-host evolution.</p><p>We have now modified the title to read: “In-host population dynamics of <italic>M. tuberculosiscomplex</italic> during active disease” to de-emphasize the study of strict treatment failure and address the reviewer’s comments below. We believe this title accurately reflects the contents of the manuscript and welcome additional suggestions for edits.</p><disp-quote content-type="editor-comment"><p>(1.2) Also, the selection of the resistance variants during therapy has been shown several times in previous papers and is the natural process for resistance development. So, the information presented is not really new.</p></disp-quote><p>We agree with the reviewer that the selection of resistance variants during therapy has been demonstrated previously. This was not the main finding of our study. Our goal in reporting this result was to extend previous analyses by providing a quantification of the rate of resistance acquisition. We also sought to provide a comparison of the rate of resistance amplification between drug resistant and drug susceptible isolates, and most importantly to study allele frequency dynamics. Specifically, the analysis where we determine the allele frequency above which we observed consistent fixation of the drug resistant allele, is novel and has not been previously reported. We also demonstrate a pattern of clonal interference for sites in resistance genes not known to be canonical resistance variants e.g. in Figure 2A <italic>katG</italic> L159F. Lastly and beyond drug resistance acquisition we study most of the Mtb genome (i.e., not just loci implicated in antibiotic resistance) to investigate selection for variants unrelated to antibiotic treatment. We believe these analyses to be novel and expand our knowledge of Mtb evolution.</p><disp-quote content-type="editor-comment"><p>(1.3) What I find interesting is the calculation of the &quot;Lowest&quot; low frequency level that can be detected. However, the paragraph is not really conclusive and the authors should refer to recently published papers e.g. Dreyer et al., 2020.</p></disp-quote><p>We thank the reviewer for pointing out the paper by Dreyer et al., 2020, we have now cited this reference in the paragraph describing the lowest frequency level. We have also revised this paragraph to add clarity to the results. In particular, we note that we are specifically interested in the “lowest” low frequency level that can be detected at <italic>the first time point</italic> and predict the fixation of that mutant allele at <italic>the second time</italic> point. This is to be distinguished from the ability to detect very low allele frequencies (~1%) accurately at a single time point which is what Dreyer et al., focus on by validating against simulated sequence data and in vitro mixtures. Notably Dreyer et al., report false positive calls only at allele frequencies lower than 5% in their study. All analyses in our paper focus on alleles at a frequency of &gt;5%, requiring minimum depth of 25x, and read depth of at least 5 reads supporting the alternate allele. The average depth of read coverage in our study isolates was 186x. Hence the Dreyer et al. results strongly support our choice of allele frequency thresholds and depth as having a very low risk of SNPs called due to sequence error alone.</p><disp-quote content-type="editor-comment"><p>(1.4) Overall, the paper is well written, however, more data esp. patient data need to be provided to allow a clear cut interpretation of the data presented.</p></disp-quote><p>We thank the reviewer for describing the manuscript as such, we aimed to communicate the results with accuracy and clarity. In response to the reviewers’ requests for additional patient data, we have now aggregated treatment from the source studies and added this metadata to Supplementary file 2 for each longitudinal isolate. We include columns that indicate the timing of sampling of Mtb relative to treatment, the treatment regimen administered and final patient outcome (and relevant details). Patient outcomes are defined as follows: Delayed culture conversion (sputum culture positive at baseline and ≥ 2 months treatment initiation with genomic analysis consistent with clonal infection), Failure or Relapse (sputum culture positive at baseline and ≥ 4.5 months treatment initiation with genomic analysis consistent with clonal infection), Failure or Relapse or Default (sputum culture positive at interval of ≥ 4.5 months with genomic analysis consistent with clonal infection, only partial treatment data is available) or N/A if date data was of low resolution, not available or no treatment data was available. We also determined Reinfection and Mixed infection based on the genomic analysis. This information has been added to the Materials and methods section.</p><p>We have now also repeated our analyses on the allele frequency dynamics within antibiotic resistance (AR) loci and rates of resistance amplification using the subset of 121/200 patients with confirmed failure or relapse (Supplementary file 2) corresponding to the following sections:</p><p>Results: In-host pathogen dynamics in antibiotic resistance loci, Allele frequency &gt;19% predicts subsequent fixation of resistance variants, Determinants of antibiotic resistance acquisition and microbiological treatment failure). For all 200 cases the order of sampling was available, but for 195/200 (119/121 confirmed failure subjects) we also had the exact dates of sampling which were required for some analyses. We found that the analysis conclusions were unchanged between both the 121 subset and the full 200 patient sample. We have now added the results of these analyses in the supplementary text as Appendix 3.</p><disp-quote content-type="editor-comment"><p>(1.5) Introduction - I would refer to M. tuberculosis complex</p></disp-quote><p>The reviewer is correct, as we have included Lineages 5 and 6, we have now revised “<italic>Mycobacterium tuberculosis”</italic> to <italic>“Mycobacterium tuberculosis</italic> complex<italic>”</italic> in the title and Introduction.</p><disp-quote content-type="editor-comment"><p>Results</p><p>(1.6) Subsection “Identifying clonal Mtb populations in-host”- where are the data concerning MTB lineage shown?</p></disp-quote><p>The data are shown in Figure 5A. We have now added a reference to this figure in the text.</p><disp-quote content-type="editor-comment"><p>(1.7) Subsection “In-host pathogen dynamics in antibiotic resistance loci” – how can you differentiate contaminations from real low frequency variants if you have only the two datapoints? What statistical method did you use to distinguish sequencing errors from real SNPs?</p></disp-quote><p>The analysis involved multiple steps of quality control including control for contamination, steps to differentiate sequencing error from real low frequency SNPs at each time point and steps to determine the significance of allele frequency changes between time points. Contamination of sequencing read data was evaluated for each isolate and does not require comparison of data at different time points. We used a highly cited method for contamination detection, Kraken, (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/24580807&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 24580807</ext-link>), that was recently also validated for use in Mtb (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/32122347&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 32122347</ext-link>). Patients with one or more isolates without &gt; 95% of reads mapping to Mtb complex were excluded as contaminated. We required a minimum read depth of 25x at each time point and that the mutant allele frequency changes by at least 5% between paired (longitudinal) isolates for us to include the drug resistance SNP in this analysis. To ensure a minimum of 5 reads supports each mutant allele called (with min. read depth of 25x), we also required an additional filter if the change in allele frequencies fell between 5% and 20%: these SNPs were only retained if the mutant allele (in both isolates) that had an allele frequency &gt; 0% was supported by at least 5 reads at either time point. As we trimmed all reads to a phred error of 1/100, the depth and allele frequency threshold associated with a probability of sequencing error of (1/100)^(5)=1x10<sup>-10</sup>. Furthermore, we studied alleles changing in allele frequency by &gt;5% only in drug resistance genes as they are known to be under selection. For other genes in the genome, we only count SNPs that have increased in allele frequency by &gt; 70% as significant for <italic>in host</italic> evolution based on our comparison with serial paired samples that were only passaged in vitro. We found that smaller allele frequency changes may be observed simply because of in vitro passage. These methods are detailed in the Materials and methods section.</p><p>To aid the readers, we have added the following descriptions of our methodology directly into the Results:</p><p>Subsection “Identifying clonal Mtb populations in-host”: “We required that no indels be present in the pileup supporting any SNP call, dropped SNP calls in repetitive regions and enforced a read depth ≥ 25x and alternate allele depth of ≥5 reads.”</p><p>Subsection “In-host pathogen dynamics in antibiotic resistance loci”: “To investigate temporal dynamics related to antibiotic pressure, we identified non-synonymous and intergenic SNPs within a set of 36 predetermined resistance loci associated with antibiotic resistance (Farhat et al., 2016, 2013) (Supplementary file 5 ) that changed in allele frequency by ≥5% (Sun et al., 2012) and ensuring that support of the alternate allele was ≥5 reads at each time point (Materials and methods ).”</p><disp-quote content-type="editor-comment"><p>(1.8) Subsection “Allele frequency &gt;19% predicts subsequent fixation of resistance variants” – this paragraph is not really clear to me – what was the lowest frequency then that can be detected and results in treatment failure? Please look at Dreyer et al., 2020.</p></disp-quote><p>We apologize about this lack of clarity. The goal of this analysis is to find the lowest mutant allele frequency that can be detected at the first time point and can be used to predict the fixation of the mutant allele in the sample taken at the second time point. The majority of our isolates were collected from patients with treatment failure, and we did not have a comparator group of patients cured with treatment to causally link resistance allele frequency with treatment failure. We do not claim to do this in the text, and we believe this association between allele frequency with the probability of fixation lays the foundation for future studies that can causally link to treatment failure. We identified this lowest allele frequency at the first time point to be 19%.</p><p>In response to the reviewer’s comment and other comments below, we have now simplified the paragraph to add clarity as follows:</p><p>“We aimed to measure the lowest AR allele frequency that can accurately predict the fixation of resistance alleles later in time (Dreyer et al., 2020; Sun et al., 2012; Zhang et al., 2016). We examined all 1,919 SNPs that varied by at least 5% in allele frequency (AF), and discarded 20 SNPs that were fixed at AF &gt; 75% in both isolates. We calculated the true positive rate (TPR) and false positive rate (FPR) for varying values of AF at the first time point (AF1) ∈{0,1,2,…,99,100}% (Figure 2C, Materials and methods) allowing a maximum FPR of 5%. We found the optimal classification threshold to be with an associated sensitivity of 27.0% and a specificity of 95.8%.”</p><disp-quote content-type="editor-comment"><p>(1.9) Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – don't understand this sentence – If the patients have treatment failure, they are treated all the time – or?</p></disp-quote><p>We apologize about this lack of clarity. We have now added treatment information into Supplementary file 2 as detailed above. We have also rephrased this sentence to make it​ clearer and pointed readers to the supplementary table that has the treatment metadata. The sentence currently reads:</p><p>“Among the set of 195/200 patients with clonal infection and sampling date, AR acquisition was more likely as the time between sampling increased with the OR of AR acquisition being 1.023 per 30day increment (95% CI 1.002, 1.045, P=0.035 Logistic Regression).”</p><disp-quote content-type="editor-comment"><p>(1.10) Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – the finding that pre-existing resistance mutations are leading to a higher chance of failure and resistance development is not surprising. What needs to be provided here are data on the effectivity of the given regimen. Meaning, one needs to know with how many active drugs a patient was treated when the low level resistance mutation was present. That at the ends most likely defines the chance of resistance development. This information thus is crucial for the interpretation of this paragraph.</p></disp-quote><p>We thank the reviewer for asking for additional clarity on this point. In Supplementary file 2 we now add detailed treatment data, and classify outcomes as detailed above including “delayed culture conversion” or “treatment failure and relapse”. These outcomes were defined by persistent positive cultures on treatment detected beyond 2 months and 4.5 months from treatment initiation. Detailed drugs administered are also given in Supplementary file 2, Drug resistance profiles are also given in Supplementary file 21. We have now repeated this analysis focusing specifically on patients with confirmed failure i.e. excluding patients where treatment details were not available or for which we could not exclude default. We explored the specific question the reviewer raised about the number of effective drugs received by the 11 patients that demonstrated evidence for resistance acquisition. We found that in 9/11 cases the patient received fewer than four effective drugs. This result has now been added to the Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure”.</p><disp-quote content-type="editor-comment"><p>(1.11) Subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure” – the argument may be valid, however, without data on the treatment regimen and the resistance data at a given timepoint, this is hard to made. This information needs to be provided.</p></disp-quote><p>For ease of reference here is the argument the reviewer is referring to – “We also quantified genome-wide Mtb diversity in-host among the patients with microbiological treatment failure….”</p><p>We have now aggregated patient treatment from the source studies and added this metadata to Supplementary file 2 for each longitudinal isolate (see above). We include columns that indicate the timing of sampling of Mtb relative to treatment, the treatment regimen administered and final patient outcome (and relevant details). We believe this updated treatment regimen and resistance data better support the claim made in the text. We have now also revised the text to read:</p><p>“We also quantified genome-wide Mtb diversity in-host among the patients with persistent or relapsed infection for ≥2 months. We reasoned that if these patients are not on or not adherent to effective antibiotic treatment, their effective pathogen population size may be large and prone to more genetic drift or turnover of minority variants with and without selection (Trauner et al., 2017).”</p><disp-quote content-type="editor-comment"><p>(1.12) Subsection “Antibiotic resistance and PE/PPE genes vary while antigens remain conserved” – variants in PE/PPE genes – in the opinion of this reviewer, a valid data analysis with regard to SNPs is impossible in repeat regions. Simply as you cannot really allocate repeat block to a particular region in the reference genome. In addition, what are you doing with regions not present in the reference genome? If the authors want to keep this analyses in, they need to include a different way of confirmation e.g. Sanger sequencing of the regions from the strains of interest.</p></disp-quote><p>The reviewer is correct to raise concern about variant calling in repetitive regions of the Mtb genome. We want to emphasize that while the PE/PPE genes are usually excluded from genomic analyses, a large proportion of these regions are not repetitive and allow for variant calling with high accuracy. Many PE/PPE genes which do contain repetitive sequence content also contain uniquely alignable sequences. We have extensively validated that the variant calls made in our study are not due to mapping or sequencing error in the following three ways:</p><p>1) We used extensive filters for calling all SNPs (see Materials and methods). (2) We simulated all the SNPs that we observed in-host (including the ones detected in the PE/PPE regions) in a diverse set of complete genomes, simulated short-read sequencing data from these genomes then mapped the simulated reads to H37Rv and called SNPs with our pipeline. These simulations demonstrated that we could reliably call the SNPs we observed in-host from short-read sequencing data (see subsection “Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions”, subsection “SNP Calling simulations in repetitive genomic regions” and Appendix 1). Finally, (3) We used paired PacBio and Illumina sequences (taken from the same isolates) and compared the congruence of base calls across the genome using the PacBio sequences as a “ground truth” since PacBio reads are much longer and are used routinely to sequence repetitive regions. After comparing the congruence of calls between both sequencing technologies, we only called SNPs in regions of the genome where the base calls from Illumina agreed with the PacBio base calls (see subsection “Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions”, subsection “Empirical Score for Difficult-to-Call Regions” and Appendix 2).</p><disp-quote content-type="editor-comment"><p>(1.13) Subsection “Identifying candidate pathoadaptive loci from genome-wide variation” – resistance related genes need to be excluded here – or? That these have convergent evolution is not a surprise due to the selection pressure. This also induced the diversity. To really look into that, you would need to look into pan-susceptible isolates. What are the resistance types here?</p></disp-quote><p>We agree with the reviewer that convergent evolution occurs within the antibiotic resistance genes due to selection pressure. We included the drug resistance loci as a “Positive control” or comparator set of genes. Given that we hypothesize that other selection pressures (e.g., immune-related) might account for some of the other diversity, the observation of a strong signal in these resistance genes is a good sign that our methodology works and provides a point of reference of the strength of the selection signal. We break down the SNP frequencies that we detect in-host into different categories: Essential, Non-Essential, Antigen, PE/PPE and Antibiotic Resistance genes so we separate the diversity stemming from antibiotic pressure from the diversity likely arising from other selection pressure or genetic drift. See Figure 6 C-D in the main text to demonstrate this.​</p><disp-quote content-type="editor-comment"><p>(1.14) Subsection “In-host mutations display phylogenetic convergence across multiple global lineages” – How do you distinguish site of likely sequencing error from real convergent evolution sites?</p></disp-quote><p>For variant calling in the globally representative 20,352 isolates we used conservative variant calling filters to minimize the effects of sequencing error. Briefly, for each call, we required mean base quality &gt; 20, mean mapping quality &gt; 30, excluded SNPs called by reads also supporting an insertion or deletion, a minimum depth of 25 reads (Materials and methods). Additionally, we restricted our analysis to regions of the genome that were confirmed callable with high accuracy when compared with PacBio sequencing data (Subsection “Empirical Score for Difficult-to-Call Regions”) and excluded positions located within mobile genetic elements (subsection “Variant Calling/SNP Calling”). We also required that the allele called for a given SNP site be pure <italic>i.e.</italic> supported by at least 90% of the reads (Materials and methods ) otherwise we marked a missing call at that SNP site in that isolate. Finally, we also dropped SNP sites that had missing calls in &gt; 25% of isolates and then dropped isolates that had missing calls &gt; 25% of SNP sites to further exclude low-quality SNP sites and low-quality sequenced isolates (Materials and methods). Taken together, we believe this extensive filtering removed sites prone to sequencing error.</p><disp-quote content-type="editor-comment"><p>Discussion</p><p>(1.15) Please distinguish what is your finding and what is reported by others.</p></disp-quote><p>We have rearranged this paragraph to aid the reader in distinguishing our findings (later in the paragraph) from what has been reported by others (moved to earlier in the paragraph).</p><p>“In our Mtb populations sequenced from active TB patients enriched for negative treatment outcomes, we find a wealth of dynamics in genetic loci associated with antibiotic resistance, including a high turnover of minor variants. Known factors that determine treatment outcome are complex and include severity of lung disease, cavitation and adherence to treatment among others (Imperial et al., 2018). Additionally, resistance acquisition in the course of one infection is comparatively rare in most pathogenic bacteria (Llewelyn et al., 2017). Here, we observe that 9% of patients with confirmed delayed culture conversion, failure and relapse amplify resistance over time. Our findings of a higher rate of resistance acquisition in patients with MDR at the outset and with time between sampling, emphasize the importance of appropriately tailoring treatment regimens as well as close surveillance for microbiological clearance and resistance acquisition by phenotypic or genotypic means. The observed high rate of resistance acquisition also emphasizes Mtb’s biological adaptability and the long duration of drug pressure in vivo. In addition to clonal acquisition of resistance, we find that sequencing revealed a substantial proportion of mixed infection or reinfection (28% of samples collected ≥2months apart). This high percentage suggests that patient treatment and control of disease transmission can be better guided if pathogen sequencing is routinely performed for cases with persistent positive cultures especially in high TB prevalence settings where reinfection is more likely. Reinfection can also introduce strains with a different antibiotic susceptibility profile requiring adjustment in the treatment regimen.”</p><disp-quote content-type="editor-comment"><p>(1.16) “such patients”? What patients are you referring to? In your paper, you are not detailing patient characteristics. So, conclusions are difficult. In addition, I would be really cautious with the statement &quot;relatively high percentage&quot; develop resistance. To assay that, you would need to have a well defined clinical trial in which you have clear cut in depth data for the patients characteristics incl. treatment data for the course of the treatment in relation to fixed and emerging resistances. At best with more than two serial isolates (see also comments above).</p></disp-quote><p>We have modified this calculation to include only the 119 subjects that we have confirmed delayed culture conversion, failure or relapse, have appropriate isolate date collection data for both timepoints and have detailed treatment data (Supplementary file 2).</p><p>“Here, we observe that 9% of patients with confirmed delayed culture conversion, failure and relapse amplify resistance over time.”</p><disp-quote content-type="editor-comment"><p>(1.17) Please reflect a bit more about the status of the published research here. What have others reported and detected? There have been several case reports already looking into that and other studies using new bioinformatic tools are available also.</p></disp-quote><p>In response to the reviewers comment, we have now added a few sentences to add background on what others have reported and clarity to this paragraph. This additional information relates our study to similar research questions tackled by other studies and cited (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/28424085&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 28424085</ext-link>) and (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/32398743&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 32398743</ext-link>) accordingly.</p><p>“While prior studies have investigated the lowest resistance allele frequency that can be detected in clinical sputum samples (Dreyer et al., 2020; Trauner et al., 2017), there is little information on the clinical relevance of these low frequency variants. We provide a proof-of-concept analysis that minor AR alleles, occurring at a frequency ≥19%, can predict fixation of the variant with a specificity &gt;95% of mutations in-host, although we find the sensitivity of this threshold to be low. The low sensitivity is because the majority of alleles that sweep to fixation are actually not detectable at all at the first time point, suggesting that more frequent sampling may be needed. In the future, higher depth and more frequent sequencing can elucidate more clearly the role of minor AR allele detection in clinical management of TB treatment.”</p><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>The manuscript by Vargas et al., presents a re-analysis of longitudinal samples for 307 TB patients. The analysis allows the authors to infer patterns of genetic diversity within a patient over time and the impact of different selective forces including treatment and host-pathogen interactions. The manuscript documents very well the analysis done as well as the different steps to reach the conclusions. However, in my opinion it lacks clarity and to some point novelty as it identified mostly known targets of drug resistance or targets of H-P difficult to validate. The methodology is however novel and relevant showing that in-host variation can be robustly analyzed in MTBC. I have some minor and major comments I would like the authors to address:</p><p>(2.1) One major problem is that the authors lack associated clinical, demographic and epidemiological information about the cases under study. This is a major limitations as the authors can only look at the impact of bacterial genetic background (as infer from the WGS analysis) but not at relevant factor that we know or we suspect can influence levels of genetic diversity: HIV status, diabetes, treatment regimen, clinical adherence, nature of the lesions….etc.The authors correctly pool all the cases as failure (delayed culture conversion, treatment failure, relapse) but the reasons behind can be many and can impact the interpretation of some results. For example, H-P interaction loci will likely be linked to HIV status.</p></disp-quote><p>We thank the reviewer for raising this concern. It is similar to concerns raised by reviewer 1 to which we provide detailed response and describe edits to the manuscript above in Response 1.4. We believe strongly that the manuscript is now significantly improved as a result of this feedback. There are some metadata patient characteristics that we were unable to locate and aggregate such as HIV status, diabetes and nature of the lesions. We agree with the reviewer that such host characteristics (e.g., HIV) may alter selective pressures on the pathogen during active disease. Future studies examining the association between HIV and convergent evolution in genes identified here to be associated with H-P (host-pathogen) interactions will be worthwhile. We note that a large proportion of isolates included in this study derive from low HIV incidence settings specifically: Peru, the United Kingdom and China that are the source sites for 144/307 patients. Additionally, although some isolates from Witney et al., and Bryant et al., 163/307 were isolated from patients with HIV, all enrolled HIV patients had to have a CD4 count &gt;250 in those two studies. Hence, we expect the overall proportion of patients with HIV/AIDS to be low in our dataset. Though we lack this specific metadata, we believe that our validation of results across a large dataset of WGS data (~20k samples) constitutes an important first step and adds a new research direction for studying H-P relationships using convergent evolution. We demonstrate that in addition to loci involved the acquisition of antibiotic resistance loci implicated in modulation of innate host-immunity appear to be under positive selection.</p><disp-quote content-type="editor-comment"><p>(2.2) Subsection “Identifying clonal Mtb populations in-host”. The 7 SNPs threshold to discard a clonal infection seems misleading. Such an amount of variants could be compatible with clonal diversification in some cases with longitudinal samples taken several months apart and under antibiotic treatment pressure. From Figure 1C we can see the first reinfection case would have a 19-SNP difference, which seems compatible with the presence of two genotypes. But how would you classify a patient with a 10-SNP difference if you had it?</p></disp-quote><p>The reviewer is correct, the minimum SNP difference that we see among the reinfection cases is 19 SNPs and the largest SNP difference that we see among clonal infection is 7 SNPs. The 44 reinfection cases had a median SNP distance = 708 with IQR = [250.5, 1086.5] showing a large separation between reinfection cases and clonal infection cases. As such, our effective threshold for reinfection amounts to &gt; 18 SNPs and our threshold for clonal cases was &lt; 7 SNPs in our data. We agree with the reviewer that a pair of isolates 10-SNPs apart is likely consistent with clonal infection. The prior literature on the subject reports the use of a 12-SNP threshold suitable for epidemiological links via genomic sequencing (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/23158499&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 23158499</ext-link>, cited 667 times). We did not have any such borderline cases in our dataset. In response to the reviewer’s comment, we have now added the median SNP distance and IQR range for the reinfection cases to the main text to communicate than none of these cases were borderline.</p><disp-quote content-type="editor-comment"><p>(2.3) Figure 3. Related with the above question. One there is selection for a DR allele you expect a sweep of a particular clone, in practice this should translate in a decrease of diversity from Sample 1 to sample 2. Can you see this effect in Figure 3? Can you mark values for cases in which selection of DR is involved? Even more, in other patients where selection for H-P interaction is likely going on, can you see the effect? This will reinforce the idea that indeed those loci are involved in positive selection</p></disp-quote><p>We thank the reviewer for their suggestion. We have modified Figure 3A in the following way (Figure 3 legend): “The number of hSNPs called in the second sample isolated vs the number of hSNPs called in the first sample isolated from each of 178 subjects (median T1=13.5 hSNPs, median T2=13.5 hSNPs). The dashed line is y = x. Red denotes 27/178 patients who had an antibiotic resistance in-host SNP arise between sampling (median T1=15.0 hSNPs, median T2=11.0 hSNPs), blue denotes 5/178 patients who had a putative host-adaptive in-host SNP (Rv1944c, Rv0095c, PPE18, PPE54, PPE60) arise between sampling (median T1=19.0 hSNPs, median T2=6.0 hSNPs).” We observe a lower median number of hSNPs at the second time point for the subjects in which DR alleles sweep to fixation and in which putative H-P alleles sweep to fixation, demonstrating a reduction in diversity and reinforcing the idea that these loci may be involved with positive selection. We have added this observation to the second-to-last paragraph of the Discussion: “Consistent with the idea that positive selection is acting on alleles within these loci, we observe a reduction in diversity at the second time point for the subjects in which drug resistant alleles sweep to fixation and in which putative host-pathogen alleles sweep to fixation (Figure 3A).”</p><disp-quote content-type="editor-comment"><p>(2.4) Subsection “In-host pathogen dynamics in antibiotic resistance loci”. The 5% change threshold seems low if the variant is already at low frequency in the first sample. Of course, this is heavily influenced by read depth, but e.g. a change from 4% to 9% may be contributed by as little as 3 reads at 60X. The value could be adjusted dynamically according to the read depth stat for every sample.</p></disp-quote><p>We thank the reviewer for raising this concern. It is similar to two concerns raised by reviewer 1. We refer to the reviewer to a detailed description of depth and other quality control criteria and their discussion in relation to a recent publication on the topic in Mtb by Dreyer et al., under Response 1.3 and 1.7 above. Under these two responses we also list detailed edits to the text to add clarity. We note that we only looked at small allele frequency changes, down to 5%, in drug resistance genes as they are known to be under selection. In other genes in the genome, we only count SNPs that have increased in allele frequency by &gt; 70% as significant for in host evolution based on our comparison with serial paired samples that were only evolved in vitro.</p><disp-quote content-type="editor-comment"><p>(2.5) Subsection “In-host pathogen dynamics in antibiotic resistance loci”. Wouldn't we expect a slightly higher mutation rate in this type of patients enriched with resistant strains and who failed treatment? On one hand because DR varaints are being fixed but also because It is expected to have some kind of hitchhiking effect during positive selection of DR variants in a clonal population. How the rate varies patient by patient?</p></disp-quote><p>We thank the reviewer for this suggestion. In Figure 6B, we observe that fixed</p><p>SNPs accumulate at an average rate of 0.56 SNPs per genome per year (95% CI = [0.408, 0.708], <italic>P</italic> = 7x10<sup>-12</sup>) when regressing the number of SNPs per subject on the timing between isolate collection for 195/200 subjects with isolate collection dates. Per the suggestion above, we also regressed the number of SNPs per subject on the timing between collection for 119 confirmed failure/relapse subjects. In this subset of patients, we observe that SNPs accumulate at an average rate of 0.64 SNPs per genome per year (95% CI = [0.432, 0.84], <italic>P</italic> = 7x10<sup>-9</sup>). Although this rate is marginally higher, the 95% confidence intervals overlap substantially. The observed number of new mutations is expectedly stochastic and varied considerably from patient to patient with only 71/200 subjects developing ≥1 in-host SNP and the maximum number of in-host SNPs observed being 5. The analysis of the mutation rate in the 119-patient subset has now been added to the supplement along with a supporting figure and table (Appendix 3—figure 3, Supplementary file 2).</p><disp-quote content-type="editor-comment"><p>(2.6) Subsection “Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions”. Do the frequencies of SNPs in PE/PPE genes correlate in PacBio vs Illumina results? Also, are they mostly fixed or variable? A scatter plot maybe good here. This has implications to discuss about adaptation to host and the rate at which that would happen. In general, there is little information about the pacbio analysis and if it validates not just PE/PPE variation but other variation described for the patients sequenced with both technologies.</p></disp-quote><p>Calling low allele frequency variants in microbial genomes using PacBio sequencing data does not yet have an established methodology. This is in part due to PacBio reads having much lower per base accuracy compared with Illumina; for our PacBio data, reads have an estimated per base error rate of 10%. We note that the study of within sample diversity with PacBio sequencing is an active area of research, and we hope that new methods will be developed to use PacBio data in this space in the near future. For this study, we did use paired PacBio and Illumina sequences (taken from the same isolates) and compared the congruence of base calls at high allele frequencies (&gt;75%) across the genome. For this analysis we used PacBio based assemblies as a “ground truth” since PacBio reads are much longer and are used routinely to sequence repetitive regions. (see subsection “Simulations and PacBio sequencing demonstrate a low false-positive rate in repetitive regions”, subsection “Empirical score for difficult-to-call regions” and Appendix 2).</p><p>We have added the following sentence for clarity: “While the high per base error rate makes it difficult to call low allele frequency variants in microbial genomes, we made use of the PacBio sequencing data to assess fixed variant calls.” in the context of the following paragraph: “Second, we assessed the congruence in variant calls between short-read Illumina data and long-read PacBio data for a set of isolates that underwent sequencing with both technologies (Materials and methods). Unlike Illumina generated reads, PacBio reads are much longer and have randomly distributed error profiles (Rhoads and Au, 2015). With high coverage, PacBio sequencing can reliably reconstruct full microbial genomes and identify SNPs in repetitive regions. While the high per base error rate makes it difficult to call low allele frequency variants in microbial genomes, we made use of the PacBio sequencing data to assess fixed variant calls. The comparison with PacBio assemblies confirmed empirically a low rate of false positive base calls in genomic regions where we observed in-host SNPs (Materials and methods).”</p><disp-quote content-type="editor-comment"><p>(2.7) Discussion. Any reference to compare the presented value given the varied sources of the samples?</p></disp-quote><p>To our knowledge our paper is the first to measure the rate of resistance acquisition at this scale using longitudinal whole genome sequencing in patients with documented treatment failure. We also show a significant association with pre-existing resistance. There is unfortunately limited context from the literature to add to this section, but we hope the revisions we have made with regards to supplying treatment regimen data, as well as the number of effective drugs (see Response 1.10) now improves the interpretation of this result.</p><disp-quote content-type="editor-comment"><p>(2.8) Discussion. If we talk about reinfections, not only is routinely performed sequencing advisable as the authors suggest, but also on subsequent samples from the patient to identify the second strain and adjust the treatment if needed.</p></disp-quote><p>The reviewer raises a great point which has been incorporated into the Discussion by adding the following sentence: “Reinfection can also introduce strains with a different antibiotic susceptibility profile requiring adjustment in the treatment regimen.”</p><disp-quote content-type="editor-comment"><p>(2.9) Discussion. The low sensitivity of the 19% frequency threshold could be explained by the fact that random mutations appearing along a genotype that has acquired a drug resistance SNP (or any other variant that increases its fitness) will get fixated even if they have no phenotypic effect. How many synonymous mutations fall in this category? If you narrow down only to nonsynoymous will you increase sensitivity?</p></disp-quote><p>The sensitivity of the 19% frequency threshold may be explained by the detection of several competing clones at low frequency (AF &lt; 19%) (Figure 2A, Figure 2—figure supplement 1) early on in treatment and then the subsequent fixation of one of these clones at a later point in time. The sensitivity may also be explained by the observation that most antibiotic resistance alleles that fix in the second time point are undetectable in the first time point as we now note in the revised text and describe further under Response 1.8. Additionally, our methodology restricted this analysis to study only intergenic and non-synonymous mutations that are located within antibiotic resistance loci. This detail on our filtering process for calling the variants for this analysis can be found in the Materials and methods ​section. We have added several sentences to further explain our observations to the discussion.</p><p>Edits to the manuscript are described above under response 1.8 and response 1.17.</p><disp-quote content-type="editor-comment"><p>(2.10) Subsection “In-host mutations display phylogenetic convergence across multiple global lineages”. In-host diversity. The authors analyze the gene-bygen diversity in-host. However, selection in-host does not necessarily reflect epidemiological success. The authors have the chance to look at it by comparing diversity within the host versus diversity between hosts (not just with the serial sample dataset but comparing to the reference collection of global isolates). Is there a correlation between in-host diversity vs between-host diversity?</p></disp-quote><p>We thank the reviewer for raising this point. We agree that selection in host does not reflect epidemiological success. This is specifically why we conducted the analysis assessing phylogenetic convergence across 20,352 isolates. Specifically reasoning that pathoadaptive mutations observed to sweep to fixation in-host and not compromise pathogen transmissibility are likely to arise independently within other subjects and in separate geographic regions in a convergent manner (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/23995135&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 23995135</ext-link>). This is demonstrated by the detection of convergent resistance mutations (Figure 7B-C, Figure 7—figure supplement 1, Figure 7—figure supplement 2) along with other hypothesized pathoadaptive mutations. We agree that analyses comparing in-host and between host diversity would be valuable, but these would have to rely on available linked in-host and transmission/outbreak data at a sufficient scale. This type of data is not available to us and would likely require large data prospectively collected for this purpose.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>In this manuscript, the authors compile a significant body of work analyzing in-host population dynamics of Mycobacterium tuberculosis. The authors appropriately make use of publicly available data to compile a large dataset of paired samples in the same study participants over time, and the laboratory, bioinformatic, and statistical methods employed were well-designed to answer the questions of this manuscript. Overall, the analysis of 200 study participants from 8 studies has several important findings, including the low frequency of new resistance mutations in these participants, the importance of heteroresistance, in which minority variants representing {greater than or equal to}19% of reads predicted fixation in future samples, the significant contribution of prior resistance to development of new resistance, the greater role of drug resistance-associated mutations developing during drug treatment rather than new epitope-related mutations, confirmation of the development of new mutations in samples representing globally diverse lineages, and confirmation of a mutation rate within these samples that matches that of previous studies. Their findings suggest an important contribution of WGS to the prediction of treatment failure due to the potential superiority of WGS over phenotypic and rapid testing methods to identify heteroresistance at the start of treatment, which could presumably reduce the risk of treatment failure by indicating the need to adjust treatment regimens early.</p><p>(3.1) While this manuscript represents a significant contribution to the field, a few considerations ought to be addressed. First, the use of public data, while laudable and appropriate for the aims of this study, introduces significant heterogeneity in the timing of sample collection and specific treatment regimens received. This does not, on its own, negatively affect the work of the manuscript, but the extent to which the authors combine these varied treatments and sample collection time points into a discussion of treatment failure and the relative contributions of resistance-associated mutations vs selection from the host's innate immune system requires further discussion. Supplementary file 1 summarized the heterogeneity of treatment regimens and sample collection time points to the extent that they are available. This indicates that some samples were collected before, during, and after treatment. Similarly, participants received diverse combinations of isoniazid, rifampin, rifapentine, pyrazinamide, ethambutol, streptomycin, and moxifloxacin with randomization of participants within each study introducing significant heterogeneity of selection pressure and time frames across samples in this study.</p></disp-quote><p>The reviewer raises several important points. We agree that the manuscript lacked detail on treatment outcomes and the text may have focused too heavily on the discussion of treatment failure that is challenged in interpretation by the meta-analysis design. We have taken these critiques to heart and made extensive revisions as follows: We have generated a new and detailed patient treatment metadata Table (Supplementary file 2) that details treatment regimens received and describes exactly when samples were collected relative to treatment initiation. We refer the referee to response 1.4 above for additional details on this meta-data. We confirmed 121 cases of treatment failure based on treatment regimen data and sampling times, and an additional 57 cases had limited treatment data and include failure/relapse or default/treatment interruption. In the grand majority of cases, n = 117/121, samples were collected at the start of treatment and in follow up when treatment failure or relapse was identified.</p><p>We replicated all analyses focused on treatment/drug resistance (Appendix 3, Appendix 3—figure 1, Appendix 3—figure 2) specifically in the 121 cases of confirmed treatment failure and provide these results in the supplement. We refer the referee to Response 1.4 in which we report the findings of these replicate analyses and compare them to the analyses reported in the manuscript on the larger sample.</p><p>We have also now revised the text to change the focus from treatment failure to persistent active disease:</p><p>– Changed title from “In-host population dynamics of <italic>M. tuberculosis</italic>​ during treatment failure” to “In-host population dynamics of <italic>M. tuberculosis</italic> during active disease”</p><p>– Changed sentence from “Of the 178/200 subjects with unsuccessful treatment outcome” to “Of the 178/200 subjects with persistent clonal infection &gt; 2 months” in Abstract</p><p>– Deleted the statement “All study subjects had either recently completed treatment or were receiving treatment when samples were collected…” from Results.</p><p>– Changed sentence from “In our Mtb populations sequenced from active TB patients enriched for delayed culture conversion, treatment failure and relapse…” to “In our Mtb populations sequenced from active TB patients enriched for negative treatment outcomes…”</p><p>– Added a section to the supplement titled “Appendix 3 – Antibiotic Resistance Analyses for Confirmed Failure and Relapse Patients” as well as new Appendix 3 – figure 1, Appendix 3—figure 2.</p><disp-quote content-type="editor-comment"><p>(3.2) The authors note a mutation frequency similar to the range derived in the absence of drug pressure (subsection “Characteristics of mutations in-host”) and refer in the Discussion section to inadequate therapy, which is hard to interpret with such heterogeneity. Due to the impact of baseline resistance (defined as allelic frequency &gt;75%) and MDR disease on the development of new mutations during treatment, greater discussion of individual sample timing and duration/type of drug pressure would be helpful, as would be discussion of the allelic frequency threshold used to define prior resistance. Figure 6 confirms the significant impact of drug pressure as the primary driver of these mutations, rather than mutations in epitope encoding genes, so the extent to which mutations in epitope encoding genes are specifically varying in response to host activity (or not varying) is not clearly related to the host response from the data as presented.</p></disp-quote><p>We thank the reviewer for raising several points in this comment. First to clarify, baseline drug resistance was inferred from the whole genome sequencing data using a well validated set of 177 mutations at an allele frequency threshold (&gt;40%). Selection of these mutations and validation of this allele frequency threshold was previously described (PMID <ext-link ext-link-type="uri" xlink:href="https://www-ncbi-nlm-nih-gov.ezp-prod1.hul.harvard.edu/pubmed/26910495">26910495</ext-link> and cited 90 times to date). This study made use of 1,319 clinical Mtb isolates with known drug resistance phenotypes. The data were randomly split into training and validation sets containing 67% and 33% of the isolates (respectively). The diagnostic set of mutations was determined using random forest predictive modeling in which a weighted model was run with serially smaller subsets of mutations to identify a minimal set of mutations to predict resistance to first- and second-line TB drugs. The resulting set of mutations predicted INH resistance with a sensitivity of 94% and specificity of 94% on the validation isolate set and predicted RIF resistance with a sensitivity of 93% and specificity of 95% on the validation isolate set. In response to the reviewer comments we now provide further details in subsection (“Pre-existing Genotypic Resistance”, have updated Supplementary file 2 to include columns that indicated genotypic susceptibility or resistance to Isoniazid and Rifampicin in the first collected sample for all subjects) and have included Supplementary file 21 that contains the genotypic resistance predictions for 13 antibiotics for all 614 longitudinal isolates from the 307 patients in our study.</p><p>With regards to the point raised about the observed evolutionary mutation rate over time, we want to note that prior publications relied on serial samples from patients receiving antibiotic treatment and measured a similar rate (Walker, 2013 and personal communication with the author). The measured genome-wide mutation rate has been consistent between non-human primate Mtb evolutionary experiments (Ford et al., <ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/21516081&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 21516081</ext-link>), Bayesian molecular clock estimation (Menardo, 2019,) and in-host pathogen evolution that largely come from patients receiving antibiotic treatment at least for some interval (Walker, <ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/27701423&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 27701423</ext-link>). This is likely the case as drug exposure can result in selective pressure but in only a short section of the genome, and averaging mutation rates across all regions of the genome, as is done for mutation rate calculations, and the stochastic nature of mutation accumulation at this short time scale washes out this localized increase in diversity. We also refer the referee to Response 2.5 where we describe an analysis measuring the mutation rate in the subset of patients confirmed to have treatment failure.</p><p>With regards to the third point about lack of sample timing, we note that sample collection dates were available for all patients analysed 119 patients with confirmed failure and 195 total, with clonal infection. In response to the reviewer comments and previous comments from reviewer 1 and 2, we reran the assessment of new AR mutation development in the subset confirmed to be treatment failure and relapse, i.e. excluding cases of possible default for which treatment durations and regimens are not well documented. We measure a point estimate of AR of 9%, (95% <italic>CI</italic> [5.2%, 15.8%] <italic>binconf</italic> function in R) compared with the rate of 15% (95% <italic>CI</italic> [10.6%, 21.2%] <italic>binconf</italic> function in R) we observed across all persistent clonal infections We note the substantial overlap of the confidence intervals for these point estimates. The association between resistance acquisition and MDR also held among the 178/200 subset of with persistent or relapsed infection &gt;2months (OR=3.85, <italic>P</italic> = 2.2x10-4 Fisher’s exact test) and among the 119/121 subjects with confirmed failure (OR=3.9, <italic>P</italic> = 4x10-2 Fisher’s exact test).</p><p>On the last point raised on in-host evolution of non-AR regions, we postulate that important drivers for evolution of other regions of the genome include host immunity and Mtb’s metabolic needs during chronic active infection. These selective forces may be independent of drug exposure and treatment regimen details as long as chronic infection, as evidenced by persistent bacterial growth from patient samples, is maintained. We specifically found evidence for selection on bacterial proteins involved with innate immune interactions and cobalamin biosynthesis proteins among other pathways described in the discussion. The study of these forces is very novel and has not been attempted previously at this scale, and we are the first to document evidence of non-AR based selection in host​​. We believe this is an important contribution to the literature on Mtb evolution and host-pathogen interactions more generally.</p><disp-quote content-type="editor-comment"><p>(3.3) Similarly, the authors note that mutations in drug resistance-associated loci is common and occurs across distinct Mtb lineages. While the finding of phylogenetic convergence is important, an alternative framing of the finding would be to consider these sites to vary in the presence of drug pressure independent of lineage.</p></disp-quote><p>We agree with the reviewer. Our test of phylogenetic convergence, the independent occurrence of certain mutations occurring in different genetic backgrounds, assesses whether a mutation arose multiple times independent of genetic background/lineage (annotated signature of positive selection). Our aim in this analysis was precisely the assessment of mutations and genes relevant to a phenotype independent of genetic background. This signal has previously been used to infer regions of positive selection in the context of antibiotic resistance (<ext-link ext-link-type="uri" xlink:href="https://hollis.harvard.edu/openurl/01HVD/HVD_URL?url_ver=Z39.88-2004&amp;rft_val_fmt=info:ofi/fmt:kev:mtx:journal&amp;__char_set=utf8&amp;rft_id=info:pmid/23995135&amp;rfr_id=info:sid/libx%3Ahul.harvard&amp;rft.genre=article">PMID: 23995135</ext-link>, cited 364 times). It may be possible that some background mutations or genes interact with newly acquired mutations to mediate the phenotype, but our approach is not designed to assess these situations.</p><disp-quote content-type="editor-comment"><p>(3.4) Of related concern, the authors defined study participants as having met criteria for failed therapy due to culture positivity after only 2 months of treatment. While patients receiving appropriate treatment ideally develop early culture conversion, many sources would require a longer time frame than 2 months to assign treatment failure, particularly if these participants were receiving experimental study regimens. Due to the heterogeneity introduced by the sample inclusion strategy, the conclusions of the manuscript with respect to negative treatment outcomes should be presented with greater weight placed on time between samples and time since starting therapy. Reframing the findings of the study as changes that occur during treatment, rather than changes occurring in the setting of failure, could also improve the support for the conclusions of the manuscript.</p></disp-quote><p>We thank the reviewer for raising this point. We refer the referee to Response 1.1 where we detail edits to the title and Responses 1.4 and 3.1 in which we detail how we generated a new and detailed patient treatment metadata Table (Supplementary file 2) that details treatment regimens received, and describes exactly when samples were collected relative to treatment initiation.</p><disp-quote content-type="editor-comment"><p>(3.5) For example, the authors identify a strong correlation between the SNP diversity of each participant's first sample and second sample and conclude that this demonstrates ineffective therapy (subsection “Determinants of antibiotic resistance acquisition and microbiological treatment failure”). How did this correlate among those participants who were not thought to have failed therapy? Did this vary with time to culture conversion among those converted later (that is, differentiating between those with &quot;appropriate culture conversion&quot;, &quot;delayed culture conversion,&quot; and true &quot;failure&quot;). A comparison between participants with eventual success and those who either never converted or who changed regimens due to emerging resistance would help differentiate this issue. Alternatively, a comparison between the SNP diversity among the 44 participants excluded due to different strains might better support the conclusion in the discussion that the sustained diversity identified is due to the absence of effective therapy.</p></disp-quote><p>We thank the reviewer for raising this point and we agree that these analyses would be insightful. Unfortunately, we are limited in the data available to us for the comparison groups. We only have 4 patients that are on effective therapy and the analysis of the SNP diversity between those patients’ first and second samples will be confounded by the much longer time duration of sampling for the failure cases. Furthermore, we don’t have data on time to culture conversion for our longitudinal samples, we only know that the majority were persistently culture positive at more than 2 months. A comparison between the SNP diversity between first and second samples among the 44 participants that have been classified as a reinfection (different strains) would be difficult to interpret as these populations are non-clonal and the diversity between them will be dominated by lineage or ancestral SNP differences.</p><disp-quote content-type="editor-comment"><p>(3.6) Finally, the Introduction is a bit confusing, combining discussions of the impact of host immunity, drug pressure, and microbiological and sequencing biases in selection of bacterial subpopulations. The result is that the reader is left confused about how to frame their interpretation of the study findings. This may be improved by a simpler introduction of the problem of unknown in-host variation over time and a summary of experiments that explain which bacterial factors should be considered at baseline to help predict future changes in Mtb isolates as well as the fixation of mutations in different genomic loci over time.</p></disp-quote><p>We thank the reviewer for their comment. We agree that the Introduction combined many discussions that interrupted the flow of the manuscript. To make the Introduction simpler and help with the flow, we have now revised it. The Introduction now flows from explaining the importance of studying the temporal dynamics of Mtb in-host, a short explanation of the importance in studying minor allele frequencies longitudinally for drug treatment, the barriers to studying the temporal dynamics of Mtb populations in-host using WGS, our snapshot of our sample and some of our main results.</p></body></sub-article></article>