<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">89862</article-id><article-id pub-id-type="doi">10.7554/eLife.89862</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.89862.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Ecology</subject></subj-group></article-categories><title-group><article-title>Microbes with higher metabolic independence are enriched in human gut microbiomes under stress</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name><surname>Veseli</surname><given-names>Iva</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-2390-5286</contrib-id><email>iva.veseli@gmail.com</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Chen</surname><given-names>Yiqun T</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4100-1507</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Schechter</surname><given-names>Matthew S</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8435-3203</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Vanni</surname><given-names>Chiara</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-1124-1147</contrib-id><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Fogarty</surname><given-names>Emily C</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8957-9922</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Watson</surname><given-names>Andrea R</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0128-6795</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Jabri</surname><given-names>Bana</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7427-4424</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Blekhman</surname><given-names>Ran</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-3218-613X</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Willis</surname><given-names>Amy D</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-2802-4317</contrib-id><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Yu</surname><given-names>Michael K</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9560-2017</contrib-id><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="fn" rid="con10"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Fernàndez-Guerra</surname><given-names>Antonio</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8679-490X</contrib-id><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="fn" rid="con11"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Füssel</surname><given-names>Jessika</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4210-2318</contrib-id><email>jessika.fuessel@uol.de</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="fn" rid="con12"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Eren</surname><given-names>A Murat</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9013-4827</contrib-id><email>meren@hifmb.de</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="aff" rid="aff10">10</xref><xref ref-type="aff" rid="aff11">11</xref><xref ref-type="aff" rid="aff12">12</xref><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con13"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/024mw5h28</institution-id><institution>Biophysical Sciences Program, The University of Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/024mw5h28</institution-id><institution>Department of Medicine, The University of Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Data Science Institute and Department of Biomedical Data Science, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/024mw5h28</institution-id><institution>Committee on Microbiology, The University of Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04ers2y35</institution-id><institution>MARUM Center for Marine Environmental Sciences, University of Bremen</institution></institution-wrap><addr-line><named-content content-type="city">Bremen</named-content></addr-line><country>Germany</country></aff><aff id="aff6"><label>6</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00cvxb145</institution-id><institution>Department of Biostatistics, University of Washington</institution></institution-wrap><addr-line><named-content content-type="city">Seattle</named-content></addr-line><country>United States</country></aff><aff id="aff7"><label>7</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02sn5gb64</institution-id><institution>Toyota Technological Institute at Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff8"><label>8</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/035b05819</institution-id><institution>Lundbeck Foundation GeoGenetics Centre, GLOBE Institute, University of Copenhagen</institution></institution-wrap><addr-line><named-content content-type="city">Copenhagen</named-content></addr-line><country>Denmark</country></aff><aff id="aff9"><label>9</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0060pja03</institution-id><institution>Institute for Chemistry and Biology of the Marine Environment, University of Oldenburg</institution></institution-wrap><addr-line><named-content content-type="city">Oldenburg</named-content></addr-line><country>Germany</country></aff><aff id="aff10"><label>10</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02385fa51</institution-id><institution>Marine ‘Omics Bridging Group, Max Planck Institute for Marine Microbiology</institution></institution-wrap><addr-line><named-content content-type="city">Bremen</named-content></addr-line><country>Germany</country></aff><aff id="aff11"><label>11</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00tea5y39</institution-id><institution>Helmholtz Institute for Functional Marine Biodiversity</institution></institution-wrap><addr-line><named-content content-type="city">Oldenburg</named-content></addr-line><country>Germany</country></aff><aff id="aff12"><label>12</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/032e6b942</institution-id><institution>Alfred Wegener Institute for Polar and Marine Research</institution></institution-wrap><addr-line><named-content content-type="city">Bremerhaven</named-content></addr-line><country>Germany</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Turnbaugh</surname><given-names>Peter J</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/043mz5j54</institution-id><institution>University of California, San Francisco</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Garrett</surname><given-names>Wendy S</given-names></name><role>Senior Editor</role><aff><institution>Harvard T.H. Chan School of Public Health</institution><country>United States</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>16</day><month>05</month><year>2025</year></pub-date><volume>12</volume><elocation-id>RP89862</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-06-14"><day>14</day><month>06</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-05-26"><day>26</day><month>05</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.05.10.540289"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-09-08"><day>08</day><month>09</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.89862.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-11-20"><day>20</day><month>11</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.89862.2"/></event></pub-history><permissions><copyright-statement>© 2023, Veseli et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Veseli et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-89862-v1.pdf"/><abstract><p>A wide variety of human diseases are associated with loss of microbial diversity in the human gut, inspiring a great interest in the diagnostic or therapeutic potential of the microbiota. However, the ecological forces that drive diversity reduction in disease states remain unclear, rendering it difficult to ascertain the role of the microbiota in disease emergence or severity. One hypothesis to explain this phenomenon is that microbial diversity is diminished as disease states select for microbial populations that are more fit to survive environmental stress caused by inflammation or other host factors. Here, we tested this hypothesis on a large scale, by developing a software framework to quantify the enrichment of microbial metabolisms in complex metagenomes as a function of microbial diversity. We applied this framework to over 400 gut metagenomes from individuals who are healthy or diagnosed with inflammatory bowel disease (IBD). We found that high metabolic independence (HMI) is a distinguishing characteristic of microbial communities associated with individuals diagnosed with IBD. A classifier we trained using the normalized copy numbers of 33 HMI-associated metabolic modules not only distinguished states of health vs IBD, but also tracked the recovery of the gut microbiome following antibiotic treatment, suggesting that HMI is a hallmark of microbial communities in stressed gut environments.</p></abstract><abstract abstract-type="plain-language-summary"><title>eLife digest</title><p>The human gut hosts an array of microbes that form a complex community beginning shortly after birth. These microbes prime the immune system, help extract nutrients from the diet and offer protection against pathogens. Decades of research have shown that individuals who suffer from inflammatory bowel diseases (IBD) or other systemic disorders tend to have far less variety of gut microbes compared to healthy individuals. Yet it remains unclear to what extent the difference in microbial diversity is the cause of the disease or a consequence of it.</p><p>In 2023, a study suggested that the usual teamwork between different kinds of microbes breaks down during disease. Many microbes depend on each other to provide certain nutrients, while others can survive on their own. It could be that people with IBD lose most of the ‘dependent’ microbes and retain those that are more self-sufficient and thus able to survive in the stressed and deteriorating gut environment.</p><p>To test this hypothesis, Veseli et al. – who are part of the research group that performed the 2023 study – developed a computer program to quantify self-sufficient gut microbes in large numbers of stool samples collected from healthy individuals and patients with IBD. This revealed that individuals with IBD had higher numbers of self-sufficient microbes, while healthy people also harbored microbes that depended on others for the provision of essential metabolites. External disruptions to the gut homeostasis, such as antibiotics, resulted in a similar selection for independent microbes.</p><p>These findings support the idea that changes in the gut microbiome are more likely a by-product of disease, rather than its cause and offer important ecological clues for microbial therapies that aim to restore gut health. While this perspective assigns a more neutral role for gut microbial communities in non-transmissible diseases, more research is needed to see if an enrichment of self-sufficient microbes could negatively influence disease progression.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>gut microbiome</kwd><kwd>microbial metabolism</kwd><kwd>metabolic reconstruction</kwd><kwd>inflammatory bowel disease</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100023581</institution-id><institution>National Science Foundation Graduate Research Fellowship Program</institution></institution-wrap></funding-source><award-id>1746045</award-id><principal-award-recipient><name><surname>Veseli</surname><given-names>Iva</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution>National Institutes of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R35 GM133420</award-id><principal-award-recipient><name><surname>Willis</surname><given-names>Amy D</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution>Stanford Data Science Postdoctoral Fellowship</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Chen</surname><given-names>Yiqun T</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35 GM128716</award-id><principal-award-recipient><name><surname>Blekhman</surname><given-names>Ran</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution>University of Chicago International Student Fellowship</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Fogarty</surname><given-names>Emily C</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>RC2 DK122394</award-id><principal-award-recipient><name><surname>Eren</surname><given-names>A Murat</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication. Open access funding provided by Max Planck Society.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Higher biosynthetic capacity of gut microbes in individuals diagnosed with noncommunicable diseases or taking antibiotics suggests that diversity loss and 'dysbiosis' result from microbiome restructuring in response to ecosystem disruption.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The human gut is home to a diverse assemblage of microbial cells that form complex communities (<xref ref-type="bibr" rid="bib36">Coyte et al., 2015</xref>). This gut microbial ecosystem is established almost immediately after birth and plays a lifelong role in human well-being by contributing to immune system maturation and functioning (<xref ref-type="bibr" rid="bib16">Belkaid and Hand, 2014</xref>; <xref ref-type="bibr" rid="bib137">Maynard et al., 2012</xref>), extracting dietary nutrients (<xref ref-type="bibr" rid="bib76">Hijova, 2019</xref>), providing protection against pathogens (<xref ref-type="bibr" rid="bib100">Khosravi and Mazmanian, 2013</xref>), metabolizing drugs (<xref ref-type="bibr" rid="bib237">Zimmermann et al., 2019</xref>), and more (<xref ref-type="bibr" rid="bib103">Knight et al., 2017</xref>). There is no universal definition of a healthy gut microbiome (<xref ref-type="bibr" rid="bib48">Fan and Pedersen, 2021</xref>), but associations between host disease states and changes in microbial community composition have sparked great interest in the therapeutic potential of gut microbes (<xref ref-type="bibr" rid="bib26">Cani, 2018</xref>; <xref ref-type="bibr" rid="bib201">Sorbara and Pamer, 2022</xref>) and led to the emergence of hypotheses that directly link disruptions of the gut microbiome to noncommunicable diseases of complex etiology (<xref ref-type="bibr" rid="bib25">Byndloss and Bäumler, 2018</xref>).</p><p>Inflammatory bowel diseases (IBDs), which describe a heterogeneous group of chronic inflammatory disorders (<xref ref-type="bibr" rid="bib192">Shan et al., 2022</xref>), represent an increasingly common health risk around the globe (<xref ref-type="bibr" rid="bib94">Kaplan, 2015</xref>). Understanding the role of gut microbiota in IBD has been a major area of focus in human microbiome research. Studies focusing on individual microbial taxa that typically change in relative abundance in IBD patients have proposed a range of host-microbe interactions that may contribute to disease manifestation and progression (<xref ref-type="bibr" rid="bib89">Joossens et al., 2011</xref>; <xref ref-type="bibr" rid="bib184">Schirmer et al., 2019</xref>; <xref ref-type="bibr" rid="bib73">Henke et al., 2019</xref>; <xref ref-type="bibr" rid="bib131">Machiels et al., 2014</xref>). However, even within well-constrained cohorts, a large proportion of variability in the taxonomic composition of the microbiota is unexplained, and the proportion of variability explained by disease status is low (<xref ref-type="bibr" rid="bib62">Gevers et al., 2014</xref>; <xref ref-type="bibr" rid="bib183">Schirmer et al., 2018</xref>; <xref ref-type="bibr" rid="bib126">Lloyd-Price et al., 2019</xref>; <xref ref-type="bibr" rid="bib99">Khan et al., 2019</xref>). As neither individual taxa nor broad changes in microbial community composition yield effective predictors of disease (<xref ref-type="bibr" rid="bib104">Knox et al., 2019</xref>; <xref ref-type="bibr" rid="bib115">Lee and Chang, 2021</xref>), the role of gut microbes in the etiology of IBD – or the extent to which they are bystanders to disease – remains unclear (<xref ref-type="bibr" rid="bib99">Khan et al., 2019</xref>).</p><p>The marked decrease in microbial diversity in IBD is often associated with the loss of Firmicutes populations and an increased representation of a relatively small number of taxa, such as Bacteroides, Enterococcaceae, and others (<xref ref-type="bibr" rid="bib167">Prindiville et al., 2000</xref>; <xref ref-type="bibr" rid="bib181">Saitoh et al., 2002</xref>; <xref ref-type="bibr" rid="bib182">Sartor, 2006</xref>; <xref ref-type="bibr" rid="bib176">Rhodes, 2007</xref>; <xref ref-type="bibr" rid="bib41">Devkota et al., 2012</xref>; <xref ref-type="bibr" rid="bib131">Machiels et al., 2014</xref>; <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>; <xref ref-type="bibr" rid="bib126">Lloyd-Price et al., 2019</xref>). Why a handful of taxa that also typically occur in healthy individuals in lower abundances (<xref ref-type="bibr" rid="bib115">Lee and Chang, 2021</xref>; <xref ref-type="bibr" rid="bib151">Nishida et al., 2018</xref>) tend to dominate the IBD microbiome is a fundamental but open question to gain insights into the ecological underpinnings of the gut microbial ecosystem under IBD. Going beyond taxonomic summaries, a recent metagenome-wide metabolic modeling study revealed a significant loss of cross-feeding partners as a hallmark of IBD, where microbial interactions were disrupted in IBD-associated microbial communities compared to those found in healthy individuals (<xref ref-type="bibr" rid="bib133">Marcelino et al., 2023</xref>). This observation is in line with another recent work that proposed that the extent of ‘metabolic independence’ (characterized by the genomic presence of a set of key metabolic modules for the synthesis of essential nutrients) is a determinant of microbial survival in IBD (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>). It is conceivable that the disrupted metabolic interactions among microbes observed in IBD (<xref ref-type="bibr" rid="bib133">Marcelino et al., 2023</xref>) indicate an environment that lacks the ecosystem services provided by a complex network of microbial interactions, and selects for those organisms that harness high metabolic independence (HMI) (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>). This interpretation offers an ecological mechanism to explain the dominance of populations with specific metabolic features in IBD and requires further investigation.</p><p>Here, we implemented a high-throughput, taxonomy-independent strategy to estimate metabolic capabilities of microbial communities directly from metagenomes and investigate whether the enrichment of populations with HMI predicts IBD in the human gut. We benchmarked our findings using representative genomes associated with the human gut and their distribution in healthy individuals as well as those who have been diagnosed with IBD. Our results suggest that high metabolic potential (indicated by a set of 33 largely biosynthetic metabolic modules) provides enough signal to consistently distinguish gut microbiomes under stress from those that are in homeostasis, providing deeper insights into adaptive processes initiated by stress conditions that promote rare members of gut microbiota to dominance during disease.</p></sec><sec id="s2" sec-type="results|discussion"><title>Results and discussion</title><p>We compiled 2893 publicly available stool metagenomes from 13 different studies, 5 of which explicitly studied the IBD gut microbiome (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1a–c</xref>). The average sequencing depth varied across individual datasets (4.2 to 60.3 million paired-end reads, with a median value of 21.4, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1c</xref>). To improve the sensitivity and accuracy of our downstream analyses that depend on metagenomic assembly, we excluded samples with less than 25 million reads, resulting in a set of 408 relatively deeply sequenced metagenomes from 10 studies (26.4 to 61.9 million paired-end reads, with a median value of 37.0, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1b</xref>, Appendix 1, Methods), which we de novo assembled individually. The final dataset included individuals who were healthy (n = 229), diagnosed with IBD (n = 101), or suffered from other gastrointestinal conditions (‘non-IBD’, n = 78). In accordance with previous observations of reduced microbial diversity in IBD (<xref ref-type="bibr" rid="bib108">Kostic et al., 2014</xref>; <xref ref-type="bibr" rid="bib149">Nagalingam and Lynch, 2012</xref>; <xref ref-type="bibr" rid="bib104">Knox et al., 2019</xref>), the estimated number of populations based on the occurrence of bacterial single-copy core genes (SCGs) present in these metagenomes was higher in healthy individuals than those diagnosed with IBD (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>).</p><sec id="s2-1"><title>Estimating normalized copy numbers of metabolic modules from metagenomic assemblies</title><p>Gaining insights into microbial metabolism requires accurate estimates of the presence/absence and completion of metabolic modules. While a myriad of tools address this task for single genomes (<xref ref-type="bibr" rid="bib130">Machado et al., 2018</xref>; <xref ref-type="bibr" rid="bib12">Aziz et al., 2008</xref>; <xref ref-type="bibr" rid="bib10">Arkin et al., 2018</xref>; <xref ref-type="bibr" rid="bib158">Palù et al., 2022</xref>; <xref ref-type="bibr" rid="bib189">Shaffer et al., 2020</xref>; <xref ref-type="bibr" rid="bib61">Geller-McGrath et al., 2023</xref>; <xref ref-type="bibr" rid="bib240">Zorrilla et al., 2021</xref>; <xref ref-type="bibr" rid="bib236">Zhou et al., 2022</xref>; <xref ref-type="bibr" rid="bib238">Zimmermann et al., 2021</xref>), working with complex environmental metagenomes poses additional challenges due to the large number of organisms that are present in metagenomic assemblies. A few tools can estimate community-level metabolic potential from metagenomes without relying on the reconstruction of individual population genomes or reference-based approaches (<xref ref-type="bibr" rid="bib232">Ye and Doak, 2009</xref>; <xref ref-type="bibr" rid="bib96">Karp et al., 2021</xref>; <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). These high-level summaries of module presence and redundancy in a given environment are suitable for most surveys of metabolic capacity, particularly for microbial communities of similar richness. However, since the frequency of observed metabolic modules will increase as the number of distinct microbial populations in a habitat increases, investigations of metabolic determinants of survival across environmental conditions with substantial differences in microbial richness may suffer from ambiguous observations from quantitative data. For instance, the estimated copy number of a given metabolic module may be identical between two metagenomes but its enrichment may be relatively higher in the metagenome with a lower alpha diversity, revealing its potential role in overcoming environment-specific selective pressures that influence an entire community. Working solely with raw copy numbers of metabolic modules without a normalization step that considers the microbial richness will thus shroud potentially critical insights. To quantify the differential enrichment of metabolic modules between metagenomes generated from healthy individuals and those from individuals diagnosed with IBD, we implemented a new software framework (<ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/anvi-estimate-metabolism">https://anvio.org/m/anvi-estimate-metabolism</ext-link>) that reconstructs metabolic modules from genomes and metagenomes, and a means to calculate the per-population copy number (PPCN) of modules in metagenomes to account for potential differences in microbial richness (Methods, Appendices 1 and 2). Briefly, the PPCN estimates the proportion of microbes in a community with a particular metabolic capacity (<xref ref-type="fig" rid="fig1">Figure 1</xref>, <xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>) by normalizing observed metabolic module copy numbers with the ‘number of microbial populations in a given metagenome’, which we estimate using the SCGs without relying on the reconstruction of individual genomes. Our validation of this method using simulated metagenomic data demonstrated that it is accurate in capturing metagenome-level metabolic capacity relative to genome-level metabolic capacity estimated from the same data (Appendix 2, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Conceptual diagram of per-population copy number (PPCN) calculation.</title><p>Each step of the calculation is demonstrated in (<bold>A</bold>) for a sample with high diversity (six microbial populations) and in (<bold>B</bold>) for a sample with low diversity (three populations). Metagenome sequences are shown as black lines. The left panel shows the single-copy core genes (SCGs) annotated in the metagenome (indicated by letters), with a barplot showing the counts for different SCGs. The dashed black line indicates the mode of the counts, which is taken as the estimate of the number of populations. The middle panel shows the annotations of metabolic modules (indicated by boxes and numerically labeled), with a barplot showing the copy number of each module (for more details on how this copy number is computed, see Appendix 1 and <xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>). The right panel shows the equation for PPCN, with the barplots indicating the PPCN values for each metabolic module in each sample and arrows differentiating between different types of modules based on the comparison of their normalized copy numbers between samples.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-fig1-v1.tif"/></fig></sec><sec id="s2-2"><title>Key biosynthetic modules are enriched in microbial populations from IBD samples</title><p>To gain insight into potential metabolic determinants of microbial survival in the IBD gut environment, we assessed the distribution of metabolic modules within samples from each group (IBD and healthy) with and without using PPCN normalization. Without normalizing, module copy numbers were overall higher in healthy samples (<xref ref-type="fig" rid="fig2">Figure 2A</xref>) and modules exhibited weak differential occurrence between cohorts (<xref ref-type="fig" rid="fig2">Figure 2B and C</xref>, <xref ref-type="fig" rid="app1fig3">Appendix 1—figure 3</xref>). The application of PPCN reversed this trend, and most metabolic modules were elevated in IBD (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5</xref>). This observation is influenced by two independent aspects of the healthy and IBD microbiota. The first one is the increased representation of microbial organisms with smaller genomes in healthy individuals (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>), which increases the likelihood that the overall copy number of a given metabolic module is below the actual number of populations. In contrast, one of the hallmarks of the IBD microbiota is the generally increased representation of organisms with larger genomes (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>). The second aspect is that the generally higher diversity of microbes in healthy individuals increases the denominator of the PPCN. This results in a greater reduction in the PPCN of metabolic modules that are not shared across all members of the diverse gut microbial populations in health.</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Comparison of metabolic potential across healthy and inflammatory bowel disease (IBD) cohorts.</title><p>Panels <bold>A–C</bold> show unnormalized copy number data and the remaining panels show normalized per-population copy number (PPCN) data. (<bold>A</bold>) Scatterplot of module copy number in IBD samples (x-axis) and healthy samples (y-axis). Transparency of points indicates the p-value of the module in a Wilcoxon rank-sum test for enrichment (based on copy number data), and color indicates whether the module is enriched in the IBD samples (in this study), enriched in the good colonizers from the fecal microbiota transplant (FMT) study (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>), or enriched in both. The pink dashed line indicates the effect size threshold applied to modules when determining their enrichment in IBD. (<bold>B</bold>) Heatmap of unnormalized copy numbers for all modules. The 33 modules that were found to be IBD-enriched based on PPCN data are highlighted by the red bar on the left. Sample group is indicated by the blue (healthy) and red (IBD) bars on the bottom. (<bold>C</bold>) Boxplots of median copy number for each module enriched in the FMT colonizers from <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>, in the healthy samples (blue) and the IBD samples (red). Solid lines connect the same module in each plot. (<bold>D</bold>) Scatterplot of module PPCN values in IBD samples (x-axis) and healthy samples (y-axis). Transparency and color of points are defined as in panel <bold>A</bold>, but based on PPCN data. The pink dashed line indicates the effect size threshold applied to modules when determining their enrichment in IBD. (<bold>E</bold>) Heatmap of PPCN values for all modules. Side bars defined as in (<bold>B</bold>). (<bold>F</bold>) Boxplots of median PPCN values for modules enriched in the FMT colonizers from <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>, in the healthy samples (blue) and the IBD samples (red). Lines defined as in (<bold>D</bold>). Modules that were also enriched in the IBD samples (in this study) are highlighted in red. (<bold>G</bold>) Boxplots of PPCN values for individual modules in the healthy samples (blue) and the IBD samples (red). All example modules were enriched in both this study and in <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-fig2-v1.tif"/></fig><p>To go beyond this general trend and identify modules that were highly conserved in the IBD group, we first selected those that passed a relatively high statistical significance threshold in our enrichment test (Wilcoxon rank-sum test, FDR-adjusted p-value &lt; 2e-10). We then accounted for effect size by ranking these modules according to the difference between their median PPCN in IBD samples and their median PPCN in healthy samples, and keeping only those in the top 50% (which translated to an effect size threshold of &gt; 0.12). This stringent filtering revealed a set of 33 metabolic modules that were significantly enriched in metagenomes obtained from individuals diagnosed with IBD (<xref ref-type="fig" rid="fig2">Figure 2D and E</xref>), 17 of which matched the modules that were associated with HMI previously (<xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>; <xref ref-type="fig" rid="fig2">Figure 2F</xref>). This result suggests that the PPCN normalization is an important step in comparative analyses of metabolisms between samples with different levels of microbial diversity.</p><p>The majority of the metabolic modules that were significantly enriched in the microbiomes of IBD patients encoded biosynthetic capabilities (23 out of 33) that resolved to amino acid metabolism (33%), carbohydrate metabolism (21%), cofactor and vitamin biosynthesis (15%), nucleotide biosynthesis (12%), lipid biosynthesis (6%), and energy metabolism (6%) (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2a</xref>). In contrast to previous reports based on reference genomes (<xref ref-type="bibr" rid="bib62">Gevers et al., 2014</xref>; <xref ref-type="bibr" rid="bib146">Morgan et al., 2012</xref>), amino acid synthesis and carbohydrate metabolism were not reduced in the IBD gut microbiome in our dataset. Rather, our results were in accordance with a more recent finding that predicted amino acid secretion potential is increased in the microbiomes of individuals with IBD (<xref ref-type="bibr" rid="bib71">Heinken et al., 2021</xref>).</p><p>Within our set of 33 modules that were enriched in IBD, it is notable that all the biosynthesis and central carbohydrate modules are directly or indirectly linked via shared enzymes and metabolites. Each enriched module shared on average 25.6% of its enzymes and 40.2% of metabolites with the other enriched modules, and overall 18.2% of enzymes and 20.4% of compounds across these modules were shared (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2a</xref>). Thus, modules may be enriched not just due to the importance of their immediate end products, but also because of their role in the larger metabolic network. The few standalone modules that were enriched included the efflux pump MepA and the beta-lactam resistance system, which are associated with drug resistance. These capacities may provide an advantage since antibiotics are a common treatment for IBDs (<xref ref-type="bibr" rid="bib152">Nitzan et al., 2016</xref>), but are not necessarily related to the systematic enrichment of biosynthesis modules that likely provide resilience to general environmental stress rather than to a specific stressor such as antibiotics.</p><p>Microbiome data generated by different groups can result in systematic biases that may outweigh biological differences between otherwise similar samples (<xref ref-type="bibr" rid="bib129">Lozupone et al., 2013</xref>; <xref ref-type="bibr" rid="bib197">Sinha et al., 2017</xref>; <xref ref-type="bibr" rid="bib33">Clausen and Willis, 2022</xref>). The potential impact of such biases constitutes an important consideration for meta-analyses such as ours that analyze publicly available metagenomes from multiple sources. To account for cohort biases, we conducted an analysis of our data on a per-cohort basis. All cohorts within a given group exhibited similar distributions of PPCN values, which indicates that the trends we observed above result from an overall between-group difference in signal rather than a cohort-specific signal (<xref ref-type="fig" rid="app1fig6">Appendix 1—figure 6B and C</xref>). Another source of potential bias stems from the annotation efficiency of gene function. For instance, we noticed that, independent of the annotation strategy, a smaller proportion of genes resolved to known functions in metagenomic assemblies of samples from healthy individuals compared to the samples from individuals who were diagnosed with IBD (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4</xref>). This highlights the possibility that samples from healthy individuals merely appear to harbor less metabolic capabilities due to missing annotations. Indeed, we found that the normalized copy numbers of most metabolic modules were reduced in the healthy group, where 84% of KEGG modules (98 out of 118) have significantly lower median copy numbers (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5C</xref>, Appendix 1). While the presence of a bias between the two cohorts is clear, the source of this bias and its implications are not. One hypothesis that could explain this phenomenon is that the increased proportion of unknown functions in environments where populations with low metabolic independence (LMI) thrive is due to our inability to identify distant homologs of even well-studied functions in poorly studied novel genomes through public databases. If true, this would indeed impair our ability to annotate genes using state-of-the-art functional databases and bias metabolic module completion estimates. Such a limitation would warrant a careful reconsideration of common workflows and studies that rely on public resources to characterize gene function in complex environments. Another hypothesis that could explain our observation is that the general absence of microbes with smaller genomes in culture had a historical impact on the characterization of novel functions that represent a relatively larger fraction of their gene repertoire. If true, this would suggest that the unknown functions are unlikely essential for well-studied metabolic capabilities. Furthermore, HMI and LMI genomes may be indistinguishable with respect to the distribution of such novel genes, but the increased number of genes in HMI genomes that resolve to well-studied metabolisms would reduce the proportion of known functions in LMI genomes, and thus in metagenomes where they thrive. While testing these hypotheses falls outside the scope of our work, we find the latter hypothesis more likely due to examples in literature that have successfully identified genes that belong to known metabolisms in some of the most obscure organisms via annotation strategies similar to those we have used in our work (<xref ref-type="bibr" rid="bib84">Jaffe et al., 2020</xref>; <xref ref-type="bibr" rid="bib50">Farag et al., 2020</xref>).</p><p>Taken together, these results (1) demonstrate that the PPCN normalization is an important consideration for investigations of metabolic enrichment in complex microbial communities as a function of microbial diversity, and (2) reveal that the enrichment of HMI populations in an environment offers a high-resolution marker to resolve different levels of environmental stress.</p></sec><sec id="s2-3"><title>Reference genomes with higher metabolic independence are overrepresented in the gut metagenomes of individuals with IBD</title><p>So far, our findings demonstrate an overall, metagenome-level trend of increasing HMI within gut microbial communities as a function of IBD status without considering the individual genomes that contribute to this signal. Since we can measure the extent of metabolic independence as defined in our study based on the completion of a few key metabolic modules for any given genome, we next considered a genome-based approach to further benchmark our findings by investigating whether publicly available microbial genomes that appear to have properties of HMI are more commonly found in individuals diagnosed with IBD.</p><p>To identify a set of microbial genomes that are generally associated with the human gut environment, we cast a broad net by surveying the ecology of 19,226 genomes in the Genome Taxonomy Database (GTDB) (<xref ref-type="bibr" rid="bib162">Parks et al., 2022</xref>) that belonged to three major phyla: Bacteroidetes, Firmicutes, and Proteobacteria, which represent the vast majority of microbial diversity in the human gut environment (<xref ref-type="bibr" rid="bib226">Woting and Blaut, 2016</xref>; <xref ref-type="bibr" rid="bib210">Turnbaugh et al., 2009</xref>). As these phyla also include a large number of taxa that primarily occur outside of the human gut, we only kept for downstream analyses those that were detected in at least 2% of the participants of the Human Microbiome Project (HMP) (<xref ref-type="bibr" rid="bib80">Human Microbiome Project Consortium, 2012</xref>; <xref ref-type="fig" rid="app1fig8">Appendix 1—figure 8</xref>, Methods). Of the final set of 338 reference genomes that passed our filters, 258 (76.3%) resolved to Firmicutes, 60 (17.8%) to Bacteroidetes, and 20 (5.9%) to Proteobacteria. Most of these genomes resolved to families common to the colonic microbiota, such as Lachnospiraceae (30.0%), Ruminococcaceae/Oscillospiraceae (23.1%), and Bacteroidaceae (10.1%) (<xref ref-type="bibr" rid="bib11">Arumugam et al., 2011</xref>), while 5.9% belonged to poorly studied families with temporary code names (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3a</xref>). Finally, we performed a more comprehensive read recruitment analysis on this smaller set of genomes using all deeply sequenced metagenomes from cohorts that included healthy, non-IBD, and IBD samples (<xref ref-type="fig" rid="fig3">Figure 3</xref>). This provided us with a quantitative summary of the detection patterns of GTDB genome representatives common to the human gut across our dataset.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Identification of high metabolic independence (HMI) genomes and their distribution across gut samples.</title><p>(<bold>A</bold>) The phylogeny of 338 gut-associated genomes from the Genome Taxonomy Database (GTDB) along with the following data, from top to bottom: taxonomic classification as assigned by GTDB; proportion of healthy samples with at least 50% detection of the genome sequence; proportion of inflammatory bowel disease (IBD) samples with at least 50% detection of the genome sequence; square-root normalized ratio of percent abundance in IBD samples to percent abundance in healthy samples; metabolic independence score (sum of completeness scores of 33 HMI-associated metabolic modules); whether (red) or not (white) the genome is classified as having HMI with a threshold score of 26.4; heatmap of completeness scores for each of the 33 HMI-associated metabolic modules (0% completeness is white and 100% completeness is black). Module name is shown on the right and colored according to its category of metabolism. (<bold>B</bold>) Boxplot showing the proportion of healthy (blue) or IBD (red) samples in which genomes of each class are detected ≥ 50%, with p-values from a Wilcoxon rank-sum test on the underlying data. (<bold>C</bold>) Barplot showing the proportion of detected genomes (with ≥ 50% genome sequence covered by at least 1 read) in each sample that are classified as HMI, for each group of samples. The black lines show the median for each group: 37.0% for IBD samples, 25.5% for non-IBD samples, and 18.4% for healthy samples.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-fig3-v1.tif"/></fig><p>We assumed that a given genome had HMI if its average completeness of the 33 HMI-associated metabolic modules was at least 80%, equivalent to a summed metabolic independence score of 26.4 (Methods). Given the number of ways a genome can pass or fail this threshold, this arbitrary cutoff has significant shortcomings, which was demonstrated by the fact that several species in the <italic>Bacteroides</italic> group were not classified as HMI despite their frequent dominance of the gut microbiome of individuals with IBD (<xref ref-type="bibr" rid="bib181">Saitoh et al., 2002</xref>; <xref ref-type="bibr" rid="bib222">Wexler, 2007</xref>; <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>) (Appendix 1). That said, the genomes that were classified as HMI by this approach were consistently higher in their detection and abundance in IBD samples (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). It is likely that there are multiple ways to have HMI which are not fully captured by the 33 IBD-enriched metabolic modules identified in this study. Across all genomes, the mean metabolic independence score was 24.0 (Q1: 19.9, Q3: 25.7). We identified 17.5% (59) of the reference genomes as HMI. HMI genomes were on average substantially larger (3.8 Mbp) than non-HMI genomes (2.9 Mbp) and encoded more genes (3634 vs 2683 genes, respectively), which is in accordance with the reduced metabolic potential of non-HMI populations (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3a</xref>). Our read recruitment analysis showed that HMI reference genomes were present in a significantly higher proportion of IBD samples compared to non-HMI genomes (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, p &lt; 1e-5, Wilcoxon rank-sum test). Similarly, the fraction of HMI populations was significantly higher within a given IBD sample compared to samples classified as ‘non-IBD’ and those from healthy individuals (<xref ref-type="fig" rid="fig3">Figure 3C</xref>, p &lt; 1e-24, Kruskal-Wallis rank-sum test). In contrast, the detection of HMI populations and non-HMI populations was similar in healthy individuals (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, p = 0.267, Wilcoxon rank-sum test). The intestinal environment of healthy individuals likely supports both HMI and non-HMI populations, wherein ‘metabolic diversity’ is maintained by metabolic interactions such as cross-feeding. Indeed, loss of cross-feeding interactions in the gut microbiome appears to be associated with a number of human diseases, including IBD (<xref ref-type="bibr" rid="bib133">Marcelino et al., 2023</xref>). This interpretation is further supported by the fact that the top two HMI-associated modules are required for the synthesis of cobalamin from glutamate. Auxotrophy for cobalamin biosynthesis is common among gut bacteria that rely on cross-feeding for this essential cofactor (<xref ref-type="bibr" rid="bib38">Degnan et al., 2014a</xref>; <xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>; <xref ref-type="bibr" rid="bib98">Kelly et al., 2019</xref>) (Appendix 1).</p><p>Overall, the classification of reference gut genomes as HMI and their enrichment in individuals diagnosed with IBD strongly supports the contribution of HMI to stress resilience of individual microbial populations. We note that survival in a disturbed gut environment will likely require a wide variety of additional functions that are not covered in the list of metabolic modules we consider to determine HMI status – e.g., see <xref ref-type="bibr" rid="bib39">Degnan et al., 2014b</xref>; <xref ref-type="bibr" rid="bib134">Martens et al., 2014</xref>; <xref ref-type="bibr" rid="bib239">Zong et al., 2020</xref>; <xref ref-type="bibr" rid="bib55">Feng et al., 2020</xref>; <xref ref-type="bibr" rid="bib65">Goodman et al., 2009</xref>; <xref ref-type="bibr" rid="bib166">Powell et al., 2016</xref>. Indeed, there may be many ways for a microbe to be metabolically independent, and our strategy likely failed to identify some HMI populations. Nonetheless, these data suggest that HMI serves as a reliable proxy for the identification of microbial populations that are particularly resilient.</p></sec><sec id="s2-4"><title>HMI-associated metabolic potential predicts general stress on gut microbes</title><p>Our analysis identified HMI as an emergent property of gut microbial communities associated with individuals diagnosed with IBD. This community-level signal translates to individual microbial populations and provides insights into the microbial ecology of stressed gut environments. HMI-associated metabolic modules were enriched at the community level, and microbial populations encoding these modules were more prevalent in individuals with IBD than in healthy individuals. Furthermore, the copy number of these modules and the proportion of HMI populations reflect the severity of environmental stress and translate to host health states (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5B</xref>, <xref ref-type="fig" rid="fig3">Figure 3C</xref>). The ecological implications of these observations suggest that HMI may serve as a predictor of general stress in the human gut environment.</p><p>So far, efforts to identify IBD using microbial markers have presented classifiers based on (1) taxonomy in pediatric IBD patients (<xref ref-type="bibr" rid="bib159">Papa et al., 2012</xref>; <xref ref-type="bibr" rid="bib62">Gevers et al., 2014</xref>), (2) community composition in combination with clinical data (<xref ref-type="bibr" rid="bib68">Halfvarson et al., 2017</xref>), (3) untargeted metabolomics and/or species-level relative abundance from metagenomes (<xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>), and (4) k-mer-based sequence variants in metagenomes that can be linked to microbial genomes associated with IBD (<xref ref-type="bibr" rid="bib175">Reiter et al., 2022</xref>). Performance varied both between and within studies according to the target classes and data types used for training and validation of each classifier (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4a</xref>). For those studies reporting accuracy, a maximum accuracy of 77% was achieved based on either metabolite profiles (for prediction of IBD subtype) (<xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>) or k-mer-based sequence variants (for differentiating between IBD and non-IBD samples) (<xref ref-type="bibr" rid="bib175">Reiter et al., 2022</xref>). Some studies reported performance as area under the receiver operating characteristic curve (AUROCC), a typical measure of classifier utility describing both sensitivity (ability to correctly identify the disease) and specificity (ability to correctly identify absence of disease). For this metric the highest value was 0.92, achieved by <xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>, when using metabolite profiles, with or without species abundance data, for classifying IBD vs non-IBD. However, the majority of these classifiers were trained and tested on a relatively small group of individuals that all come from the same region, i.e., clinical studies confined to a specific hospital. Though some had high performance, they either relied on data that are inaccessible to most laboratories and clinics considering that untargeted metabolomics analyses are difficult to reproduce (<xref ref-type="bibr" rid="bib105">Koek et al., 2011</xref>; <xref ref-type="bibr" rid="bib121">Lin et al., 2020</xref>), or they required complex k-mer-based models without the resolution to differentiate gradients in host health (<xref ref-type="bibr" rid="bib175">Reiter et al., 2022</xref>). These classifiers thus have limited translational potential across global clinical settings and do not provide an ecological framework to explain the observed shifts in community composition and activity. For practical use as a diagnostic tool, a microbiome-based classifier for IBD should rely on an ecologically meaningful, easy to measure, and high-level signal that is robust to host variables like lifestyle, geographical location, and ethnicity. HMI could potentially fill this gap as a metric related to the ecological filtering that defines microbial community changes in the IBD gut microbiome.</p><p>We trained a logistic regression classifier to explore the applicability of HMI as a noninvasive diagnostic tool for IBD. The classifier’s predictors were the PPCNs of IBD-enriched metabolic modules in a given metagenome. Across the 330 deeply sequenced IBD and healthy samples included in this analysis, the classifier had high sensitivity and specificity (<xref ref-type="fig" rid="fig4">Figure 4</xref>). It correctly identified (on average) 76.8% of samples from individuals diagnosed with IBD and 89.5% of samples representing healthy individuals, for an overall accuracy of 85.6% and an average AUROCC of 0.832 (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4c</xref>). Our model outperforms (<xref ref-type="bibr" rid="bib62">Gevers et al., 2014</xref>; <xref ref-type="bibr" rid="bib68">Halfvarson et al., 2017</xref>; <xref ref-type="bibr" rid="bib175">Reiter et al., 2022</xref>) or has comparable performance to <xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>; <xref ref-type="bibr" rid="bib159">Papa et al., 2012</xref> the previous attempts to classify IBD from fecal samples in more restrictively defined cohorts. It also has the advantage of being a simple model, utilizing a relatively low number of features compared to the other classifiers. Determining whether such a model has broader utility as a diagnostic tool requires further research and validation; however, these results demonstrate the potential of HMI as an accessible diagnostic marker of IBD. Due to the lack of time-series studies that include individuals in the pre-diagnosis phase of IBD development, we cannot test the applicability of HMI to predict IBD onset (<xref ref-type="bibr" rid="bib126">Lloyd-Price et al., 2019</xref>).</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Performance of our metagenome classifier trained on per-population copy numbers (PPCNs) of inflammatory bowel disease (IBD)-enriched modules.</title><p>(<bold>A</bold>) Receiver operating characteristic (ROC) curves for 25-fold cross-validation. Each fold used a random subset of 80% of the data for training and the other 20% for testing. In each fold, we calculated a set of IBD-enriched modules from the training dataset and used the PPCN of these modules to train a logistic regression model whose performance was evaluated using the test dataset. Light gray lines show the ROC curve for each fold, the dark blue line shows the mean ROC curve, the gray area delineates the confidence interval for the mean ROC, and the pink dashed line indicates the benchmark performance of a naive (random guess) classifier. (<bold>B</bold>) Confusion matrix for each fold of the random cross-validation. Categories of classification, from top left to bottom right, are: true positives (correctly classified IBD samples), false positives (incorrectly classified healthy samples), false negatives (incorrectly classified IBD samples), and true negatives (correctly classified healthy samples). Each fold is represented by a box within each category. Opacity of the box indicates the proportion of samples in that category, and the actual proportion is written within the box with one significant digit. Underlying data for this matrix can be accessed in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4d</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-fig4-v1.tif"/></fig><p>Yet, the gradient of metabolic independence reflected by per-population module copy number and the relative increase in the number of HMI populations detected in non-IBD samples (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5B</xref>, <xref ref-type="fig" rid="fig3">Figure 3C</xref>) suggests that the degree of HMI in the gut microbiome may be indicative of general gut stress, such as the stress induced by antibiotic use. Antibiotics can cause long-lasting perturbations of the gut microbiome – including reduced diversity, emergence of opportunistic pathogens, increased microbial load, and development of highly resistant strains – with potential implications for host health (<xref ref-type="bibr" rid="bib171">Ramirez et al., 2020</xref>). We applied our metabolism classifier to a metagenomic dataset that reflects the changes in the microbiome of healthy people before, during, and up to 6 months following a 4-day antibiotic treatment (<xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>). The resulting pattern of sample classification corresponds to the posttreatment decline and subsequent recovery of species richness documented in the study by <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>.</p><p>All pretreatment samples were classified as ‘healthy’ followed by a decline in the proportion of ‘healthy’ samples to a minimum 8 days posttreatment, and a gradual increase until 180 days posttreatment, when over 90% of samples were classified as ‘healthy’ (<xref ref-type="fig" rid="fig5">Figure 5</xref>, <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4b</xref>). In other words, the increase in the HMI metric serves as an indicator of stress in the gut microbiome, regardless of whether that stress arises from the IBD condition or the application of antibiotics. These observations support the role of HMI as an ecological driver of microbial resilience during gut stress caused by a variety of environmental perturbations and demonstrate its diagnostic power in reflecting gut microbiome state.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Classification results on an antibiotic time-series dataset from <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>.</title><p>Note that antibiotic treatment was taken on days 1–4. (<bold>A</bold>) Samples collected per subject during the time series. (<bold>B</bold>) Species richness data (figure created using data from <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>). (<bold>C</bold>) Classification of each sample by the metabolism classifier profiled in <xref ref-type="fig" rid="fig4">Figure 4</xref>. Samples with insufficient sequencing depth were not classified. (<bold>D</bold>) Proportion of classes assigned to samples per day in the time series. Samples classified as ‘healthy’ by the model were considered to have ‘no stress’ (blue), while samples classified as inflammatory bowel disease (‘IBD’) were considered to be under ‘stress’ (red).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-fig5-v1.tif"/></fig></sec><sec id="s2-5"><title>Conclusions</title><p>Overall, our observations that stem from the analysis of hundreds of reference genomes, deeply sequenced gut metagenomes, and multiple categories of human disease states suggest that environmental stress in the human gut – whether it is associated with inflammation, cancer, or antibiotic use – promotes the survival and relative expansion of microbial populations with HMI. These results establish HMI as a high-level metric to classify gradients of human health states through the gut microbiota that is robust to ethnic, geographical, or lifestyle factors. Taken together with recent evidence that models altered ecological relationships within gut microbiomes under stress due to disrupted metabolic cross-feeding (<xref ref-type="bibr" rid="bib71">Heinken et al., 2021</xref>; <xref ref-type="bibr" rid="bib133">Marcelino et al., 2023</xref>), our data support the hypothesis that the reduction in microbial diversity, or more generally ‘dysbiosis’, is an emergent property of microbial communities responding to disease pathogenesis or other external factors such as antibiotic use that disrupt the gut microbial ecosystem. This paradigm depicts microbes as bystanders by default, rather than perpetrators or drivers of noncommunicable human diseases, and provides an ecological framework to explain the frequently observed reduction in microbial diversity associated with IBD and other noncommunicable human diseases and disorders.</p></sec></sec><sec id="s3" sec-type="methods"><title>Methods</title><p>A bioinformatics workflow that further details all analyses described below and gives access to reproducible data products is available at the URL <ext-link ext-link-type="uri" xlink:href="https://merenlab.org/data/ibd-gut-metabolism/">https://merenlab.org/data/ibd-gut-metabolism/</ext-link>.</p><sec id="s3-1"><title>A new framework for metabolism estimation</title><p>We developed a new program ‘anvi-estimate-metabolism’ (<ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/anvi-estimate-metabolism">https://anvio.org/m/anvi-estimate-metabolism</ext-link>), which uses gene annotations to estimate ‘completeness’ and ‘copy number’ of metabolic modules that are defined in terms of enzyme accession numbers. By default, this tool works on metabolic modules from the KEGG MODULE database (<xref ref-type="bibr" rid="bib92">Kanehisa et al., 2012</xref>; <xref ref-type="bibr" rid="bib93">Kanehisa et al., 2023</xref>) which are defined by KEGG KOfams (<xref ref-type="bibr" rid="bib9">Aramaki et al., 2020</xref>), but user-defined modules based on a variety of functional annotation sources are also accepted as input. Completeness estimates describe the percentage of steps (typically, enzymatic reactions) in a given metabolic module that are encoded in a genome or a metagenome. Likewise, copy number summarizes the number of distinct sets of enzyme annotations that collectively encode the complete module. This program offers two strategies for estimating metabolic potential: a ‘stepwise’ strategy with equivalent treatment for alternative enzymes – i.e., enzymes that can catalyze the same reaction in a given metabolic module – and a ‘pathwise’ strategy that accounts for all possible variations of the module. Appendix 1 file includes more information on these two strategies and the completeness/copy number calculations. For the analysis of metagenomes, we used stepwise copy number of KEGG modules. Briefly, the calculation of stepwise copy number is done as follows: the copy number of each step in a module (typically, one chemical reaction or conversion) is individually evaluated by translating the step definition into an arithmetic expression that summarizes the number of annotations for each required enzyme. In cases where multiple enzymes or an enzyme complex are needed to catalyze the reaction, we take the minimum number of annotations across these components. In cases where there are alternative enzymes that can each catalyze the reaction individually, we sum the number of annotations for each alternative. Once the copy number of each step is computed, we then calculate the copy number of the entire module by taking the minimum copy number across all the individual steps. The use of minimums results in a conservative estimate of module copy number such that only copies of the module with all enzymes present are counted. For the analysis of genomes, we calculated the stepwise completeness of KEGG modules. This calculation is similar to the one described above for copy number, except that the step definition is translated into a Boolean expression that, once evaluated, indicates the presence or absence of each step in the module. Then, the completeness of the modules is computed as the proportion of present steps in the module.</p></sec><sec id="s3-2"><title>Metagenomic datasets and sample groups</title><p>We acquired publicly available gut metagenomes from 13 different studies (<xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>; <xref ref-type="bibr" rid="bib54">Feng et al., 2015</xref>; <xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>; <xref ref-type="bibr" rid="bib126">Lloyd-Price et al., 2019</xref>; <xref ref-type="bibr" rid="bib168">Qin et al., 2012</xref>; <xref ref-type="bibr" rid="bib169">Quince et al., 2015</xref>; <xref ref-type="bibr" rid="bib172">Rampelli et al., 2015</xref>; <xref ref-type="bibr" rid="bib173">Raymond et al., 2016</xref>; <xref ref-type="bibr" rid="bib183">Schirmer et al., 2018</xref>; <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>; <xref ref-type="bibr" rid="bib212">University of Sydney, 2022</xref>; <xref ref-type="bibr" rid="bib221">Wen et al., 2017</xref>; <xref ref-type="bibr" rid="bib229">Xie et al., 2016</xref>). The studies were chosen based on the following criteria: (1) they included shotgun metagenomes of fecal matter (primarily stool, but some ileal pouch luminal aspirate samples (<xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>) are also included); (2) they sampled from people living in industrialized countries in the case where a study (<xref ref-type="bibr" rid="bib172">Rampelli et al., 2015</xref>) included samples from hunter-gatherer populations, only the samples from industrialized areas were included in our analysis; (3) they included samples from people with IBD and/or they included samples from people without gastrointestinal (GI) disease or inflammation; and (4) clear metadata differentiating between case and control samples was available. A full description of the studies and samples can be found in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1a–c</xref>. We grouped samples according to the health status of the sample donor. Briefly, the ‘IBD’ group of samples includes those from people diagnosed with Crohn’s disease, ulcerative colitis (UC), or pouchitis. The ‘non-IBD’ group contains non-IBD controls, which includes both healthy people presenting for routine cancer screenings and people with benign or nonspecific symptoms that are not clinically diagnosed with IBD. Colorectal cancer patients from <xref ref-type="bibr" rid="bib54">Feng et al., 2015</xref>, were also put into the ‘non-IBD’ group on the basis that tumors in the GI tract may arise from local inflammation (<xref ref-type="bibr" rid="bib109">Kraus and Arber, 2009</xref>) and represent a source of gut stress without an accompanying diagnosis of IBD. Finally, the ‘HEALTHY’ group contains samples from people without GI-related diseases or inflammation. Note that only control or pretreatment samples were taken from the studies covering type 2 diabetes (<xref ref-type="bibr" rid="bib168">Qin et al., 2012</xref>), ankylosing spondylitis (<xref ref-type="bibr" rid="bib221">Wen et al., 2017</xref>), antibiotic treatment (<xref ref-type="bibr" rid="bib173">Raymond et al., 2016</xref>), and dietary intervention <xref ref-type="bibr" rid="bib212">University of Sydney, 2022</xref>; these controls were all assigned to the ‘HEALTHY’ group. At least one study (<xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>) included samples from obese people, and these were also included in the ‘HEALTHY’ group.</p></sec><sec id="s3-3"><title>Processing of metagenomes</title><p>We made single assemblies of most gut metagenomes using the anvi’o metagenomics workflow implemented in the program ‘anvi-run-workflow’ (<xref ref-type="bibr" rid="bib191">Shaiber et al., 2020</xref>). This workflow uses Snakemake (<xref ref-type="bibr" rid="bib107">Köster and Rahmann, 2012</xref>), and a tutorial is available at the URL <ext-link ext-link-type="uri" xlink:href="https://merenlab.org/anvio-workflows/">https://merenlab.org/anvio-workflows/</ext-link>. Briefly, the workflow includes quality filtering using ‘iu-filter-quality-minoche’ (<xref ref-type="bibr" rid="bib46">Eren et al., 2013</xref>); assembly with IDBA-UD (<xref ref-type="bibr" rid="bib163">Peng et al., 2012</xref>) (using a minimum contig length of 1000); gene calling with Prodigal v2.6.3 (<xref ref-type="bibr" rid="bib82">Hyatt et al., 2010</xref>); tRNA identification with tRNAscan-SE v2.0.7 (<xref ref-type="bibr" rid="bib29">Chan and Lowe, 2019</xref>); as well as annotations of ribosomal RNAs (<xref ref-type="bibr" rid="bib186">Seemann, 2018</xref>), single-copy core genes (SCGs) for Bacteria, Archaea, and Protista, KEGG KOfams (<xref ref-type="bibr" rid="bib9">Aramaki et al., 2020</xref>), NCBI Clusters of Orthologous Group (COGs) (<xref ref-type="bibr" rid="bib59">Galperin et al., 2021</xref>), and Pfam protein families (release 33.1, <xref ref-type="bibr" rid="bib145">Mistry et al., 2021</xref>). The aforementioned annotations relied on HMMER v3.3.2 (<xref ref-type="bibr" rid="bib44">Eddy, 2011</xref>) as well as Diamond v0.9.14.115 (<xref ref-type="bibr" rid="bib24">Buchfink et al., 2015</xref>). As part of this workflow, all single assemblies were converted into anvi’o contigs databases. Samples from <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>, were processed differently because they contained merged reads rather than individual paired-end reads: no further quality filtering was run on these samples, we assembled them individually using MEGAHIT (<xref ref-type="bibr" rid="bib119">Li et al., 2015</xref>), and we used the anvi’o contigs workflow to perform all subsequent steps described for the metagenomics workflow above. Note that we used a version of KEGG downloaded in December 2020 (for reproducibility, the hash of the KEGG snapshot available via ‘anvi-setup-kegg-kofams’ is 45b7cc2e4fdc). Additionally, the annotation program ‘anvi-run-kegg-kofams’ includes a heuristic for annotating hits with bitscores that are just below the KEGG-defined threshold (<xref ref-type="bibr" rid="bib90">Kananen et al., 2025</xref>), which is further described at <ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/anvi-run-kegg-kofams/">https://anvio.org/m/anvi-run-kegg-kofams/</ext-link>.</p></sec><sec id="s3-4"><title>Genomic dataset</title><p>We downloaded all reference genomes for ‘species’ cluster representatives from the GTDB, release 95.0 (<xref ref-type="bibr" rid="bib160">Parks et al., 2018</xref>; <xref ref-type="bibr" rid="bib161">Parks et al., 2020</xref>), and processed them with the same anvi’o gene annotation workflow described above.</p></sec><sec id="s3-5"><title>Estimation of the number of microbial populations per metagenome</title><p>We used SCG sets belonging to each domain of microbial life (Bacteria, Archaea, Protista) to estimate the number of populations from each domain present in a given metagenomic sample. For each domain, we calculated the number of populations by taking the mode of the number of copies of each SCG in the set. We then summed the number of populations from each domain to get a total number of microbial populations within each sample. We accomplished this using SCG annotations provided by ‘anvi-run-hmms’ (which was run during metagenome processing) and a custom script relying on the anvi’o class ‘NumGenomesEstimator’ (see reproducible workflow).</p></sec><sec id="s3-6"><title>Removal of samples with low sequencing depth</title><p>We observed that, at lower sequencing depths, our estimates for the number of populations in a metagenomic sample were moderately correlated with sequencing depth (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>, R &gt; 0.5). These estimates rely on having accurate counts of SCGs, so we hypothesized that lower-depth samples were systematically missing SCGs, especially from populations with lower abundance. Since accurate population number estimates are critical for proper normalization of module copy numbers, keeping these lower-depth samples would have introduced a bias into our metabolism analyses. To address this, we removed samples with low sequencing depth from downstream analyses using a sequencing depth threshold of 25 million reads, such that the remaining samples exhibited a weaker correlation (R &lt; 0.5) between sequencing depth and number of estimated populations. We kept samples for which both the R1 file and the R2 file contained at least 25 million reads (and for the <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>, dataset, we kept samples containing at least 25 million merged reads). This produced our final sample set of 408 metagenomes.</p></sec><sec id="s3-7"><title>Estimation of normalized module copy numbers in metagenomes</title><p>We ran ‘anvi-estimate-metabolism’, in genome mode and with the ‘--add-copy-number’ flag, on each individual metagenome assembly to compute stepwise copy numbers for KEGG modules from the combined gene annotations of all populations present in the sample. We then divided these copy numbers by the number of estimated populations within each sample to obtain a PPCN for each module.</p></sec><sec id="s3-8"><title>Selection of IBD-enriched modules</title><p>We used a one-sided Mann-Whitney-Wilcoxon test with an FDR-adjusted p-value threshold of p ≤ 2e-10 on the per-sample PPCN values for each module individually to identify the modules that were most significantly enriched in the IBD sample group compared to the healthy group. We calculated the median PPCN of each metabolic module in the IBD samples, and again in the healthy samples. After filtering for p-values ≤ 2e-10, we also applied a minimum effect size threshold based on the median PPCN in each group (M<sub>IBD</sub> - M<sub>Healthy</sub> ≥ 0.12) – this threshold was calculated by taking the mean effect size over all modules that passed the p-value threshold. The set originally contained 34 modules that passed both thresholds, but we removed one redundant module (M00006) which represents the first half of another module in the set (M00004).</p></sec><sec id="s3-9"><title>Test for enrichment of biosynthesis modules</title><p>We used a one-sided Fisher’s exact test (also known as hypergeometric test, see e.g., <xref ref-type="bibr" rid="bib20">Boyle et al., 2004</xref>) for testing the independence between the metabolic modules identified to be IBD-enriched (i.e. using the methods described in ‘Selection of IBD-enriched modules’) and functionality (i.e. modules annotated to be involved in biosynthesis).</p></sec><sec id="s3-10"><title>Module comparisons</title><p>Because the 33 IBD-enriched modules were selected using PPCNs of healthy and IBD samples, statistical tests comparing PPCN distributions for these modules need to be interpreted with care, because the hypotheses were selected and tested on the same dataset (<xref ref-type="bibr" rid="bib56">Fithian et al., 2014</xref>). Therefore, to assess the statistical validity of the identified IBD-enriched modules, we performed the following repeated sample-split analysis: we first randomly split the IBD and healthy samples into the equal-sized training and validation sets. We select IBD-enriched modules in the training set using the Mann-Whitney-Wilcoxon test, and then compute the p-values on the validation set. We repeat this sample split analysis 1000 times with an FDR-adjusted p-value threshold of 1e-10 on the first split; most identified modules (89.4%; 95% CI: [87.5%, 91.3%]) on the training sets remain significant at a slightly less stringent threshold (1e-8) on the validation sets. This indicates that the approach we used to identify IBD-enriched modules yields stable and statistically significant results on this dataset.</p></sec><sec id="s3-11"><title>Metagenome classification</title><p>We trained logistic regression models to classify samples as ‘IBD’ or ‘healthy’ using PPCNs of IBD-enriched modules as features. We ran a 25-fold cross-validation pipeline on the set of 330 healthy and IBD metagenomes in our analysis, using an 80% train – 20% test random split of the data in each fold. The pipeline included selection of IBD-enriched modules within the training samples using the same strategy as described above, followed by training and testing of a logistic regression model as implemented in the ‘sklearn’ Python package. We set the ‘penalty’ parameter of the model to ‘None’ and the ‘max_iter’ parameter to 20,000 iterations, and we used the same random state in each fold to ensure changes in performance only come from differences in the training data rather than differences in model initialization. To summarize the overall performance of the classifier, we took the mean (over all folds) of each performance metric.</p><p>We trained a final classifier using the 33 IBD-enriched modules selected earlier from the entire set of 330 healthy and IBD metagenomes. We then applied this classifier to the metagenomic samples from <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>, which we processed in the same way as the other samples in our analysis (including removal of samples with low sequencing depth and calculation of PPCNs of KEGG modules for use as input features to the classifier model).</p></sec><sec id="s3-12"><title>Identification of gut microbial genomes from the GTDB</title><p>We took 19,226 representative genomes from the GTDB species clusters belonging to the phyla Firmicutes, Bacteroidetes, and Proteobacteria, which are most common in the human gut microbiome (<xref ref-type="bibr" rid="bib226">Woting and Blaut, 2016</xref>). To evaluate which of these genomes might represent gut microbes in a computationally tractable manner, we ran the anvi’o ‘EcoPhylo’ workflow (<ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/ecophylo">https://anvio.org/m/ecophylo</ext-link>) to contextualize these populations within 150 healthy gut metagenomes from the HMP (<xref ref-type="bibr" rid="bib80">Human Microbiome Project Consortium, 2012</xref>). Briefly, the EcoPhylo workflow (1) recovers sequences of a gene family of interest from each genome and metagenomic sample in the analysis, (2) clusters resulting sequences and picks representative sequences using mmseqs2 (<xref ref-type="bibr" rid="bib205">Steinegger and Söding, 2017</xref>), and (3) uses the representative sequences to rapidly summarize the distribution of each population cluster across the metagenomic samples through metagenomic read recruitment analyses. Here, we used the Ribosomal Protein S6 as our gene of interest, since it was the most frequently assembled SCG in our set of GTDB genomes. We clustered the Ribosomal Protein S6 sequences from GTDB genomes at 94% nucleotide identity.</p><p>To identify genomes that were likely to represent gut microbes, we selected genomes whose ribosomal protein S6 belonged to a gene cluster where at least 50% of the representative sequence was covered (i.e. detection ≥ 0.5×) in more than 10% of samples (i.e. n &gt; 15). There are 100 distinct individuals represented in the 150 HMP gut metagenomes – 56 of which were sampled just once and 46 of which were sampled at 2 or 3 time points – so this threshold is equivalent to detecting the genome in 5–15% of individuals. From this selection we obtained a set of 836 genomes; however, these were not exclusively gut microbes, as some non-gut populations have similar ribosomal protein S6 sequences to gut microbes and can therefore pass this selection step. To eliminate these, we mapped our set of 330 healthy and IBD metagenomes to the 836 genomes using the anvi’o metagenomics workflow and extracted genomes whose entire sequence was at least 50% covered (i.e. detection ≥ 0.5×) in over 2% (n &gt; 6) of these samples. Our final set of 338 genomes was used in downstream analysis.</p></sec><sec id="s3-13"><title>Genome phylogeny</title><p>To create the phylogeny, we identified the following ribosomal proteins that were annotated in at least 90% (n = 304) of the genomes: Ribosomal_S6, Ribosomal_S16, Ribosomal_L19, Ribosomal_L27, Ribosomal_S15, Ribosomal_S20p, Ribosomal_L13, Ribosomal_L21p, Ribosomal_L20, and Ribosomal_L9_C. We used ‘anvi-get-sequences-for-hmm-hits’ to extract the amino acid sequences for these genes, align the sequences using MUSCLE v3.8.1551 (<xref ref-type="bibr" rid="bib45">Edgar, 2004</xref>), and concatenate the alignments. We used trimAl v1.4.rev15 (<xref ref-type="bibr" rid="bib27">Capella-Gutiérrez et al., 2009</xref>) to remove any positions containing more than 50% of gap characters from the final alignment. Finally, we built the tree with IQtree v2.2.0.3 (<xref ref-type="bibr" rid="bib144">Minh et al., 2020</xref>), using the WAG model and running 1000 bootstraps.</p></sec><sec id="s3-14"><title>Determination of HMI status for genomes</title><p>We estimated metabolic potential for each genome with ‘anvi-estimate-metabolism’ (in genome mode) to get stepwise completeness scores for each KEGG module, and then we used the script ‘anvi-script-estimate-metabolic-independence’ to give each genome a metabolic independence score based on completeness of the 33 IBD-enriched modules. Briefly, the latter script calculates the score by summing the completeness scores of each module of interest. Genomes were classified as having HMI if their score was greater than or equal to 26.4. We calculated this threshold by requiring these 33 modules to be, on average, at least 80% complete in a given genome.</p></sec><sec id="s3-15"><title>Genome distribution across sample groups</title><p>We mapped the gut metagenomes from the healthy, non-IBD, and IBD groups to each genome using the anvi’o metagenomics workflow in reference mode. We used ‘anvi-summarize’ to obtain a matrix of genome detection across all samples. We summarized this data as follows: for each genome, we computed the proportion of samples in each group in which at least 50% of the genome sequence was covered by at least 1 read (≥ 50% detection). For each sample, we calculated the proportion of detected genomes that were classified as HMI. We also computed the percent abundance of each genome in each sample by dividing the number of reads mapping to that genome by the total number of reads in the sample.</p></sec><sec id="s3-16"><title>Visualizations</title><p>We used ggplot2 (<xref ref-type="bibr" rid="bib223">Wickham, 2016</xref>) to generate most of the initial data visualizations. The phylogeny and heatmap in <xref ref-type="fig" rid="fig3">Figure 3</xref> were generated by the anvi’o interactive interface and the ROC curves in <xref ref-type="fig" rid="fig4">Figure 4</xref> were generated using the pyplot package of matplotlib (<xref ref-type="bibr" rid="bib81">Hunter, 2007</xref>). These visualizations were refined for publication using Inkscape, an open-source graphical editing software that is available at <ext-link ext-link-type="uri" xlink:href="https://inkscape.org/">https://inkscape.org/</ext-link>.</p></sec><sec id="s3-17"><title>Supplementary table files</title><p>Supplementary Table files and our Appendix files can be accessed at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.22679080">https://doi.org/10.6084/m9.figshare.22679080</ext-link>.</p></sec></sec></body><back><sec sec-type="additional-information" id="s4"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Validation, Visualization, Methodology, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Formal analysis, Visualization, Writing – review and editing, Supported statistical analyses</p></fn><fn fn-type="con" id="con3"><p>Software, Writing – review and editing</p></fn><fn fn-type="con" id="con4"><p>Resources, Writing – review and editing</p></fn><fn fn-type="con" id="con5"><p>Resources, Data curation, Writing – review and editing</p></fn><fn fn-type="con" id="con6"><p>Resources, Data curation, Writing – review and editing</p></fn><fn fn-type="con" id="con7"><p>Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con8"><p>Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con9"><p>Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con10"><p>Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con11"><p>Resources, Writing – review and editing</p></fn><fn fn-type="con" id="con12"><p>Conceptualization, Formal analysis, Supervision, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con13"><p>Conceptualization, Software, Formal analysis, Visualization, Writing – original draft, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s5"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Samples and cohorts used in this study.</title><p>(a) Description of studies/cohorts providing publicly available gut metagenomes from healthy people, non-inflammatory bowel disease (IBD) controls, and people with IBD. For each study, we note the sample groups it contributes metagenomes to; whether or not those samples were sufficiently deeply sequenced to be included in the main analyses; the country of origin of the samples; the sample type (fecal metagenome or ileal pouch luminal aspirate); the number of samples it contributes to each group before and after applying the sequencing depth threshold; and cohort details/exclusions as described within the study. (b) Description of 408 samples included in the primary analyses of this manuscript (i.e. those with sufficient sequencing depth of ≥ 25 million reads), including their associated diagnosis (ulcerative colitis (UC), Crohn’s disease (CD), non-IBD, healthy, colorectal cancer with adenoma (CRC_ADENOMA), or colorectal cancer with carcinoma (CRC_CARCINOMA)); study of origin; sample group; sequencing depth; and number of microbial populations estimated to be represented within the metagenome. (c) Description of all samples initially considered and their SRA accession numbers. (d) The number of gene calls and the number/proportion of annotations per gene call for KOfams, Clusters of Orthologous Groups (COGs), and Pfams in each sample. (e) The number of genes with at least one functional annotation and the number of tRNAs in each sample from the subset of deeply sequenced samples. (f) Description of the 57 antibiotic time-series gut metagenomes from <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref> used for classifier testing, including SRA accession number; sampling day in the time series; sequencing depth; and estimated numbers of microbial populations represented in the sample.</p></caption><media xlink:href="elife-89862-supp1-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Metabolism data in metagenomes.</title><p>(a) Description of the 33 KEGG modules enriched in inflammatory bowel disease (IBD) samples, including: module name, KEGG categorization, and definition; their median per-population copy numbers (PPCNs) in the healthy sample group and IBD sample group; the p-value, FDR-adjusted p-value, and W statistic from the per-module Wilcoxon rank-sum test used to determine enrichment in IBD; the difference between its median PPCN in IBD samples and median PPCN in healthy samples (‘effect size’); the fraction of samples in which the module occurs with nonzero copy number; whether the module is also enriched in the high metabolic independence (HMI) populations analyzed in <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>; the number of total enzymes in the module; the number of total compounds in the modules; and the numbers and proportions of shared enzymes or compounds between this module and the other IBD-enriched modules. (b) Description of all 179 KEGG modules with nonzero copy number in at least one metagenome. Most of the columns match the corresponding column in sheet (b) with the exception of the ‘enrichment status’ column, which indicates whether the module was found to be enriched in the IBD samples in this study (‘IBD_ENRICHED’), in the high-metabolic independence genomes in <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref> (‘HMI_ENRICHED’), in both (‘HMI_AND_IBD’), or in neither (‘OTHER’). (c) Matrix of stepwise copy number of each module in each deeply sequenced gut metagenome. (d) Per-population copy number of each module in each deeply sequenced gut metagenome in the IBD, non-IBD, and healthy sample groups. (e) Per-population copy number of each module in each antibiotic time-series sample from <xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>.</p></caption><media xlink:href="elife-89862-supp2-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Genome Taxonomy Database (GTDB) genome data.</title><p>(a) List of 338 GTDB representative genomes identified as gut microbes, their taxonomy, metabolic independence score, classification as high metabolic independence (‘HMI’) or not (‘non-HMI’), genome length in base pairs, and number of gene calls. (b) Matrix of stepwise completeness of each module in each genome. (c) Matrix of genome detection in each deeply sequenced gut metagenome in the inflammatory bowel disease (IBD), non-IBD, and healthy sample groups. (d) Percent abundance of each genome in each deeply sequenced gut metagenome. (e) Per-genome proportion of samples from each sample group that the genome is detected in using a threshold of 50% (i.e. at least half of the genome sequence is covered by at least one sequencing read in a given sample). (f) Per-sample proportion of detected genomes that are classified as HMI. (g) Average completion of each IBD-enriched module within the HMI genome group and the non-HMI genome group, as well as the difference between these values. (h) Genome-level results when using different HMI score thresholds for determining HMI status. Each threshold is shown both as the average percent completeness required for the 33 IBD-enriched modules and as the HMI score above which a genome is considered HMI. Results for each threshold include the number of genomes assigned as HMI, the percent of genomes assigned as HMI (out of 338), the number and percent of <italic>Bacteroides</italic> genomes assigned as HMI, the mean genome size and mean number of gene calls for both HMI and non-HMI genomes, the p-values and W statistics of the Wilcoxon rank-sum tests comparing detection of HMI vs non-HMI genomes (1) in IBD samples and (2) in healthy samples, and the p-value of the Kruskal-Wallis rank-sum test comparing the fraction of genomes classified as HMI in IBD samples vs healthy/non-IBD samples.</p></caption><media xlink:href="elife-89862-supp3-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Metagenome classifier information.</title><p>(a) Details and performance of previously published classifiers for inflammatory bowel disease (IBD) and IBD subtypes. For each classifier, we summarize the cohort details as described by the study; the size of training datasets and validation datasets (if any); the type(s) of samples, data, and extracted features used for classification; the target classes (i.e. what the samples were being classified as); the classifier type and training/validation strategy; and the performance metrics as reported by the study. (b) Classification of each (<xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>) metagenome by our logistic regression model trained for distinguishing IBD vs healthy samples on the basis of PPCN data for IBD-enriched modules. This table describes whether the sample was classified as healthy (‘HEALTHY’) or stressed (‘IBD’, which we consider to be equivalent to an identification of gut stress), and also whether the sample had low sequencing depth (&lt;25 million reads) or not. (c) Summary of the performance of our metagenome classifier across different training/validation strategies using the IBD and healthy metagenome samples. It also includes the details of our final classifier trained on all 330 samples, though performance data is not available for this model since there were no IBD/healthy samples left for validation – however, see manuscript for its performance on the (<xref ref-type="bibr" rid="bib157">Palleja et al., 2018</xref>) antibiotic time-series dataset. The subsequent sheets include per-fold data and performance information for each train-test strategy: (d) random split cross-validation (25-fold) on PPCN data; (e) leave-two-studies-out cross-validation (24-fold); and (f) (10-fold) cross-validation leaving out samples from the two dominating studies in our dataset (<xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>; <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>).</p></caption><media xlink:href="elife-89862-supp4-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Details of available software for metabolism estimation.</title><p>For each tool (including the one published in this study), we summarize: the software category (based upon the tool’s architecture and mode of use); its metabolism reconstruction strategy (whether it is a pathway prediction tool or a modeling tool or both); the data source(s) it uses for enzyme and metabolic pathway information; how it calculates pathway completeness or generates models (depending on reconstruction strategy); what input and output types it accepts/generates; any additional capabilities as advertised by the tool’s publication; whether or not the tool is open-source; the program type; and what language(s) it is developed in (if known). The reference publication and code repository or webpage for each tool is also included.</p></caption><media xlink:href="elife-89862-supp5-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>Data from validation of the per-population copy number (PPCN) approach with simulated metagenomic data.</title><p>(a) Distribution (mean and standard deviation) of PPCN and PPCN error (computed relative to either average genomic completeness or average genomic copy number) across each validation test case, as well as the proportion of correct, off-by-one, and off-by-two community size estimates in each test case. (b) Spearman’s correlation test results between sample parameters (genome size, community size, and diversity level) and important values computed in our approach (PPCN, PPCN accuracy metrics, and accuracy of community size estimates). Negative values are shown in red and nonsignificant p-values are highlighted in blue. (c) Normalized and rank-ordered relative abundance data from the top 20 most abundant microbial populations in healthy gut metagenomes that we used to recreate a ‘typical’ relative abundance curve for our simulated metagenomes, based upon data from <xref ref-type="bibr" rid="bib15">Beghini et al., 2021</xref>. Each initial column provides the data from a single sample, and the final four columns describe: the average relative abundance at each rank order; the averages when scaled such that the minimum abundance is 1; the corresponding (integer) coverage values for each scaled average relative abundance value; and the coverage values when increased by 20× for sufficient sequencing depth for assembly.</p></caption><media xlink:href="elife-89862-supp6-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-89862-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s6"><title>Data availability</title><p>Accession numbers for publicly available data are listed in our Supplementary Tables that are also accessible at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.22679080">https://doi.org/10.6084/m9.figshare.22679080</ext-link>. Anvi'o contigs databases of our assemblies for the 408 deeply-sequenced metagenomes, as well as assemblies of the Palleja et al. 2018 metagenomes are available at can be accessed at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.7897987">https://doi.org/10.5281/zenodo.7897987</ext-link>. Finally the URL <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.7883421">https://doi.org/10.5281/zenodo.7883421</ext-link> gives access to anvi'o contigs databases for the 338 Genome Taxonomy Database (<ext-link ext-link-type="uri" xlink:href="https://gtdb.ecogenomic.org/">GTDB</ext-link>) reference genomes that represent populations that are prevalent in human gut metagenomes.</p><p>The following datasets were generated:</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Veseli</surname><given-names>I</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2024">2024</year><data-title>Supplementary Tables for Veseli et al. 2023</data-title><source>Figshare</source><pub-id pub-id-type="doi">10.6084/m9.figshare.22679080</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset2"><person-group person-group-type="author"><name><surname>Veseli</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Palleja et al. 2018 Metagenome Assemblies for Veseli et al. 2023</data-title><source>Zenodo</source><pub-id pub-id-type="doi">10.5281/zenodo.7897987</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset3"><person-group person-group-type="author"><name><surname>Veseli</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>GTDB Genome Contigs DBs for Veseli et al. 2023</data-title><source>Zenodo</source><pub-id pub-id-type="doi">10.5281/zenodo.7883421</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset4"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Cai</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>F</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><data-title>A metagenome-wide association study of gut microbiota in type 2 diabetes</data-title><source>NCBI Sequence Read Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra?term=SRA050230">SRA050230</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset5"><person-group person-group-type="author"><name><surname>Le Chatelier</surname><given-names>E</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Prifti</surname><given-names>E</given-names></name><name><surname>Hildebrand</surname><given-names>F</given-names></name><name><surname>Falony</surname><given-names>G</given-names></name><name><surname>Almeida</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><data-title>Richness of human gut microbiome correlates with metabolic markers</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB4336">PRJEB4336</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset6"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name><name><surname>Jia</surname><given-names>H</given-names></name><name><surname>Stadlmayr</surname><given-names>A</given-names></name><name><surname>Tang</surname><given-names>L</given-names></name><name><surname>Lan</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2015">2015</year><data-title>Gut microbiome development along the colorectal adenoma-carcinoma sequence</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB7774">PRJEB7774</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset7"><person-group person-group-type="author"><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Sirota-Madi</surname><given-names>A</given-names></name><name><surname>Avila-Pacheco</surname><given-names>J</given-names></name><name><surname>Fornelos</surname><given-names>N</given-names></name><name><surname>Haiser</surname><given-names>HJ</given-names></name><name><surname>Reinker</surname><given-names>S</given-names></name><name><surname>Vatanen</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Gut microbiome structure and metabolic activity in inflammatory bowel disease</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA400072">PRJNA400072</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset8"><person-group person-group-type="author"><name><surname>Lloyd-Price</surname><given-names>J</given-names></name><name><surname>Arze</surname><given-names>C</given-names></name><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name><name><surname>Schirmer</surname><given-names>M</given-names></name><name><surname>Pacheco</surname><given-names>JA</given-names></name><name><surname>Poon</surname><given-names>TW</given-names></name><name><surname>Andrews</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Longitudinal Multi’omics of the Human Microbiome in Inflammatory Bowel Disease</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA398089">PRJNA398089</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset9"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Cai</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>F</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><data-title>A metagenome-wide association study of gut microbiota in type 2 diabetes</data-title><source>NCBI Sequence Read Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra?term=SRA045646">SRA045646</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset10"><person-group person-group-type="author"><name><surname>Quince</surname><given-names>C</given-names></name><name><surname>Ijaz</surname><given-names>UZ</given-names></name><name><surname>Loman</surname><given-names>N</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Saulnier</surname><given-names>D</given-names></name><name><surname>Russell</surname><given-names>J</given-names></name><name><surname>Haig</surname><given-names>SJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><data-title>Exclusive enteral nutrition modulates the faecal metagenome in paediatric Crohn’s disease not by enriching the abundance of presumably ‘beneficial’ commensals but by suppressing ‘dysbiotic’ bacteria</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJEB7576">PRJEB7576</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset11"><person-group person-group-type="author"><name><surname>Raymond</surname><given-names>F</given-names></name><name><surname>Ouameur</surname><given-names>AA</given-names></name><name><surname>Déraspe</surname><given-names>M</given-names></name><name><surname>Iqbal</surname><given-names>N</given-names></name><name><surname>Gingras</surname><given-names>H</given-names></name><name><surname>Dridi</surname><given-names>B</given-names></name><name><surname>Leprohon</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2016">2016</year><data-title>The initial state of the human gut microbiome determines its reshaping by antibiotics</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB8094">PRJEB8094</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset12"><person-group person-group-type="author"><name><surname>Schirmer</surname><given-names>M</given-names></name><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Lloyd-Price</surname><given-names>J</given-names></name><name><surname>McIver</surname><given-names>LJ</given-names></name><name><surname>Schwager</surname><given-names>R</given-names></name><name><surname>Poon</surname><given-names>TW</given-names></name><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>Dynamics of metatranscription in the inflammatory bowel disease gut microbiome</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA389280/">PRJNA389280</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset13"><person-group person-group-type="author"><collab>University of Sydney</collab></person-group><year iso-8601-date="2016">2016</year><data-title>metagenome fecal microbiota, Ilumina seq reads of 12 individuals at 2 timepoints</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJEB6092/">PRJEB6092</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset14"><person-group person-group-type="author"><name><surname>Vineis</surname><given-names>JH</given-names></name><name><surname>Ringus</surname><given-names>DL</given-names></name><name><surname>Morrison</surname><given-names>HG</given-names></name><name><surname>Delmont</surname><given-names>TO</given-names></name><name><surname>Dalal</surname><given-names>S</given-names></name><name><surname>Raffals</surname><given-names>LH</given-names></name><name><surname>Antonopoulos</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2016">2016</year><data-title>Patient-Specific Bacteroides Genome Variants in Pouchitis</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA46881">PRJNA46881</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset15"><person-group person-group-type="author"><name><surname>Wen</surname><given-names>C</given-names></name><name><surname>Zheng</surname><given-names>Z</given-names></name><name><surname>Shao</surname><given-names>T</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Xie</surname><given-names>Z</given-names></name><name><surname>Chatelier</surname><given-names>EL</given-names></name><name><surname>He</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2017">2017</year><data-title>Quantitative metagenomics reveals unique gut microbiome biomarkers in ankylosing spondylitis</data-title><source>NCBI Sequence Read Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra/?term=SRP100575">SRP100575</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset16"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>N</given-names></name><name><surname>Yang</surname><given-names>F</given-names></name><name><surname>Li</surname><given-names>A</given-names></name><name><surname>Prifti</surname><given-names>E</given-names></name><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Chatelier</surname><given-names>EL</given-names></name><name><surname>Yao</surname><given-names>J</given-names></name><name><surname>Wu</surname><given-names>L</given-names></name><name><surname>Zhou</surname><given-names>J</given-names></name><name><surname>Ni</surname><given-names>S</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Pons</surname><given-names>N</given-names></name><name><surname>Batto</surname><given-names>JM</given-names></name><name><surname>Kennedy</surname><given-names>SP</given-names></name><name><surname>Leonard</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2014">2014</year><data-title>Alterations of the human gut microbiome in liver cirrhosis</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB6337">PRJEB6337</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset17"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>H</given-names></name><name><surname>Guo</surname><given-names>R</given-names></name><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Lan</surname><given-names>Z</given-names></name><name><surname>Qin</surname><given-names>B</given-names></name><name><surname>Ward</surname><given-names>KJ</given-names></name><name><surname>Zhong</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><data-title>Shotgun Metagenomics of 250 Adult Twins Reveals Genetic and Environmental Impacts on the Gut Microbiome</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB9584">PRJEB9584</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset18"><person-group person-group-type="author"><name><surname>Palleja</surname><given-names>A</given-names></name><name><surname>Mikkelsen</surname><given-names>KH</given-names></name><name><surname>Forslund</surname><given-names>SK</given-names></name><name><surname>Kashani</surname><given-names>A</given-names></name><name><surname>Allin</surname><given-names>KH</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>Hansen</surname><given-names>TH</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>Gut resistome modulates resilience and recovery of gut microbiota after broad-spectrum antibiotic treatment</data-title><source>EBI European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJEB20800">PRJEB20800</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset19"><person-group person-group-type="author"><name><surname>Rampelli</surname><given-names>S</given-names></name><name><surname>Schnorr</surname><given-names>SL</given-names></name><name><surname>Consolandi</surname><given-names>C</given-names></name><name><surname>Turroni</surname><given-names>S</given-names></name><name><surname>Severgnini</surname><given-names>M</given-names></name><name><surname>peano</surname><given-names>C</given-names></name><name><surname>Brigidi</surname><given-names>P</given-names></name><name><surname>Crittenden</surname><given-names>AN</given-names></name><name><surname>Henry</surname><given-names>AG</given-names></name><name><surname>Candela</surname><given-names>M</given-names></name></person-group><source>NCBI BioProject</source><year iso-8601-date="2015">2015</year><data-title>Metagenome Sequencing of the Hadza Hunter-Gatherer Gut Microbiota</data-title><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA278393">PRJNA278393</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset20"><person-group person-group-type="author"><name><surname>Parks</surname><given-names>DH</given-names></name><name><surname>Chuvochina</surname><given-names>M</given-names></name><name><surname>Rinke</surname><given-names>C</given-names></name><name><surname>Mussig</surname><given-names>AJ</given-names></name><name><surname>Chaumeil</surname><given-names>PA</given-names></name><name><surname>Hugenholtz</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>GTDB: an ongoing census of bacterial and archaeal diversity through a phylogenetically consistent, rank normalized and complete genome-based taxonomy</data-title><source>GTDB</source><pub-id pub-id-type="accession" xlink:href="https://gtdb.ecogenomic.org/">release95.0</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Christopher Quince for advice on statistical significance testing. IV acknowledges support from the National Science Foundation Graduate Research Fellowship under Grant No. 1746045; ADW acknowledges support from the National Institutes of General Medical Sciences under R35 GM133420. YTC acknowledges support from the Stanford Data Science Postdoctoral Fellowship. RB acknowledges support from the National Institutes of Health under R35 GM128716. ECF acknowledges support from the University of Chicago International Student Fellowship. Additional support for ECF and AME came from an NIH NIDDK grant (RC2 DK122394) to AME.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Agren</surname><given-names>R</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Shoaie</surname><given-names>S</given-names></name><name><surname>Vongsangnak</surname><given-names>W</given-names></name><name><surname>Nookaew</surname><given-names>I</given-names></name><name><surname>Nielsen</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The RAVEN toolbox and its use for generating a genome-scale metabolic model for Penicillium chrysogenum</article-title><source>PLOS Computational Biology</source><volume>9</volume><elocation-id>e1002980</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1002980</pub-id><pub-id pub-id-type="pmid">23555215</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Agus</surname><given-names>A</given-names></name><name><surname>Planchais</surname><given-names>J</given-names></name><name><surname>Sokol</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Gut microbiota regulation of tryptophan metabolism in health and disease</article-title><source>Cell Host &amp; Microbe</source><volume>23</volume><fpage>716</fpage><lpage>724</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2018.05.003</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aite</surname><given-names>M</given-names></name><name><surname>Chevallier</surname><given-names>M</given-names></name><name><surname>Frioux</surname><given-names>C</given-names></name><name><surname>Trottier</surname><given-names>C</given-names></name><name><surname>Got</surname><given-names>J</given-names></name><name><surname>Cortés</surname><given-names>MP</given-names></name><name><surname>Mendoza</surname><given-names>SN</given-names></name><name><surname>Carrier</surname><given-names>G</given-names></name><name><surname>Dameron</surname><given-names>O</given-names></name><name><surname>Guillaudeux</surname><given-names>N</given-names></name><name><surname>Latorre</surname><given-names>M</given-names></name><name><surname>Loira</surname><given-names>N</given-names></name><name><surname>Markov</surname><given-names>GV</given-names></name><name><surname>Maass</surname><given-names>A</given-names></name><name><surname>Siegel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Traceability, reproducibility and wiki-exploration for “à-la-carte” reconstructions of genome-scale metabolic models</article-title><source>PLOS Computational Biology</source><volume>14</volume><elocation-id>e1006146</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006146</pub-id><pub-id pub-id-type="pmid">29791443</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akram</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Citric acid cycle and role of its intermediates in metabolism</article-title><source>Cell Biochemistry and Biophysics</source><volume>68</volume><fpage>475</fpage><lpage>478</lpage><pub-id pub-id-type="doi">10.1007/s12013-013-9750-1</pub-id><pub-id pub-id-type="pmid">24068518</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alekshun</surname><given-names>MN</given-names></name><name><surname>Levy</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Molecular mechanisms of antibacterial multidrug resistance</article-title><source>Cell</source><volume>128</volume><fpage>1037</fpage><lpage>1050</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2007.03.004</pub-id><pub-id pub-id-type="pmid">17382878</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alkhalaf</surname><given-names>LM</given-names></name><name><surname>Ryan</surname><given-names>KS</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Biosynthetic manipulation of tryptophan in bacteria: pathways and mechanisms</article-title><source>Chemistry &amp; Biology</source><volume>22</volume><fpage>317</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1016/j.chembiol.2015.02.005</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>An</surname><given-names>D</given-names></name><name><surname>Na</surname><given-names>C</given-names></name><name><surname>Bielawski</surname><given-names>J</given-names></name><name><surname>Hannun</surname><given-names>YA</given-names></name><name><surname>Kasper</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Membrane sphingolipids as essential molecular signals for Bacteroides survival in the intestine</article-title><source>PNAS</source><volume>108 Suppl 1</volume><fpage>4666</fpage><lpage>4671</lpage><pub-id pub-id-type="doi">10.1073/pnas.1001501107</pub-id><pub-id pub-id-type="pmid">20855611</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name><name><surname>Luo</surname><given-names>C</given-names></name><name><surname>Yajnik</surname><given-names>V</given-names></name><name><surname>Khalili</surname><given-names>H</given-names></name><name><surname>Garber</surname><given-names>JJ</given-names></name><name><surname>Stevens</surname><given-names>BW</given-names></name><name><surname>Cleland</surname><given-names>T</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Gut microbiome function predicts response to anti-integrin biologic therapy in inflammatory bowel diseases</article-title><source>Cell Host &amp; Microbe</source><volume>21</volume><fpage>603</fpage><lpage>610</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2017.04.010</pub-id><pub-id pub-id-type="pmid">28494241</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aramaki</surname><given-names>T</given-names></name><name><surname>Blanc-Mathieu</surname><given-names>R</given-names></name><name><surname>Endo</surname><given-names>H</given-names></name><name><surname>Ohkubo</surname><given-names>K</given-names></name><name><surname>Kanehisa</surname><given-names>M</given-names></name><name><surname>Goto</surname><given-names>S</given-names></name><name><surname>Ogata</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>KofamKOALA: KEGG Ortholog assignment based on profile HMM and adaptive score threshold</article-title><source>Bioinformatics</source><volume>36</volume><fpage>2251</fpage><lpage>2252</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btz859</pub-id><pub-id pub-id-type="pmid">31742321</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arkin</surname><given-names>AP</given-names></name><name><surname>Cottingham</surname><given-names>RW</given-names></name><name><surname>Henry</surname><given-names>CS</given-names></name><name><surname>Harris</surname><given-names>NL</given-names></name><name><surname>Stevens</surname><given-names>RL</given-names></name><name><surname>Maslov</surname><given-names>S</given-names></name><name><surname>Dehal</surname><given-names>P</given-names></name><name><surname>Ware</surname><given-names>D</given-names></name><name><surname>Perez</surname><given-names>F</given-names></name><name><surname>Canon</surname><given-names>S</given-names></name><name><surname>Sneddon</surname><given-names>MW</given-names></name><name><surname>Henderson</surname><given-names>ML</given-names></name><name><surname>Riehl</surname><given-names>WJ</given-names></name><name><surname>Murphy-Olson</surname><given-names>D</given-names></name><name><surname>Chan</surname><given-names>SY</given-names></name><name><surname>Kamimura</surname><given-names>RT</given-names></name><name><surname>Kumari</surname><given-names>S</given-names></name><name><surname>Drake</surname><given-names>MM</given-names></name><name><surname>Brettin</surname><given-names>TS</given-names></name><name><surname>Glass</surname><given-names>EM</given-names></name><name><surname>Chivian</surname><given-names>D</given-names></name><name><surname>Gunter</surname><given-names>D</given-names></name><name><surname>Weston</surname><given-names>DJ</given-names></name><name><surname>Allen</surname><given-names>BH</given-names></name><name><surname>Baumohl</surname><given-names>J</given-names></name><name><surname>Best</surname><given-names>AA</given-names></name><name><surname>Bowen</surname><given-names>B</given-names></name><name><surname>Brenner</surname><given-names>SE</given-names></name><name><surname>Bun</surname><given-names>CC</given-names></name><name><surname>Chandonia</surname><given-names>J-M</given-names></name><name><surname>Chia</surname><given-names>J-M</given-names></name><name><surname>Colasanti</surname><given-names>R</given-names></name><name><surname>Conrad</surname><given-names>N</given-names></name><name><surname>Davis</surname><given-names>JJ</given-names></name><name><surname>Davison</surname><given-names>BH</given-names></name><name><surname>DeJongh</surname><given-names>M</given-names></name><name><surname>Devoid</surname><given-names>S</given-names></name><name><surname>Dietrich</surname><given-names>E</given-names></name><name><surname>Dubchak</surname><given-names>I</given-names></name><name><surname>Edirisinghe</surname><given-names>JN</given-names></name><name><surname>Fang</surname><given-names>G</given-names></name><name><surname>Faria</surname><given-names>JP</given-names></name><name><surname>Frybarger</surname><given-names>PM</given-names></name><name><surname>Gerlach</surname><given-names>W</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name><name><surname>Greiner</surname><given-names>A</given-names></name><name><surname>Gurtowski</surname><given-names>J</given-names></name><name><surname>Haun</surname><given-names>HL</given-names></name><name><surname>He</surname><given-names>F</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Joachimiak</surname><given-names>MP</given-names></name><name><surname>Keegan</surname><given-names>KP</given-names></name><name><surname>Kondo</surname><given-names>S</given-names></name><name><surname>Kumar</surname><given-names>V</given-names></name><name><surname>Land</surname><given-names>ML</given-names></name><name><surname>Meyer</surname><given-names>F</given-names></name><name><surname>Mills</surname><given-names>M</given-names></name><name><surname>Novichkov</surname><given-names>PS</given-names></name><name><surname>Oh</surname><given-names>T</given-names></name><name><surname>Olsen</surname><given-names>GJ</given-names></name><name><surname>Olson</surname><given-names>R</given-names></name><name><surname>Parrello</surname><given-names>B</given-names></name><name><surname>Pasternak</surname><given-names>S</given-names></name><name><surname>Pearson</surname><given-names>E</given-names></name><name><surname>Poon</surname><given-names>SS</given-names></name><name><surname>Price</surname><given-names>GA</given-names></name><name><surname>Ramakrishnan</surname><given-names>S</given-names></name><name><surname>Ranjan</surname><given-names>P</given-names></name><name><surname>Ronald</surname><given-names>PC</given-names></name><name><surname>Schatz</surname><given-names>MC</given-names></name><name><surname>Seaver</surname><given-names>SMD</given-names></name><name><surname>Shukla</surname><given-names>M</given-names></name><name><surname>Sutormin</surname><given-names>RA</given-names></name><name><surname>Syed</surname><given-names>MH</given-names></name><name><surname>Thomason</surname><given-names>J</given-names></name><name><surname>Tintle</surname><given-names>NL</given-names></name><name><surname>Wang</surname><given-names>D</given-names></name><name><surname>Xia</surname><given-names>F</given-names></name><name><surname>Yoo</surname><given-names>H</given-names></name><name><surname>Yoo</surname><given-names>S</given-names></name><name><surname>Yu</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>KBase: the united states department of energy systems biology knowledgebase</article-title><source>Nature Biotechnology</source><volume>36</volume><fpage>566</fpage><lpage>569</lpage><pub-id pub-id-type="doi">10.1038/nbt.4163</pub-id><pub-id pub-id-type="pmid">29979655</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arumugam</surname><given-names>M</given-names></name><name><surname>Raes</surname><given-names>J</given-names></name><name><surname>Pelletier</surname><given-names>E</given-names></name><name><surname>Le Paslier</surname><given-names>D</given-names></name><name><surname>Yamada</surname><given-names>T</given-names></name><name><surname>Mende</surname><given-names>DR</given-names></name><name><surname>Fernandes</surname><given-names>GR</given-names></name><name><surname>Tap</surname><given-names>J</given-names></name><name><surname>Bruls</surname><given-names>T</given-names></name><name><surname>Batto</surname><given-names>J-M</given-names></name><name><surname>Bertalan</surname><given-names>M</given-names></name><name><surname>Borruel</surname><given-names>N</given-names></name><name><surname>Casellas</surname><given-names>F</given-names></name><name><surname>Fernandez</surname><given-names>L</given-names></name><name><surname>Gautier</surname><given-names>L</given-names></name><name><surname>Hansen</surname><given-names>T</given-names></name><name><surname>Hattori</surname><given-names>M</given-names></name><name><surname>Hayashi</surname><given-names>T</given-names></name><name><surname>Kleerebezem</surname><given-names>M</given-names></name><name><surname>Kurokawa</surname><given-names>K</given-names></name><name><surname>Leclerc</surname><given-names>M</given-names></name><name><surname>Levenez</surname><given-names>F</given-names></name><name><surname>Manichanh</surname><given-names>C</given-names></name><name><surname>Nielsen</surname><given-names>HB</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>Pons</surname><given-names>N</given-names></name><name><surname>Poulain</surname><given-names>J</given-names></name><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Sicheritz-Ponten</surname><given-names>T</given-names></name><name><surname>Tims</surname><given-names>S</given-names></name><name><surname>Torrents</surname><given-names>D</given-names></name><name><surname>Ugarte</surname><given-names>E</given-names></name><name><surname>Zoetendal</surname><given-names>EG</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Guarner</surname><given-names>F</given-names></name><name><surname>Pedersen</surname><given-names>O</given-names></name><name><surname>de Vos</surname><given-names>WM</given-names></name><name><surname>Brunak</surname><given-names>S</given-names></name><name><surname>Doré</surname><given-names>J</given-names></name><collab>MetaHIT Consortium</collab><name><surname>Antolín</surname><given-names>M</given-names></name><name><surname>Artiguenave</surname><given-names>F</given-names></name><name><surname>Blottiere</surname><given-names>HM</given-names></name><name><surname>Almeida</surname><given-names>M</given-names></name><name><surname>Brechot</surname><given-names>C</given-names></name><name><surname>Cara</surname><given-names>C</given-names></name><name><surname>Chervaux</surname><given-names>C</given-names></name><name><surname>Cultrone</surname><given-names>A</given-names></name><name><surname>Delorme</surname><given-names>C</given-names></name><name><surname>Denariaz</surname><given-names>G</given-names></name><name><surname>Dervyn</surname><given-names>R</given-names></name><name><surname>Foerstner</surname><given-names>KU</given-names></name><name><surname>Friss</surname><given-names>C</given-names></name><name><surname>van de Guchte</surname><given-names>M</given-names></name><name><surname>Guedon</surname><given-names>E</given-names></name><name><surname>Haimet</surname><given-names>F</given-names></name><name><surname>Huber</surname><given-names>W</given-names></name><name><surname>van Hylckama-Vlieg</surname><given-names>J</given-names></name><name><surname>Jamet</surname><given-names>A</given-names></name><name><surname>Juste</surname><given-names>C</given-names></name><name><surname>Kaci</surname><given-names>G</given-names></name><name><surname>Knol</surname><given-names>J</given-names></name><name><surname>Lakhdari</surname><given-names>O</given-names></name><name><surname>Layec</surname><given-names>S</given-names></name><name><surname>Le Roux</surname><given-names>K</given-names></name><name><surname>Maguin</surname><given-names>E</given-names></name><name><surname>Mérieux</surname><given-names>A</given-names></name><name><surname>Melo Minardi</surname><given-names>R</given-names></name><name><surname>M’rini</surname><given-names>C</given-names></name><name><surname>Muller</surname><given-names>J</given-names></name><name><surname>Oozeer</surname><given-names>R</given-names></name><name><surname>Parkhill</surname><given-names>J</given-names></name><name><surname>Renault</surname><given-names>P</given-names></name><name><surname>Rescigno</surname><given-names>M</given-names></name><name><surname>Sanchez</surname><given-names>N</given-names></name><name><surname>Sunagawa</surname><given-names>S</given-names></name><name><surname>Torrejon</surname><given-names>A</given-names></name><name><surname>Turner</surname><given-names>K</given-names></name><name><surname>Vandemeulebrouck</surname><given-names>G</given-names></name><name><surname>Varela</surname><given-names>E</given-names></name><name><surname>Winogradsky</surname><given-names>Y</given-names></name><name><surname>Zeller</surname><given-names>G</given-names></name><name><surname>Weissenbach</surname><given-names>J</given-names></name><name><surname>Ehrlich</surname><given-names>SD</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Enterotypes of the human gut microbiome</article-title><source>Nature</source><volume>473</volume><fpage>174</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/nature09944</pub-id><pub-id pub-id-type="pmid">21508958</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aziz</surname><given-names>RK</given-names></name><name><surname>Bartels</surname><given-names>D</given-names></name><name><surname>Best</surname><given-names>AA</given-names></name><name><surname>DeJongh</surname><given-names>M</given-names></name><name><surname>Disz</surname><given-names>T</given-names></name><name><surname>Edwards</surname><given-names>RA</given-names></name><name><surname>Formsma</surname><given-names>K</given-names></name><name><surname>Gerdes</surname><given-names>S</given-names></name><name><surname>Glass</surname><given-names>EM</given-names></name><name><surname>Kubal</surname><given-names>M</given-names></name><name><surname>Meyer</surname><given-names>F</given-names></name><name><surname>Olsen</surname><given-names>GJ</given-names></name><name><surname>Olson</surname><given-names>R</given-names></name><name><surname>Osterman</surname><given-names>AL</given-names></name><name><surname>Overbeek</surname><given-names>RA</given-names></name><name><surname>McNeil</surname><given-names>LK</given-names></name><name><surname>Paarmann</surname><given-names>D</given-names></name><name><surname>Paczian</surname><given-names>T</given-names></name><name><surname>Parrello</surname><given-names>B</given-names></name><name><surname>Pusch</surname><given-names>GD</given-names></name><name><surname>Reich</surname><given-names>C</given-names></name><name><surname>Stevens</surname><given-names>R</given-names></name><name><surname>Vassieva</surname><given-names>O</given-names></name><name><surname>Vonstein</surname><given-names>V</given-names></name><name><surname>Wilke</surname><given-names>A</given-names></name><name><surname>Zagnitko</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>The RAST Server: rapid annotations using subsystems technology</article-title><source>BMC Genomics</source><volume>9</volume><elocation-id>75</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2164-9-75</pub-id><pub-id pub-id-type="pmid">18261238</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bansal</surname><given-names>T</given-names></name><name><surname>Alaniz</surname><given-names>RC</given-names></name><name><surname>Wood</surname><given-names>TK</given-names></name><name><surname>Jayaraman</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The bacterial signal indole increases epithelial-cell tight-junction resistance and attenuates indicators of inflammation</article-title><source>PNAS</source><volume>107</volume><fpage>228</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1073/pnas.0906112107</pub-id><pub-id pub-id-type="pmid">19966295</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bassil</surname><given-names>AK</given-names></name><name><surname>Bourdu</surname><given-names>S</given-names></name><name><surname>Townson</surname><given-names>KA</given-names></name><name><surname>Wheeldon</surname><given-names>A</given-names></name><name><surname>Jarvie</surname><given-names>EM</given-names></name><name><surname>Zebda</surname><given-names>N</given-names></name><name><surname>Abuin</surname><given-names>A</given-names></name><name><surname>Grau</surname><given-names>E</given-names></name><name><surname>Livi</surname><given-names>GP</given-names></name><name><surname>Punter</surname><given-names>L</given-names></name><name><surname>Latcham</surname><given-names>J</given-names></name><name><surname>Grimes</surname><given-names>AM</given-names></name><name><surname>Hurp</surname><given-names>DP</given-names></name><name><surname>Downham</surname><given-names>KM</given-names></name><name><surname>Sanger</surname><given-names>GJ</given-names></name><name><surname>Winchester</surname><given-names>WJ</given-names></name><name><surname>Morrison</surname><given-names>AD</given-names></name><name><surname>Moore</surname><given-names>GBT</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>UDP-glucose modulates gastric function through P2Y14 receptor-dependent and -independent mechanisms</article-title><source>American Journal of Physiology. Gastrointestinal and Liver Physiology</source><volume>296</volume><fpage>G923</fpage><lpage>G30</lpage><pub-id pub-id-type="doi">10.1152/ajpgi.90363.2008</pub-id><pub-id pub-id-type="pmid">19164486</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Beghini</surname><given-names>F</given-names></name><name><surname>McIver</surname><given-names>LJ</given-names></name><name><surname>Blanco-Míguez</surname><given-names>A</given-names></name><name><surname>Dubois</surname><given-names>L</given-names></name><name><surname>Asnicar</surname><given-names>F</given-names></name><name><surname>Maharjan</surname><given-names>S</given-names></name><name><surname>Mailyan</surname><given-names>A</given-names></name><name><surname>Manghi</surname><given-names>P</given-names></name><name><surname>Scholz</surname><given-names>M</given-names></name><name><surname>Thomas</surname><given-names>AM</given-names></name><name><surname>Valles-Colomer</surname><given-names>M</given-names></name><name><surname>Weingart</surname><given-names>G</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zolfo</surname><given-names>M</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Segata</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Integrating taxonomic, functional, and strain-level profiling of diverse microbial communities with bioBakery 3</article-title><source>eLife</source><volume>10</volume><elocation-id>e65088</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.65088</pub-id><pub-id pub-id-type="pmid">33944776</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Belkaid</surname><given-names>Y</given-names></name><name><surname>Hand</surname><given-names>TW</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Role of the microbiota in immunity and inflammation</article-title><source>Cell</source><volume>157</volume><fpage>121</fpage><lpage>141</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.03.011</pub-id><pub-id pub-id-type="pmid">24679531</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Berg</surname><given-names>NO</given-names></name><name><surname>Dahlqvist</surname><given-names>A</given-names></name><name><surname>Lindberg</surname><given-names>T</given-names></name><name><surname>Lindstrand</surname><given-names>K</given-names></name><name><surname>Nordén</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1972">1972</year><article-title>Morphology, dipeptidases and disaccharidases of small intestinal mucosa in vitamin B 12 and folic acid deficiency</article-title><source>Scandinavian Journal of Haematology</source><volume>9</volume><fpage>167</fpage><lpage>173</lpage><pub-id pub-id-type="doi">10.1111/j.1600-0609.1972.tb00927.x</pub-id><pub-id pub-id-type="pmid">5037635</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blachier</surname><given-names>F</given-names></name><name><surname>Beaumont</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Cysteine-derived hydrogen sulfide and gut health: A matter of endogenous or bacterial origin</article-title><source>Current Opinion in Clinical Nutrition and Metabolic Care</source><volume>22</volume><fpage>68</fpage><lpage>75</lpage><pub-id pub-id-type="doi">10.1097/MCO.0000000000000526</pub-id><pub-id pub-id-type="pmid">30461448</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Borren</surname><given-names>NZ</given-names></name><name><surname>Plichta</surname><given-names>D</given-names></name><name><surname>Joshi</surname><given-names>AD</given-names></name><name><surname>Bonilla</surname><given-names>G</given-names></name><name><surname>Peng</surname><given-names>V</given-names></name><name><surname>Colizzo</surname><given-names>FP</given-names></name><name><surname>Luther</surname><given-names>J</given-names></name><name><surname>Khalili</surname><given-names>H</given-names></name><name><surname>Garber</surname><given-names>JJ</given-names></name><name><surname>Janneke van der Woude</surname><given-names>C</given-names></name><name><surname>Sadreyev</surname><given-names>R</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Alterations in fecal microbiomes and serum metabolomes of fatigued patients with quiescent inflammatory bowel diseases</article-title><source>Clinical Gastroenterology and Hepatology</source><volume>19</volume><fpage>519</fpage><lpage>527</lpage><pub-id pub-id-type="doi">10.1016/j.cgh.2020.03.013</pub-id><pub-id pub-id-type="pmid">32184182</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boyle</surname><given-names>EI</given-names></name><name><surname>Weng</surname><given-names>S</given-names></name><name><surname>Gollub</surname><given-names>J</given-names></name><name><surname>Jin</surname><given-names>H</given-names></name><name><surname>Botstein</surname><given-names>D</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Sherlock</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>GO::termfinder--open source software for accessing gene ontology information and finding significantly enriched Gene Ontology terms associated with a list of genes</article-title><source>Bioinformatics</source><volume>20</volume><fpage>3710</fpage><lpage>3715</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bth456</pub-id><pub-id pub-id-type="pmid">15297299</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bressenot</surname><given-names>A</given-names></name><name><surname>Pooya</surname><given-names>S</given-names></name><name><surname>Bossenmeyer-Pourie</surname><given-names>C</given-names></name><name><surname>Gauchotte</surname><given-names>G</given-names></name><name><surname>Germain</surname><given-names>A</given-names></name><name><surname>Chevaux</surname><given-names>JB</given-names></name><name><surname>Coste</surname><given-names>F</given-names></name><name><surname>Vignaud</surname><given-names>JM</given-names></name><name><surname>Guéant</surname><given-names>JL</given-names></name><name><surname>Peyrin-Biroulet</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Methyl donor deficiency affects small-intestinal differentiation and barrier function in rats</article-title><source>The British Journal of Nutrition</source><volume>109</volume><fpage>667</fpage><lpage>677</lpage><pub-id pub-id-type="doi">10.1017/S0007114512001869</pub-id><pub-id pub-id-type="pmid">22794784</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>EM</given-names></name><name><surname>Clardy</surname><given-names>J</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Gut microbiome lipid metabolism and its impact on host physiology</article-title><source>Cell Host &amp; Microbe</source><volume>31</volume><fpage>173</fpage><lpage>186</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2023.01.009</pub-id><pub-id pub-id-type="pmid">36758518</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bryant</surname><given-names>DA</given-names></name><name><surname>Hunter</surname><given-names>CN</given-names></name><name><surname>Warren</surname><given-names>MJ</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Biosynthesis of the modified tetrapyrroles—the pigments of life</article-title><source>Journal of Biological Chemistry</source><volume>295</volume><fpage>6888</fpage><lpage>6925</lpage><pub-id pub-id-type="doi">10.1074/jbc.REV120.006194</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buchfink</surname><given-names>B</given-names></name><name><surname>Xie</surname><given-names>C</given-names></name><name><surname>Huson</surname><given-names>DH</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Fast and sensitive protein alignment using DIAMOND</article-title><source>Nature Methods</source><volume>12</volume><fpage>59</fpage><lpage>60</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3176</pub-id><pub-id pub-id-type="pmid">25402007</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Byndloss</surname><given-names>MX</given-names></name><name><surname>Bäumler</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The germ-organ theory of non-communicable diseases</article-title><source>Nature Reviews. Microbiology</source><volume>16</volume><fpage>103</fpage><lpage>110</lpage><pub-id pub-id-type="doi">10.1038/nrmicro.2017.158</pub-id><pub-id pub-id-type="pmid">29307890</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cani</surname><given-names>PD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Human gut microbiome: hopes, threats and promises</article-title><source>Gut</source><volume>67</volume><fpage>1716</fpage><lpage>1725</lpage><pub-id pub-id-type="doi">10.1136/gutjnl-2018-316723</pub-id><pub-id pub-id-type="pmid">29934437</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Capella-Gutiérrez</surname><given-names>S</given-names></name><name><surname>Silla-Martínez</surname><given-names>JM</given-names></name><name><surname>Gabaldón</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>trimAl: A tool for automated alignment trimming in large-scale phylogenetic analyses</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1972</fpage><lpage>1973</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp348</pub-id><pub-id pub-id-type="pmid">19505945</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cevallos</surname><given-names>SA</given-names></name><name><surname>Lee</surname><given-names>JY</given-names></name><name><surname>Tiffany</surname><given-names>CR</given-names></name><name><surname>Byndloss</surname><given-names>AJ</given-names></name><name><surname>Johnston</surname><given-names>L</given-names></name><name><surname>Byndloss</surname><given-names>MX</given-names></name><name><surname>Bäumler</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Increased epithelial oxygenation links colitis to an expansion of tumorigenic bacteria</article-title><source>mBio</source><volume>10</volume><elocation-id>e02244-19</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.02244-19</pub-id><pub-id pub-id-type="pmid">31575772</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname><given-names>PP</given-names></name><name><surname>Lowe</surname><given-names>TM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>tRNAscan-SE: searching for trna genes in genomic sequences</article-title><source>Methods in Molecular Biology</source><volume>1962</volume><fpage>1</fpage><lpage>14</lpage><pub-id pub-id-type="doi">10.1007/978-1-4939-9173-0_1</pub-id><pub-id pub-id-type="pmid">31020551</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>D</given-names></name><name><surname>Dai</surname><given-names>Z</given-names></name><name><surname>Piao</surname><given-names>X</given-names></name><name><surname>Wu</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>B</given-names></name><name><surname>Zhu</surname><given-names>Y</given-names></name><name><surname>Zeng</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>L-methionine supplementation maintains the integrity and barrier function of the small-intestinal mucosa in post-weaning piglets</article-title><source>Amino Acids</source><volume>46</volume><fpage>1131</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1007/s00726-014-1675-5</pub-id><pub-id pub-id-type="pmid">24477834</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Christodoulou</surname><given-names>D</given-names></name><name><surname>Link</surname><given-names>H</given-names></name><name><surname>Fuhrer</surname><given-names>T</given-names></name><name><surname>Kochanowski</surname><given-names>K</given-names></name><name><surname>Gerosa</surname><given-names>L</given-names></name><name><surname>Sauer</surname><given-names>U</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Reserve flux capacity in the pentose phosphate pathway enables <italic>Escherichia coli’s</italic> rapid response to oxidative stress</article-title><source>Cell Systems</source><volume>6</volume><fpage>569</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2018.04.009</pub-id><pub-id pub-id-type="pmid">29753645</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Claud</surname><given-names>EC</given-names></name><name><surname>Keegan</surname><given-names>KP</given-names></name><name><surname>Brulc</surname><given-names>JM</given-names></name><name><surname>Lu</surname><given-names>L</given-names></name><name><surname>Bartels</surname><given-names>D</given-names></name><name><surname>Glass</surname><given-names>E</given-names></name><name><surname>Chang</surname><given-names>EB</given-names></name><name><surname>Meyer</surname><given-names>F</given-names></name><name><surname>Antonopoulos</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Bacterial community structure and functional contributions to emergence of health or necrotizing enterocolitis in preterm infants</article-title><source>Microbiome</source><volume>1</volume><elocation-id>20</elocation-id><pub-id pub-id-type="doi">10.1186/2049-2618-1-20</pub-id><pub-id pub-id-type="pmid">24450928</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Clausen</surname><given-names>DS</given-names></name><name><surname>Willis</surname><given-names>AD</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Evaluating replicability in microbiome data</article-title><source>Biostatistics</source><volume>23</volume><fpage>1099</fpage><lpage>1114</lpage><pub-id pub-id-type="doi">10.1093/biostatistics/kxab048</pub-id><pub-id pub-id-type="pmid">34969071</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coburn</surname><given-names>LA</given-names></name><name><surname>Gong</surname><given-names>X</given-names></name><name><surname>Singh</surname><given-names>K</given-names></name><name><surname>Asim</surname><given-names>M</given-names></name><name><surname>Scull</surname><given-names>BP</given-names></name><name><surname>Allaman</surname><given-names>MM</given-names></name><name><surname>Williams</surname><given-names>CS</given-names></name><name><surname>Rosen</surname><given-names>MJ</given-names></name><name><surname>Washington</surname><given-names>MK</given-names></name><name><surname>Barry</surname><given-names>DP</given-names></name><name><surname>Piazuelo</surname><given-names>MB</given-names></name><name><surname>Casero</surname><given-names>RA</given-names><suffix>Jr</suffix></name><name><surname>Chaturvedi</surname><given-names>R</given-names></name><name><surname>Zhao</surname><given-names>Z</given-names></name><name><surname>Wilson</surname><given-names>KT</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>L-arginine supplementation improves responses to injury and inflammation in dextran sulfate sodium colitis</article-title><source>PLOS ONE</source><volume>7</volume><elocation-id>e33546</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0033546</pub-id><pub-id pub-id-type="pmid">22428068</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Constante</surname><given-names>M</given-names></name><name><surname>Fragoso</surname><given-names>G</given-names></name><name><surname>Calvé</surname><given-names>A</given-names></name><name><surname>Samba-Mondonga</surname><given-names>M</given-names></name><name><surname>Santos</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Dietary heme induces gut dysbiosis, aggravates colitis, and potentiates the development of adenomas in mice</article-title><source>Frontiers in Microbiology</source><volume>8</volume><elocation-id>1809</elocation-id><pub-id pub-id-type="doi">10.3389/fmicb.2017.01809</pub-id><pub-id pub-id-type="pmid">28983289</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Coyte</surname><given-names>KZ</given-names></name><name><surname>Schluter</surname><given-names>J</given-names></name><name><surname>Foster</surname><given-names>KR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The ecology of the microbiome: Networks, competition, and stability</article-title><source>Science</source><volume>350</volume><fpage>663</fpage><lpage>666</lpage><pub-id pub-id-type="doi">10.1126/science.aad2602</pub-id><pub-id pub-id-type="pmid">26542567</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Culp</surname><given-names>EJ</given-names></name><name><surname>Goodman</surname><given-names>AL</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Cross-feeding in the gut microbiome: Ecology and mechanisms</article-title><source>Cell Host &amp; Microbe</source><volume>31</volume><fpage>485</fpage><lpage>499</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2023.03.016</pub-id><pub-id pub-id-type="pmid">37054671</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Degnan</surname><given-names>PH</given-names></name><name><surname>Barry</surname><given-names>NA</given-names></name><name><surname>Mok</surname><given-names>KC</given-names></name><name><surname>Taga</surname><given-names>ME</given-names></name><name><surname>Goodman</surname><given-names>AL</given-names></name></person-group><year iso-8601-date="2014">2014a</year><article-title>Human gut microbes use multiple transporters to distinguish vitamin B12 analogs and compete in the gut</article-title><source>Cell Host &amp; Microbe</source><volume>15</volume><fpage>47</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2013.12.007</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Degnan</surname><given-names>PH</given-names></name><name><surname>Taga</surname><given-names>ME</given-names></name><name><surname>Goodman</surname><given-names>AL</given-names></name></person-group><year iso-8601-date="2014">2014b</year><article-title>Vitamin B12 as a modulator of gut microbial ecology</article-title><source>Cell Metabolism</source><volume>20</volume><fpage>769</fpage><lpage>778</lpage><pub-id pub-id-type="doi">10.1016/j.cmet.2014.10.002</pub-id><pub-id pub-id-type="pmid">25440056</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>DeJongh</surname><given-names>M</given-names></name><name><surname>Formsma</surname><given-names>K</given-names></name><name><surname>Boillot</surname><given-names>P</given-names></name><name><surname>Gould</surname><given-names>J</given-names></name><name><surname>Rycenga</surname><given-names>M</given-names></name><name><surname>Best</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Toward the automated generation of genome-scale metabolic networks in the SEED</article-title><source>BMC Bioinformatics</source><volume>8</volume><elocation-id>139</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-8-139</pub-id><pub-id pub-id-type="pmid">17462086</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Devkota</surname><given-names>S</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Musch</surname><given-names>MW</given-names></name><name><surname>Leone</surname><given-names>V</given-names></name><name><surname>Fehlner-Peach</surname><given-names>H</given-names></name><name><surname>Nadimpalli</surname><given-names>A</given-names></name><name><surname>Antonopoulos</surname><given-names>DA</given-names></name><name><surname>Jabri</surname><given-names>B</given-names></name><name><surname>Chang</surname><given-names>EB</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Dietary-fat-induced taurocholic acid promotes pathobiont expansion and colitis in Il10-/- mice</article-title><source>Nature</source><volume>487</volume><fpage>104</fpage><lpage>108</lpage><pub-id pub-id-type="doi">10.1038/nature11225</pub-id><pub-id pub-id-type="pmid">22722865</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dhakan</surname><given-names>DB</given-names></name><name><surname>Maji</surname><given-names>A</given-names></name><name><surname>Sharma</surname><given-names>AK</given-names></name><name><surname>Saxena</surname><given-names>R</given-names></name><name><surname>Pulikkan</surname><given-names>J</given-names></name><name><surname>Grace</surname><given-names>T</given-names></name><name><surname>Gomez</surname><given-names>A</given-names></name><name><surname>Scaria</surname><given-names>J</given-names></name><name><surname>Amato</surname><given-names>KR</given-names></name><name><surname>Sharma</surname><given-names>VK</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The unique composition of Indian gut microbiome, gene catalogue, and associated fecal metabolome deciphered using multi-omics approaches</article-title><source>GigaScience</source><volume>8</volume><elocation-id>giz004</elocation-id><pub-id pub-id-type="doi">10.1093/gigascience/giz004</pub-id><pub-id pub-id-type="pmid">30698687</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dias</surname><given-names>O</given-names></name><name><surname>Rocha</surname><given-names>M</given-names></name><name><surname>Ferreira</surname><given-names>EC</given-names></name><name><surname>Rocha</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Reconstructing genome-scale metabolic models with merlin</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>3899</fpage><lpage>3910</lpage><pub-id pub-id-type="doi">10.1093/nar/gkv294</pub-id><pub-id pub-id-type="pmid">25845595</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eddy</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Accelerated Profile HMM Searches</article-title><source>PLOS Computational Biology</source><volume>7</volume><elocation-id>e1002195</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1002195</pub-id><pub-id pub-id-type="pmid">22039361</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>1792</fpage><lpage>1797</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id><pub-id pub-id-type="pmid">15034147</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Vineis</surname><given-names>JH</given-names></name><name><surname>Morrison</surname><given-names>HG</given-names></name><name><surname>Sogin</surname><given-names>ML</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>A filtering method to generate high quality short reads using illumina paired-end technology</article-title><source>PLOS ONE</source><volume>8</volume><elocation-id>e66643</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0066643</pub-id><pub-id pub-id-type="pmid">23799126</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Eren</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>Reads-for-assembly</data-title><version designator="swh:1:rev:cc5c9a22c8d26688e2c156878438e6fe0777ddbb">swh:1:rev:cc5c9a22c8d26688e2c156878438e6fe0777ddbb</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:7bd125c4df26fd24a3d9df69c24bb2a16a3c51d2;origin=https://github.com/merenlab/reads-for-assembly;visit=swh:1:snp:db631f474996123dccaf9c1fb223c5e6cfeb8c0e;anchor=swh:1:rev:cc5c9a22c8d26688e2c156878438e6fe0777ddbb">https://archive.softwareheritage.org/swh:1:dir:7bd125c4df26fd24a3d9df69c24bb2a16a3c51d2;origin=https://github.com/merenlab/reads-for-assembly;visit=swh:1:snp:db631f474996123dccaf9c1fb223c5e6cfeb8c0e;anchor=swh:1:rev:cc5c9a22c8d26688e2c156878438e6fe0777ddbb</ext-link></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname><given-names>Y</given-names></name><name><surname>Pedersen</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Gut microbiota in human metabolic health and disease</article-title><source>Nature Reviews. Microbiology</source><volume>19</volume><fpage>55</fpage><lpage>71</lpage><pub-id pub-id-type="doi">10.1038/s41579-020-0433-9</pub-id><pub-id pub-id-type="pmid">32887946</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname><given-names>X</given-names></name><name><surname>Lloyd</surname><given-names>CJ</given-names></name><name><surname>Palsson</surname><given-names>BO</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Reconstructing organisms in silico: genome-scale models and their emerging applications</article-title><source>Nature Reviews. Microbiology</source><volume>18</volume><fpage>731</fpage><lpage>743</lpage><pub-id pub-id-type="doi">10.1038/s41579-020-00440-4</pub-id><pub-id pub-id-type="pmid">32958892</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farag</surname><given-names>IF</given-names></name><name><surname>Biddle</surname><given-names>JF</given-names></name><name><surname>Zhao</surname><given-names>R</given-names></name><name><surname>Martino</surname><given-names>AJ</given-names></name><name><surname>House</surname><given-names>CH</given-names></name><name><surname>León-Zayas</surname><given-names>RI</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Metabolic potentials of archaeal lineages resolved from metagenomes of deep Costa Rica sediments</article-title><source>The ISME Journal</source><volume>14</volume><fpage>1345</fpage><lpage>1358</lpage><pub-id pub-id-type="doi">10.1038/s41396-020-0615-5</pub-id><pub-id pub-id-type="pmid">32066876</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Faria</surname><given-names>JP</given-names></name><name><surname>Rocha</surname><given-names>M</given-names></name><name><surname>Rocha</surname><given-names>I</given-names></name><name><surname>Henry</surname><given-names>CS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Methods for automated genome-scale metabolic model reconstruction</article-title><source>Biochemical Society Transactions</source><volume>46</volume><fpage>931</fpage><lpage>936</lpage><pub-id pub-id-type="doi">10.1042/BST20170246</pub-id><pub-id pub-id-type="pmid">30065105</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Faure</surname><given-names>M</given-names></name><name><surname>Moënnoz</surname><given-names>D</given-names></name><name><surname>Montigon</surname><given-names>F</given-names></name><name><surname>Mettraux</surname><given-names>C</given-names></name><name><surname>Breuillé</surname><given-names>D</given-names></name><name><surname>Ballèvre</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Dietary threonine restriction specifically reduces intestinal mucin synthesis in rats</article-title><source>The Journal of Nutrition</source><volume>135</volume><fpage>486</fpage><lpage>491</lpage><pub-id pub-id-type="doi">10.1093/jn/135.3.486</pub-id><pub-id pub-id-type="pmid">15735082</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Faure</surname><given-names>M</given-names></name><name><surname>Mettraux</surname><given-names>C</given-names></name><name><surname>Moennoz</surname><given-names>D</given-names></name><name><surname>Godin</surname><given-names>J-P</given-names></name><name><surname>Vuichoud</surname><given-names>J</given-names></name><name><surname>Rochat</surname><given-names>F</given-names></name><name><surname>Breuillé</surname><given-names>D</given-names></name><name><surname>Obled</surname><given-names>C</given-names></name><name><surname>Corthésy-Theulaz</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Specific amino acids increase mucin synthesis and microbiota in dextran sulfate sodium-treated rats</article-title><source>The Journal of Nutrition</source><volume>136</volume><fpage>1558</fpage><lpage>1564</lpage><pub-id pub-id-type="doi">10.1093/jn/136.6.1558</pub-id><pub-id pub-id-type="pmid">16702321</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name><name><surname>Jia</surname><given-names>H</given-names></name><name><surname>Stadlmayr</surname><given-names>A</given-names></name><name><surname>Tang</surname><given-names>L</given-names></name><name><surname>Lan</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>D</given-names></name><name><surname>Xia</surname><given-names>H</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Jie</surname><given-names>Z</given-names></name><name><surname>Su</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Xiao</surname><given-names>L</given-names></name><name><surname>Huber-Schönauer</surname><given-names>U</given-names></name><name><surname>Niederseer</surname><given-names>D</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Al-Aama</surname><given-names>JY</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Kristiansen</surname><given-names>K</given-names></name><name><surname>Arumugam</surname><given-names>M</given-names></name><name><surname>Tilg</surname><given-names>H</given-names></name><name><surname>Datz</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Gut microbiome development along the colorectal adenoma-carcinoma sequence</article-title><source>Nature Communications</source><volume>6</volume><elocation-id>6528</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms7528</pub-id><pub-id pub-id-type="pmid">25758642</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>L</given-names></name><name><surname>Raman</surname><given-names>AS</given-names></name><name><surname>Hibberd</surname><given-names>MC</given-names></name><name><surname>Cheng</surname><given-names>J</given-names></name><name><surname>Griffin</surname><given-names>NW</given-names></name><name><surname>Peng</surname><given-names>Y</given-names></name><name><surname>Leyn</surname><given-names>SA</given-names></name><name><surname>Rodionov</surname><given-names>DA</given-names></name><name><surname>Osterman</surname><given-names>AL</given-names></name><name><surname>Gordon</surname><given-names>JI</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Identifying determinants of bacterial fitness in a model of human gut microbial succession</article-title><source>PNAS</source><volume>117</volume><fpage>2622</fpage><lpage>2633</lpage><pub-id pub-id-type="doi">10.1073/pnas.1918951117</pub-id><pub-id pub-id-type="pmid">31969452</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Fithian</surname><given-names>W</given-names></name><name><surname>Sun</surname><given-names>D</given-names></name><name><surname>Taylor</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Optimal Inference After Model Selection</article-title><source>arXiv</source><pub-id pub-id-type="doi">10.48550/arXiv.1410.2597</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Sirota-Madi</surname><given-names>A</given-names></name><name><surname>Avila-Pacheco</surname><given-names>J</given-names></name><name><surname>Fornelos</surname><given-names>N</given-names></name><name><surname>Haiser</surname><given-names>HJ</given-names></name><name><surname>Reinker</surname><given-names>S</given-names></name><name><surname>Vatanen</surname><given-names>T</given-names></name><name><surname>Hall</surname><given-names>AB</given-names></name><name><surname>Mallick</surname><given-names>H</given-names></name><name><surname>McIver</surname><given-names>LJ</given-names></name><name><surname>Sauk</surname><given-names>JS</given-names></name><name><surname>Wilson</surname><given-names>RG</given-names></name><name><surname>Stevens</surname><given-names>BW</given-names></name><name><surname>Scott</surname><given-names>JM</given-names></name><name><surname>Pierce</surname><given-names>K</given-names></name><name><surname>Deik</surname><given-names>AA</given-names></name><name><surname>Bullock</surname><given-names>K</given-names></name><name><surname>Imhann</surname><given-names>F</given-names></name><name><surname>Porter</surname><given-names>JA</given-names></name><name><surname>Zhernakova</surname><given-names>A</given-names></name><name><surname>Fu</surname><given-names>J</given-names></name><name><surname>Weersma</surname><given-names>RK</given-names></name><name><surname>Wijmenga</surname><given-names>C</given-names></name><name><surname>Clish</surname><given-names>CB</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Gut microbiome structure and metabolic activity in inflammatory bowel disease</article-title><source>Nature Microbiology</source><volume>4</volume><fpage>293</fpage><lpage>305</lpage><pub-id pub-id-type="doi">10.1038/s41564-018-0306-4</pub-id><pub-id pub-id-type="pmid">30531976</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Galperin</surname><given-names>MY</given-names></name><name><surname>Makarova</surname><given-names>KS</given-names></name><name><surname>Wolf</surname><given-names>YI</given-names></name><name><surname>Koonin</surname><given-names>EV</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Expanded microbial genome coverage and improved protein family annotation in the COG database</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>D261</fpage><lpage>D9</lpage><pub-id pub-id-type="doi">10.1093/nar/gku1223</pub-id><pub-id pub-id-type="pmid">25428365</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Galperin</surname><given-names>MY</given-names></name><name><surname>Wolf</surname><given-names>YI</given-names></name><name><surname>Makarova</surname><given-names>KS</given-names></name><name><surname>Vera Alvarez</surname><given-names>R</given-names></name><name><surname>Landsman</surname><given-names>D</given-names></name><name><surname>Koonin</surname><given-names>EV</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>COG database update: focus on microbial diversity, model organisms, and widespread pathogens</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>D274</fpage><lpage>D281</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa1018</pub-id><pub-id pub-id-type="pmid">33167031</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garschagen</surname><given-names>LS</given-names></name><name><surname>Franke</surname><given-names>T</given-names></name><name><surname>Deppenmeier</surname><given-names>U</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>An alternative pentose phosphate pathway in human gut bacteria for the degradation of C5 sugars in dietary fibers</article-title><source>The FEBS Journal</source><volume>288</volume><fpage>1839</fpage><lpage>1858</lpage><pub-id pub-id-type="doi">10.1111/febs.15511</pub-id><pub-id pub-id-type="pmid">32770699</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Geller-McGrath</surname><given-names>D</given-names></name><name><surname>Konwar</surname><given-names>KM</given-names></name><name><surname>Edgcomb</surname><given-names>VP</given-names></name><name><surname>Pachiadaki</surname><given-names>M</given-names></name><name><surname>Roddy</surname><given-names>JW</given-names></name><name><surname>Wheeler</surname><given-names>TJ</given-names></name><name><surname>McDermott</surname><given-names>JE</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>MetaPathPredict: A machine learning-based tool for predicting metabolic modules in incomplete bacterial genomes</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.12.21.521254</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gevers</surname><given-names>D</given-names></name><name><surname>Kugathasan</surname><given-names>S</given-names></name><name><surname>Denson</surname><given-names>LA</given-names></name><name><surname>Vázquez-Baeza</surname><given-names>Y</given-names></name><name><surname>Van Treuren</surname><given-names>W</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Schwager</surname><given-names>E</given-names></name><name><surname>Knights</surname><given-names>D</given-names></name><name><surname>Song</surname><given-names>SJ</given-names></name><name><surname>Yassour</surname><given-names>M</given-names></name><name><surname>Morgan</surname><given-names>XC</given-names></name><name><surname>Kostic</surname><given-names>AD</given-names></name><name><surname>Luo</surname><given-names>C</given-names></name><name><surname>González</surname><given-names>A</given-names></name><name><surname>McDonald</surname><given-names>D</given-names></name><name><surname>Haberman</surname><given-names>Y</given-names></name><name><surname>Walters</surname><given-names>T</given-names></name><name><surname>Baker</surname><given-names>S</given-names></name><name><surname>Rosh</surname><given-names>J</given-names></name><name><surname>Stephens</surname><given-names>M</given-names></name><name><surname>Heyman</surname><given-names>M</given-names></name><name><surname>Markowitz</surname><given-names>J</given-names></name><name><surname>Baldassano</surname><given-names>R</given-names></name><name><surname>Griffiths</surname><given-names>A</given-names></name><name><surname>Sylvester</surname><given-names>F</given-names></name><name><surname>Mack</surname><given-names>D</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Crandall</surname><given-names>W</given-names></name><name><surname>Hyams</surname><given-names>J</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The treatment-naive microbiome in new-onset Crohn’s disease</article-title><source>Cell Host &amp; Microbe</source><volume>15</volume><fpage>382</fpage><lpage>392</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2014.02.005</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gobert</surname><given-names>AP</given-names></name><name><surname>Cheng</surname><given-names>Y</given-names></name><name><surname>Akhtar</surname><given-names>M</given-names></name><name><surname>Mersey</surname><given-names>BD</given-names></name><name><surname>Blumberg</surname><given-names>DR</given-names></name><name><surname>Cross</surname><given-names>RK</given-names></name><name><surname>Chaturvedi</surname><given-names>R</given-names></name><name><surname>Drachenberg</surname><given-names>CB</given-names></name><name><surname>Boucher</surname><given-names>J-L</given-names></name><name><surname>Hacker</surname><given-names>A</given-names></name><name><surname>Casero</surname><given-names>RA</given-names><suffix>Jr</suffix></name><name><surname>Wilson</surname><given-names>KT</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Protective role of arginase in a mouse model of colitis</article-title><source>The Journal of Immunology</source><volume>173</volume><fpage>2109</fpage><lpage>2117</lpage><pub-id pub-id-type="doi">10.4049/jimmunol.173.3.2109</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gojda</surname><given-names>J</given-names></name><name><surname>Cahova</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Gut microbiota as the link between elevated BCAA serum levels and insulin resistance</article-title><source>Biomolecules</source><volume>11</volume><elocation-id>1414</elocation-id><pub-id pub-id-type="doi">10.3390/biom11101414</pub-id><pub-id pub-id-type="pmid">34680047</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goodman</surname><given-names>AL</given-names></name><name><surname>McNulty</surname><given-names>NP</given-names></name><name><surname>Zhao</surname><given-names>Y</given-names></name><name><surname>Leip</surname><given-names>D</given-names></name><name><surname>Mitra</surname><given-names>RD</given-names></name><name><surname>Lozupone</surname><given-names>CA</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Gordon</surname><given-names>JI</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Identifying genetic determinants needed to establish a human gut symbiont in its habitat</article-title><source>Cell Host &amp; Microbe</source><volume>6</volume><fpage>279</fpage><lpage>289</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2009.08.003</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Gruss</surname><given-names>A</given-names></name><name><surname>Borezée-Durant</surname><given-names>E</given-names></name><name><surname>Lechardeur</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2012">2012</year><chapter-title>Chapter three - environmental heme utilization by heme-auxotrophic bacteria</chapter-title><person-group person-group-type="editor"><name><surname>Poole</surname><given-names>RK</given-names></name></person-group><source>In Advances in Microbial Physiology</source><publisher-name>Academic Press</publisher-name><fpage>69</fpage><lpage>124</lpage><pub-id pub-id-type="doi">10.1016/B978-0-12-394423-8.00003-2</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname><given-names>C</given-names></name><name><surname>Kim</surname><given-names>GB</given-names></name><name><surname>Kim</surname><given-names>WJ</given-names></name><name><surname>Kim</surname><given-names>HU</given-names></name><name><surname>Lee</surname><given-names>SY</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Current status and applications of genome-scale metabolic models</article-title><source>Genome Biology</source><volume>20</volume><elocation-id>121</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-019-1730-3</pub-id><pub-id pub-id-type="pmid">31196170</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Halfvarson</surname><given-names>J</given-names></name><name><surname>Brislawn</surname><given-names>CJ</given-names></name><name><surname>Lamendella</surname><given-names>R</given-names></name><name><surname>Vázquez-Baeza</surname><given-names>Y</given-names></name><name><surname>Walters</surname><given-names>WA</given-names></name><name><surname>Bramer</surname><given-names>LM</given-names></name><name><surname>D’Amato</surname><given-names>M</given-names></name><name><surname>Bonfiglio</surname><given-names>F</given-names></name><name><surname>McDonald</surname><given-names>D</given-names></name><name><surname>Gonzalez</surname><given-names>A</given-names></name><name><surname>McClure</surname><given-names>EE</given-names></name><name><surname>Dunklebarger</surname><given-names>MF</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Jansson</surname><given-names>JK</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Dynamics of the human gut microbiome in inflammatory bowel disease</article-title><source>Nature Microbiology</source><volume>2</volume><elocation-id>17004</elocation-id><pub-id pub-id-type="doi">10.1038/nmicrobiol.2017.4</pub-id><pub-id pub-id-type="pmid">28191884</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hardivillé</surname><given-names>S</given-names></name><name><surname>Hart</surname><given-names>GW</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Nutrient regulation of signaling, transcription, and cell physiology by O-GlcNAcylation</article-title><source>Cell Metabolism</source><volume>20</volume><fpage>208</fpage><lpage>213</lpage><pub-id pub-id-type="doi">10.1016/j.cmet.2014.07.014</pub-id><pub-id pub-id-type="pmid">25100062</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hashimoto</surname><given-names>T</given-names></name><name><surname>Perlot</surname><given-names>T</given-names></name><name><surname>Rehman</surname><given-names>A</given-names></name><name><surname>Trichereau</surname><given-names>J</given-names></name><name><surname>Ishiguro</surname><given-names>H</given-names></name><name><surname>Paolino</surname><given-names>M</given-names></name><name><surname>Sigl</surname><given-names>V</given-names></name><name><surname>Hanada</surname><given-names>T</given-names></name><name><surname>Hanada</surname><given-names>R</given-names></name><name><surname>Lipinski</surname><given-names>S</given-names></name><name><surname>Wild</surname><given-names>B</given-names></name><name><surname>Camargo</surname><given-names>SMR</given-names></name><name><surname>Singer</surname><given-names>D</given-names></name><name><surname>Richter</surname><given-names>A</given-names></name><name><surname>Kuba</surname><given-names>K</given-names></name><name><surname>Fukamizu</surname><given-names>A</given-names></name><name><surname>Schreiber</surname><given-names>S</given-names></name><name><surname>Clevers</surname><given-names>H</given-names></name><name><surname>Verrey</surname><given-names>F</given-names></name><name><surname>Rosenstiel</surname><given-names>P</given-names></name><name><surname>Penninger</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>ACE2 links amino acid malnutrition to microbial ecology and intestinal inflammation</article-title><source>Nature</source><volume>487</volume><fpage>477</fpage><lpage>481</lpage><pub-id pub-id-type="doi">10.1038/nature11228</pub-id><pub-id pub-id-type="pmid">22837003</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Heinken</surname><given-names>A</given-names></name><name><surname>Hertel</surname><given-names>J</given-names></name><name><surname>Thiele</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Metabolic modelling reveals broad changes in gut microbial metabolism in inflammatory bowel disease patients with dysbiosis</article-title><source>NPJ Systems Biology and Applications</source><volume>7</volume><elocation-id>19</elocation-id><pub-id pub-id-type="doi">10.1038/s41540-021-00178-6</pub-id><pub-id pub-id-type="pmid">33958598</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hellmann</surname><given-names>J</given-names></name><name><surname>Ta</surname><given-names>A</given-names></name><name><surname>Ollberding</surname><given-names>NJ</given-names></name><name><surname>Bezold</surname><given-names>R</given-names></name><name><surname>Lake</surname><given-names>K</given-names></name><name><surname>Jackson</surname><given-names>K</given-names></name><name><surname>Dirksing</surname><given-names>K</given-names></name><name><surname>Bonkowski</surname><given-names>E</given-names></name><name><surname>Haslam</surname><given-names>DB</given-names></name><name><surname>Denson</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Patient-reported outcomes correlate with microbial community composition independent of mucosal inflammation in pediatric inflammatory bowel disease</article-title><source>Inflammatory Bowel Diseases</source><volume>29</volume><fpage>286</fpage><lpage>296</lpage><pub-id pub-id-type="doi">10.1093/ibd/izac175</pub-id><pub-id pub-id-type="pmid">35972440</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Henke</surname><given-names>MT</given-names></name><name><surname>Kenny</surname><given-names>DJ</given-names></name><name><surname>Cassilly</surname><given-names>CD</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Clardy</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title><italic>Ruminococcus gnavus</italic>, a member of the human gut microbiome associated with Crohn’s disease, produces an inflammatory polysaccharide</article-title><source>PNAS</source><volume>116</volume><fpage>12672</fpage><lpage>12677</lpage><pub-id pub-id-type="doi">10.1073/pnas.1904099116</pub-id><pub-id pub-id-type="pmid">31182571</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Henry</surname><given-names>CS</given-names></name><name><surname>DeJongh</surname><given-names>M</given-names></name><name><surname>Best</surname><given-names>AA</given-names></name><name><surname>Frybarger</surname><given-names>PM</given-names></name><name><surname>Linsay</surname><given-names>B</given-names></name><name><surname>Stevens</surname><given-names>RL</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>High-throughput generation, optimization and analysis of genome-scale metabolic models</article-title><source>Nature Biotechnology</source><volume>28</volume><fpage>977</fpage><lpage>982</lpage><pub-id pub-id-type="doi">10.1038/nbt.1672</pub-id><pub-id pub-id-type="pmid">20802497</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Herrmann</surname><given-names>KM</given-names></name><name><surname>Weaver</surname><given-names>LM</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>The shikimate pathway</article-title><source>Annual Review of Plant Physiology and Plant Molecular Biology</source><volume>50</volume><fpage>473</fpage><lpage>503</lpage><pub-id pub-id-type="doi">10.1146/annurev.arplant.50.1.473</pub-id><pub-id pub-id-type="pmid">15012217</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hijova</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Gut bacterial metabolites of indigestible polysaccharides in intestinal fermentation as mediators of public health</article-title><source>Bratislavske Lekarske Listy</source><volume>120</volume><fpage>807</fpage><lpage>812</lpage><pub-id pub-id-type="doi">10.4149/BLL_2019_134</pub-id><pub-id pub-id-type="pmid">31747759</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hossain</surname><given-names>KS</given-names></name><name><surname>Amarasena</surname><given-names>S</given-names></name><name><surname>Mayengbam</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>B vitamins and their roles in gut health</article-title><source>Microorganisms</source><volume>10</volume><elocation-id>1168</elocation-id><pub-id pub-id-type="doi">10.3390/microorganisms10061168</pub-id><pub-id pub-id-type="pmid">35744686</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hove-Jensen</surname><given-names>B</given-names></name><name><surname>Andersen</surname><given-names>KR</given-names></name><name><surname>Kilstrup</surname><given-names>M</given-names></name><name><surname>Martinussen</surname><given-names>J</given-names></name><name><surname>Switzer</surname><given-names>RL</given-names></name><name><surname>Willemoës</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Phosphoribosyl Diphosphate (PRPP): Biosynthesis, enzymology, utilization, and metabolic significance</article-title><source>Microbiology and Molecular Biology Reviews</source><volume>81</volume><elocation-id>e00040-16</elocation-id><pub-id pub-id-type="doi">10.1128/MMBR.00040-16</pub-id><pub-id pub-id-type="pmid">28031352</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huergo</surname><given-names>LF</given-names></name><name><surname>Dixon</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The emergence of 2-oxoglutarate as a master regulator metabolite</article-title><source>Microbiology and Molecular Biology Reviews</source><volume>79</volume><fpage>419</fpage><lpage>435</lpage><pub-id pub-id-type="doi">10.1128/MMBR.00038-15</pub-id><pub-id pub-id-type="pmid">26424716</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>Human Microbiome Project Consortium</collab></person-group><year iso-8601-date="2012">2012</year><article-title>A framework for human microbiome research</article-title><source>Nature</source><volume>486</volume><fpage>215</fpage><lpage>221</lpage><pub-id pub-id-type="doi">10.1038/nature11209</pub-id><pub-id pub-id-type="pmid">22699610</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Matplotlib: a 2D graphics environment</article-title><source>Computing in Science &amp; Engineering</source><volume>9</volume><fpage>90</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hyatt</surname><given-names>D</given-names></name><name><surname>Chen</surname><given-names>G-L</given-names></name><name><surname>Locascio</surname><given-names>PF</given-names></name><name><surname>Land</surname><given-names>ML</given-names></name><name><surname>Larimer</surname><given-names>FW</given-names></name><name><surname>Hauser</surname><given-names>LJ</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Prodigal: prokaryotic gene recognition and translation initiation site identification</article-title><source>BMC Bioinformatics</source><volume>11</volume><elocation-id>119</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-11-119</pub-id><pub-id pub-id-type="pmid">20211023</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ijssennagger</surname><given-names>N</given-names></name><name><surname>Belzer</surname><given-names>C</given-names></name><name><surname>Hooiveld</surname><given-names>GJ</given-names></name><name><surname>Dekker</surname><given-names>J</given-names></name><name><surname>van Mil</surname><given-names>SWC</given-names></name><name><surname>Müller</surname><given-names>M</given-names></name><name><surname>Kleerebezem</surname><given-names>M</given-names></name><name><surname>van der Meer</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Gut microbiota facilitates dietary heme-induced epithelial hyperproliferation by opening the mucus barrier in colon</article-title><source>PNAS</source><volume>112</volume><fpage>10038</fpage><lpage>10043</lpage><pub-id pub-id-type="doi">10.1073/pnas.1507645112</pub-id><pub-id pub-id-type="pmid">26216954</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jaffe</surname><given-names>AL</given-names></name><name><surname>Castelle</surname><given-names>CJ</given-names></name><name><surname>Matheus Carnevali</surname><given-names>PB</given-names></name><name><surname>Gribaldo</surname><given-names>S</given-names></name><name><surname>Banfield</surname><given-names>JF</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The rise of diversity in metabolic platforms across the Candidate Phyla Radiation</article-title><source>BMC Biology</source><volume>18</volume><elocation-id>69</elocation-id><pub-id pub-id-type="doi">10.1186/s12915-020-00804-5</pub-id><pub-id pub-id-type="pmid">32560683</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jansson</surname><given-names>J</given-names></name><name><surname>Willing</surname><given-names>B</given-names></name><name><surname>Lucio</surname><given-names>M</given-names></name><name><surname>Fekete</surname><given-names>A</given-names></name><name><surname>Dicksved</surname><given-names>J</given-names></name><name><surname>Halfvarson</surname><given-names>J</given-names></name><name><surname>Tysk</surname><given-names>C</given-names></name><name><surname>Schmitt-Kopplin</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Metabolomics reveals metabolic biomarkers of Crohn’s disease</article-title><source>PLOS ONE</source><volume>4</volume><elocation-id>e6386</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0006386</pub-id><pub-id pub-id-type="pmid">19636438</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jodorkovsky</surname><given-names>D</given-names></name><name><surname>Young</surname><given-names>Y</given-names></name><name><surname>Abreu</surname><given-names>MT</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Clinical outcomes of patients with ulcerative colitis and co-existing Clostridium difficile infection</article-title><source>Digestive Diseases and Sciences</source><volume>55</volume><fpage>415</fpage><lpage>420</lpage><pub-id pub-id-type="doi">10.1007/s10620-009-0749-9</pub-id><pub-id pub-id-type="pmid">19255850</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johansson</surname><given-names>MEV</given-names></name><name><surname>Gustafsson</surname><given-names>JK</given-names></name><name><surname>Sjöberg</surname><given-names>KE</given-names></name><name><surname>Petersson</surname><given-names>J</given-names></name><name><surname>Holm</surname><given-names>L</given-names></name><name><surname>Sjövall</surname><given-names>H</given-names></name><name><surname>Hansson</surname><given-names>GC</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Bacteria penetrate the inner mucus layer before inflammation in the dextran sulfate colitis model</article-title><source>PLOS ONE</source><volume>5</volume><elocation-id>e12238</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0012238</pub-id><pub-id pub-id-type="pmid">20805871</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johansson</surname><given-names>MEV</given-names></name><name><surname>Hansson</surname><given-names>GC</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Immunological aspects of intestinal mucus and mucins</article-title><source>Nature Reviews. Immunology</source><volume>16</volume><fpage>639</fpage><lpage>649</lpage><pub-id pub-id-type="doi">10.1038/nri.2016.88</pub-id><pub-id pub-id-type="pmid">27498766</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Joossens</surname><given-names>M</given-names></name><name><surname>Huys</surname><given-names>G</given-names></name><name><surname>Cnockaert</surname><given-names>M</given-names></name><name><surname>De Preter</surname><given-names>V</given-names></name><name><surname>Verbeke</surname><given-names>K</given-names></name><name><surname>Rutgeerts</surname><given-names>P</given-names></name><name><surname>Vandamme</surname><given-names>P</given-names></name><name><surname>Vermeire</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Dysbiosis of the faecal microbiota in patients with Crohn’s disease and their unaffected relatives</article-title><source>Gut</source><volume>60</volume><fpage>631</fpage><lpage>637</lpage><pub-id pub-id-type="doi">10.1136/gut.2010.223263</pub-id><pub-id pub-id-type="pmid">21209126</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kananen</surname><given-names>K</given-names></name><name><surname>Veseli</surname><given-names>I</given-names></name><name><surname>Quiles Pérez</surname><given-names>CJ</given-names></name><name><surname>Miller</surname><given-names>SE</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Bradley</surname><given-names>PH</given-names></name><name><surname>Miller</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Adaptive adjustment of profile HMM significance thresholds improves functional and metabolic insights into microbial genomes</article-title><source>Bioinformatics Advances</source><volume>5</volume><elocation-id>vbaf039</elocation-id><pub-id pub-id-type="doi">10.1093/bioadv/vbaf039</pub-id><pub-id pub-id-type="pmid">40177264</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname><given-names>M</given-names></name><name><surname>Goto</surname><given-names>S</given-names></name><name><surname>Hattori</surname><given-names>M</given-names></name><name><surname>Aoki-Kinoshita</surname><given-names>KF</given-names></name><name><surname>Itoh</surname><given-names>M</given-names></name><name><surname>Kawashima</surname><given-names>S</given-names></name><name><surname>Katayama</surname><given-names>T</given-names></name><name><surname>Araki</surname><given-names>M</given-names></name><name><surname>Hirakawa</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>From genomics to chemical genomics: new developments in KEGG</article-title><source>Nucleic Acids Research</source><volume>34</volume><fpage>D354</fpage><lpage>D7</lpage><pub-id pub-id-type="doi">10.1093/nar/gkj102</pub-id><pub-id pub-id-type="pmid">16381885</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname><given-names>M</given-names></name><name><surname>Goto</surname><given-names>S</given-names></name><name><surname>Sato</surname><given-names>Y</given-names></name><name><surname>Furumichi</surname><given-names>M</given-names></name><name><surname>Tanabe</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>KEGG for integration and interpretation of large-scale molecular data sets</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>D109</fpage><lpage>D14</lpage><pub-id pub-id-type="doi">10.1093/nar/gkr988</pub-id><pub-id pub-id-type="pmid">22080510</pub-id></element-citation></ref><ref id="bib93"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname><given-names>M</given-names></name><name><surname>Furumichi</surname><given-names>M</given-names></name><name><surname>Sato</surname><given-names>Y</given-names></name><name><surname>Kawashima</surname><given-names>M</given-names></name><name><surname>Ishiguro-Watanabe</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>KEGG for taxonomy-based analysis of pathways and genomes</article-title><source>Nucleic Acids Research</source><volume>51</volume><fpage>D587</fpage><lpage>D592</lpage><pub-id pub-id-type="doi">10.1093/nar/gkac963</pub-id><pub-id pub-id-type="pmid">36300620</pub-id></element-citation></ref><ref id="bib94"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaplan</surname><given-names>GG</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The global burden of IBD: from 2015 to 2025</article-title><source>Nature Reviews. Gastroenterology &amp; Hepatology</source><volume>12</volume><fpage>720</fpage><lpage>727</lpage><pub-id pub-id-type="doi">10.1038/nrgastro.2015.150</pub-id><pub-id pub-id-type="pmid">26323879</pub-id></element-citation></ref><ref id="bib95"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Karp</surname><given-names>PD</given-names></name><name><surname>Latendresse</surname><given-names>M</given-names></name><name><surname>Paley</surname><given-names>SM</given-names></name><name><surname>Krummenacker</surname><given-names>M</given-names></name><name><surname>Ong</surname><given-names>QD</given-names></name><name><surname>Billington</surname><given-names>R</given-names></name><name><surname>Kothari</surname><given-names>A</given-names></name><name><surname>Weaver</surname><given-names>D</given-names></name><name><surname>Lee</surname><given-names>T</given-names></name><name><surname>Subhraveti</surname><given-names>P</given-names></name><name><surname>Spaulding</surname><given-names>A</given-names></name><name><surname>Fulcher</surname><given-names>C</given-names></name><name><surname>Keseler</surname><given-names>IM</given-names></name><name><surname>Caspi</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Pathway tools version 19.0 update: software for pathway/genome informatics and systems biology</article-title><source>arXiv</source><pub-id pub-id-type="doi">10.48550/arXiv.1510.03964</pub-id></element-citation></ref><ref id="bib96"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karp</surname><given-names>PD</given-names></name><name><surname>Midford</surname><given-names>PE</given-names></name><name><surname>Billington</surname><given-names>R</given-names></name><name><surname>Kothari</surname><given-names>A</given-names></name><name><surname>Krummenacker</surname><given-names>M</given-names></name><name><surname>Latendresse</surname><given-names>M</given-names></name><name><surname>Ong</surname><given-names>WK</given-names></name><name><surname>Subhraveti</surname><given-names>P</given-names></name><name><surname>Caspi</surname><given-names>R</given-names></name><name><surname>Fulcher</surname><given-names>C</given-names></name><name><surname>Keseler</surname><given-names>IM</given-names></name><name><surname>Paley</surname><given-names>SM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Pathway Tools version 23.0 update: software for pathway/genome informatics and systems biology</article-title><source>Briefings in Bioinformatics</source><volume>22</volume><fpage>109</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.1093/bib/bbz104</pub-id></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Katayama</surname><given-names>S</given-names></name><name><surname>Mine</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Antioxidative activity of amino acids on tissue oxidative stress in human intestinal epithelial cell model</article-title><source>Journal of Agricultural and Food Chemistry</source><volume>55</volume><fpage>8458</fpage><lpage>8464</lpage><pub-id pub-id-type="doi">10.1021/jf070866p</pub-id><pub-id pub-id-type="pmid">17883254</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kelly</surname><given-names>CJ</given-names></name><name><surname>Alexeev</surname><given-names>EE</given-names></name><name><surname>Farb</surname><given-names>L</given-names></name><name><surname>Vickery</surname><given-names>TW</given-names></name><name><surname>Zheng</surname><given-names>L</given-names></name><name><surname>Eric L</surname><given-names>C</given-names></name><name><surname>Kitzenberg</surname><given-names>DA</given-names></name><name><surname>Battista</surname><given-names>KD</given-names></name><name><surname>Kominsky</surname><given-names>DJ</given-names></name><name><surname>Robertson</surname><given-names>CE</given-names></name><name><surname>Frank</surname><given-names>DN</given-names></name><name><surname>Stabler</surname><given-names>SP</given-names></name><name><surname>Colgan</surname><given-names>SP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Oral vitamin B<sub>12</sub> supplement is delivered to the distal gut, altering the corrinoid profile and selectively depleting <italic>Bacteroides</italic> in C57BL/6 mice</article-title><source>Gut Microbes</source><volume>10</volume><fpage>654</fpage><lpage>662</lpage><pub-id pub-id-type="doi">10.1080/19490976.2019.1597667</pub-id><pub-id pub-id-type="pmid">31062653</pub-id></element-citation></ref><ref id="bib99"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname><given-names>I</given-names></name><name><surname>Ullah</surname><given-names>N</given-names></name><name><surname>Zha</surname><given-names>L</given-names></name><name><surname>Bai</surname><given-names>Y</given-names></name><name><surname>Khan</surname><given-names>A</given-names></name><name><surname>Zhao</surname><given-names>T</given-names></name><name><surname>Che</surname><given-names>T</given-names></name><name><surname>Zhang</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Alteration of Gut Microbiota in Inflammatory Bowel Disease (IBD): cause or consequence? IBD treatment targeting the gut microbiome</article-title><source>Pathogens</source><volume>8</volume><elocation-id>126</elocation-id><pub-id pub-id-type="doi">10.3390/pathogens8030126</pub-id><pub-id pub-id-type="pmid">31412603</pub-id></element-citation></ref><ref id="bib100"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khosravi</surname><given-names>A</given-names></name><name><surname>Mazmanian</surname><given-names>SK</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Disruption of the gut microbiome as a risk factor for microbial infections</article-title><source>Current Opinion in Microbiology</source><volume>16</volume><fpage>221</fpage><lpage>227</lpage><pub-id pub-id-type="doi">10.1016/j.mib.2013.03.009</pub-id><pub-id pub-id-type="pmid">23597788</pub-id></element-citation></ref><ref id="bib101"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kilstrup</surname><given-names>M</given-names></name><name><surname>Hammer</surname><given-names>K</given-names></name><name><surname>Ruhdal Jensen</surname><given-names>P</given-names></name><name><surname>Martinussen</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Nucleotide metabolism and its control in lactic acid bacteria</article-title><source>FEMS Microbiology Reviews</source><volume>29</volume><fpage>555</fpage><lpage>590</lpage><pub-id pub-id-type="doi">10.1016/j.femsre.2005.04.006</pub-id><pub-id pub-id-type="pmid">15935511</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>CJ</given-names></name><name><surname>Kovacs-Nolan</surname><given-names>JA</given-names></name><name><surname>Yang</surname><given-names>C</given-names></name><name><surname>Archbold</surname><given-names>T</given-names></name><name><surname>Fan</surname><given-names>MZ</given-names></name><name><surname>Mine</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>L-Tryptophan exhibits therapeutic function in a porcine model of dextran sodium sulfate (DSS)-induced colitis</article-title><source>The Journal of Nutritional Biochemistry</source><volume>21</volume><fpage>468</fpage><lpage>475</lpage><pub-id pub-id-type="doi">10.1016/j.jnutbio.2009.01.019</pub-id><pub-id pub-id-type="pmid">19428234</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Callewaert</surname><given-names>C</given-names></name><name><surname>Marotz</surname><given-names>C</given-names></name><name><surname>Hyde</surname><given-names>ER</given-names></name><name><surname>Debelius</surname><given-names>JW</given-names></name><name><surname>McDonald</surname><given-names>D</given-names></name><name><surname>Sogin</surname><given-names>ML</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The microbiome and human biology</article-title><source>Annual Review of Genomics and Human Genetics</source><volume>18</volume><fpage>65</fpage><lpage>86</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-083115-022438</pub-id><pub-id pub-id-type="pmid">28375652</pub-id></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Knox</surname><given-names>NC</given-names></name><name><surname>Forbes</surname><given-names>JD</given-names></name><name><surname>Van Domselaar</surname><given-names>G</given-names></name><name><surname>Bernstein</surname><given-names>CN</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The gut microbiome as a target for IBD treatment: are we there yet?</article-title><source>Current Treatment Options in Gastroenterology</source><volume>17</volume><fpage>115</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.1007/s11938-019-00221-w</pub-id><pub-id pub-id-type="pmid">30661163</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Koek</surname><given-names>MM</given-names></name><name><surname>Jellema</surname><given-names>RH</given-names></name><name><surname>van der Greef</surname><given-names>J</given-names></name><name><surname>Tas</surname><given-names>AC</given-names></name><name><surname>Hankemeier</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Quantitative metabolomics based on gas chromatography mass spectrometry: status and perspectives</article-title><source>Metabolomics</source><volume>7</volume><fpage>307</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1007/s11306-010-0254-3</pub-id><pub-id pub-id-type="pmid">21949491</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kolios</surname><given-names>G</given-names></name><name><surname>Valatas</surname><given-names>V</given-names></name><name><surname>Ward</surname><given-names>SG</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Nitric oxide in inflammatory bowel disease: A universal messenger in an unsolved puzzle</article-title><source>Immunology</source><volume>113</volume><fpage>427</fpage><lpage>437</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2567.2004.01984.x</pub-id><pub-id pub-id-type="pmid">15554920</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Köster</surname><given-names>J</given-names></name><name><surname>Rahmann</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Snakemake--a scalable bioinformatics workflow engine</article-title><source>Bioinformatics</source><volume>28</volume><fpage>2520</fpage><lpage>2522</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts480</pub-id><pub-id pub-id-type="pmid">22908215</pub-id></element-citation></ref><ref id="bib108"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kostic</surname><given-names>AD</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Gevers</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The microbiome in inflammatory bowel disease: current status and the future ahead</article-title><source>Gastroenterology</source><volume>146</volume><fpage>1489</fpage><lpage>1499</lpage><pub-id pub-id-type="doi">10.1053/j.gastro.2014.02.009</pub-id><pub-id pub-id-type="pmid">24560869</pub-id></element-citation></ref><ref id="bib109"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kraus</surname><given-names>S</given-names></name><name><surname>Arber</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Inflammation and colorectal cancer</article-title><source>Current Opinion in Pharmacology</source><volume>9</volume><fpage>405</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1016/j.coph.2009.06.006</pub-id><pub-id pub-id-type="pmid">19589728</pub-id></element-citation></ref><ref id="bib110"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kronman</surname><given-names>MP</given-names></name><name><surname>Zaoutis</surname><given-names>TE</given-names></name><name><surname>Haynes</surname><given-names>K</given-names></name><name><surname>Feng</surname><given-names>R</given-names></name><name><surname>Coffin</surname><given-names>SE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Antibiotic exposure and IBD development among children: A population-based cohort study</article-title><source>Pediatrics</source><volume>130</volume><fpage>e794</fpage><lpage>e803</lpage><pub-id pub-id-type="doi">10.1542/peds.2011-3886</pub-id><pub-id pub-id-type="pmid">23008454</pub-id></element-citation></ref><ref id="bib111"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kruger</surname><given-names>NJ</given-names></name><name><surname>von Schaewen</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>The oxidative pentose phosphate pathway: structure and organisation</article-title><source>Current Opinion in Plant Biology</source><volume>6</volume><fpage>236</fpage><lpage>246</lpage><pub-id pub-id-type="doi">10.1016/s1369-5266(03)00039-6</pub-id><pub-id pub-id-type="pmid">12753973</pub-id></element-citation></ref><ref id="bib112"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Le Chatelier</surname><given-names>E</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Prifti</surname><given-names>E</given-names></name><name><surname>Hildebrand</surname><given-names>F</given-names></name><name><surname>Falony</surname><given-names>G</given-names></name><name><surname>Almeida</surname><given-names>M</given-names></name><name><surname>Arumugam</surname><given-names>M</given-names></name><name><surname>Batto</surname><given-names>J-M</given-names></name><name><surname>Kennedy</surname><given-names>S</given-names></name><name><surname>Leonard</surname><given-names>P</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Burgdorf</surname><given-names>K</given-names></name><name><surname>Grarup</surname><given-names>N</given-names></name><name><surname>Jørgensen</surname><given-names>T</given-names></name><name><surname>Brandslund</surname><given-names>I</given-names></name><name><surname>Nielsen</surname><given-names>HB</given-names></name><name><surname>Juncker</surname><given-names>AS</given-names></name><name><surname>Bertalan</surname><given-names>M</given-names></name><name><surname>Levenez</surname><given-names>F</given-names></name><name><surname>Pons</surname><given-names>N</given-names></name><name><surname>Rasmussen</surname><given-names>S</given-names></name><name><surname>Sunagawa</surname><given-names>S</given-names></name><name><surname>Tap</surname><given-names>J</given-names></name><name><surname>Tims</surname><given-names>S</given-names></name><name><surname>Zoetendal</surname><given-names>EG</given-names></name><name><surname>Brunak</surname><given-names>S</given-names></name><name><surname>Clément</surname><given-names>K</given-names></name><name><surname>Doré</surname><given-names>J</given-names></name><name><surname>Kleerebezem</surname><given-names>M</given-names></name><name><surname>Kristiansen</surname><given-names>K</given-names></name><name><surname>Renault</surname><given-names>P</given-names></name><name><surname>Sicheritz-Ponten</surname><given-names>T</given-names></name><name><surname>de Vos</surname><given-names>WM</given-names></name><name><surname>Zucker</surname><given-names>J-D</given-names></name><name><surname>Raes</surname><given-names>J</given-names></name><name><surname>Hansen</surname><given-names>T</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Ehrlich</surname><given-names>SD</given-names></name><name><surname>Pedersen</surname><given-names>O</given-names></name><collab>MetaHIT consortium</collab></person-group><year iso-8601-date="2013">2013</year><article-title>Richness of human gut microbiome correlates with metabolic markers</article-title><source>Nature</source><volume>500</volume><fpage>541</fpage><lpage>546</lpage><pub-id pub-id-type="doi">10.1038/nature12506</pub-id><pub-id pub-id-type="pmid">23985870</pub-id></element-citation></ref><ref id="bib113"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ledder</surname><given-names>Oren</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Antibiotics in inflammatory bowel diseases: do we know what we’re doing?</article-title><source>Translational Pediatrics</source><volume>8</volume><fpage>42</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.21037/tp.2018.11.02</pub-id><pub-id pub-id-type="pmid">30881898</pub-id></element-citation></ref><ref id="bib114"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>H-S</given-names></name><name><surname>Han</surname><given-names>S-Y</given-names></name><name><surname>Ryu</surname><given-names>K-Y</given-names></name><name><surname>Kim</surname><given-names>D-H</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The degradation of glycosaminoglycans by intestinal microflora deteriorates colitis in mice</article-title><source>Inflammation</source><volume>32</volume><fpage>27</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1007/s10753-008-9099-6</pub-id><pub-id pub-id-type="pmid">19067146</pub-id></element-citation></ref><ref id="bib115"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>M</given-names></name><name><surname>Chang</surname><given-names>EB</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Inflammatory Bowel Diseases (IBD) and the microbiome-searching the crime scene for clues</article-title><source>Gastroenterology</source><volume>160</volume><fpage>524</fpage><lpage>537</lpage><pub-id pub-id-type="doi">10.1053/j.gastro.2020.09.056</pub-id><pub-id pub-id-type="pmid">33253681</pub-id></element-citation></ref><ref id="bib116"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leonardi</surname><given-names>R</given-names></name><name><surname>Zhang</surname><given-names>YM</given-names></name><name><surname>Rock</surname><given-names>CO</given-names></name><name><surname>Jackowski</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Coenzyme A: back in action</article-title><source>Progress in Lipid Research</source><volume>44</volume><fpage>125</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1016/j.plipres.2005.04.001</pub-id><pub-id pub-id-type="pmid">15893380</pub-id></element-citation></ref><ref id="bib117"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leung</surname><given-names>W</given-names></name><name><surname>Malhi</surname><given-names>G</given-names></name><name><surname>Willey</surname><given-names>BM</given-names></name><name><surname>McGeer</surname><given-names>AJ</given-names></name><name><surname>Borgundvaag</surname><given-names>B</given-names></name><name><surname>Thanabalan</surname><given-names>R</given-names></name><name><surname>Gnanasuntharam</surname><given-names>P</given-names></name><name><surname>Le</surname><given-names>B</given-names></name><name><surname>Weizman</surname><given-names>AV</given-names></name><name><surname>Croitoru</surname><given-names>K</given-names></name><name><surname>Silverberg</surname><given-names>MS</given-names></name><name><surname>Steinhart</surname><given-names>AH</given-names></name><name><surname>Nguyen</surname><given-names>GC</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Prevalence and predictors of MRSA, ESBL, and VRE colonization in the ambulatory IBD population</article-title><source>Journal of Crohn’s &amp; Colitis</source><volume>6</volume><fpage>743</fpage><lpage>749</lpage><pub-id pub-id-type="doi">10.1016/j.crohns.2011.12.005</pub-id><pub-id pub-id-type="pmid">22398097</pub-id></element-citation></ref><ref id="bib118"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Levy</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The future of antibiotics: facing antibiotic resistance</article-title><source>Clinical Microbiology and Infection: The Official Publication of the European Society of Clinical Microbiology and Infectious Diseases</source><volume>6 Suppl 3</volume><fpage>101</fpage><lpage>106</lpage><pub-id pub-id-type="pmid">11449641</pub-id></element-citation></ref><ref id="bib119"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>D</given-names></name><name><surname>Liu</surname><given-names>C-M</given-names></name><name><surname>Luo</surname><given-names>R</given-names></name><name><surname>Sadakane</surname><given-names>K</given-names></name><name><surname>Lam</surname><given-names>T-W</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>MEGAHIT: an ultra-fast single-node solution for large and complex metagenomics assembly via succinct    <italic>de Bruijn</italic>    graph</article-title><source>Bioinformatics</source><volume>31</volume><fpage>1674</fpage><lpage>1676</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv033</pub-id></element-citation></ref><ref id="bib120"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>R</given-names></name><name><surname>Liu</surname><given-names>W</given-names></name><name><surname>Piao</surname><given-names>M</given-names></name><name><surname>Zhu</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A review of the relationship between the gut microbiota and amino acid metabolism</article-title><source>Amino Acids</source><volume>49</volume><fpage>2083</fpage><lpage>2090</lpage><pub-id pub-id-type="doi">10.1007/s00726-017-2493-3</pub-id></element-citation></ref><ref id="bib121"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>Y</given-names></name><name><surname>Caldwell</surname><given-names>GW</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Lang</surname><given-names>W</given-names></name><name><surname>Masucci</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Inter-laboratory reproducibility of an untargeted metabolomics GC-MS assay for analysis of human plasma</article-title><source>Scientific Reports</source><volume>10</volume><elocation-id>10918</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-020-67939-x</pub-id><pub-id pub-id-type="pmid">32616798</pub-id></element-citation></ref><ref id="bib122"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Guo</surname><given-names>X</given-names></name><name><surname>Rao</surname><given-names>JN</given-names></name><name><surname>Zou</surname><given-names>T</given-names></name><name><surname>Xiao</surname><given-names>L</given-names></name><name><surname>Yu</surname><given-names>T</given-names></name><name><surname>Timmons</surname><given-names>JA</given-names></name><name><surname>Turner</surname><given-names>DJ</given-names></name><name><surname>Wang</surname><given-names>JY</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Polyamines regulate E-cadherin transcription through c-Myc modulating intestinal epithelial barrier function</article-title><source>American Journal of Physiology. Cell Physiology</source><volume>296</volume><fpage>C801</fpage><lpage>C10</lpage><pub-id pub-id-type="doi">10.1152/ajpcell.00620.2008</pub-id><pub-id pub-id-type="pmid">19176757</pub-id></element-citation></ref><ref id="bib123"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Breukink</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The Mmembrane steps of bacterial cell wall synthesis as antibiotic tts</article-title><source>Antibiotics</source><volume>5</volume><elocation-id>28</elocation-id><pub-id pub-id-type="doi">10.3390/antibiotics5030028</pub-id><pub-id pub-id-type="pmid">27571111</pub-id></element-citation></ref><ref id="bib124"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Hu</surname><given-names>CAA</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Therapeutic potential of amino acids in inflammatory bowel disease</article-title><source>Nutrients</source><volume>9</volume><elocation-id>920</elocation-id><pub-id pub-id-type="doi">10.3390/nu9090920</pub-id><pub-id pub-id-type="pmid">28832517</pub-id></element-citation></ref><ref id="bib125"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Llor</surname><given-names>C</given-names></name><name><surname>Bjerrum</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Antimicrobial resistance: risk associated with antibiotic overuse and initiatives to reduce the problem</article-title><source>Therapeutic Advances in Drug Safety</source><volume>5</volume><fpage>229</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1177/2042098614554919</pub-id><pub-id pub-id-type="pmid">25436105</pub-id></element-citation></ref><ref id="bib126"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lloyd-Price</surname><given-names>J</given-names></name><name><surname>Arze</surname><given-names>C</given-names></name><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name><name><surname>Schirmer</surname><given-names>M</given-names></name><name><surname>Avila-Pacheco</surname><given-names>J</given-names></name><name><surname>Poon</surname><given-names>TW</given-names></name><name><surname>Andrews</surname><given-names>E</given-names></name><name><surname>Ajami</surname><given-names>NJ</given-names></name><name><surname>Bonham</surname><given-names>KS</given-names></name><name><surname>Brislawn</surname><given-names>CJ</given-names></name><name><surname>Casero</surname><given-names>D</given-names></name><name><surname>Courtney</surname><given-names>H</given-names></name><name><surname>Gonzalez</surname><given-names>A</given-names></name><name><surname>Graeber</surname><given-names>TG</given-names></name><name><surname>Hall</surname><given-names>AB</given-names></name><name><surname>Lake</surname><given-names>K</given-names></name><name><surname>Landers</surname><given-names>CJ</given-names></name><name><surname>Mallick</surname><given-names>H</given-names></name><name><surname>Plichta</surname><given-names>DR</given-names></name><name><surname>Prasad</surname><given-names>M</given-names></name><name><surname>Rahnavard</surname><given-names>G</given-names></name><name><surname>Sauk</surname><given-names>J</given-names></name><name><surname>Shungin</surname><given-names>D</given-names></name><name><surname>Vázquez-Baeza</surname><given-names>Y</given-names></name><name><surname>White</surname><given-names>RA</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Denson</surname><given-names>LA</given-names></name><name><surname>Jansson</surname><given-names>JK</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Kugathasan</surname><given-names>S</given-names></name><name><surname>McGovern</surname><given-names>DPB</given-names></name><name><surname>Petrosino</surname><given-names>JF</given-names></name><name><surname>Stappenbeck</surname><given-names>TS</given-names></name><name><surname>Winter</surname><given-names>HS</given-names></name><name><surname>Clish</surname><given-names>CB</given-names></name><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name><collab>IBDMDB Investigators</collab></person-group><year iso-8601-date="2019">2019</year><article-title>Multi-omics of the gut microbial ecosystem in inflammatory bowel diseases</article-title><source>Nature</source><volume>569</volume><fpage>655</fpage><lpage>662</lpage><pub-id pub-id-type="doi">10.1038/s41586-019-1237-9</pub-id><pub-id pub-id-type="pmid">31142855</pub-id></element-citation></ref><ref id="bib127"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Lopez</surname><given-names>MJ</given-names></name><name><surname>Mohiuddin</surname><given-names>SS</given-names></name></person-group><year iso-8601-date="2023">2023</year><source>Biochemistry, Essential Amino Acids</source><publisher-name>StatPearls Publishing</publisher-name></element-citation></ref><ref id="bib128"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lopez-Siles</surname><given-names>M</given-names></name><name><surname>Khan</surname><given-names>TM</given-names></name><name><surname>Duncan</surname><given-names>SH</given-names></name><name><surname>Harmsen</surname><given-names>HJM</given-names></name><name><surname>Garcia-Gil</surname><given-names>LJ</given-names></name><name><surname>Flint</surname><given-names>HJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Cultured representatives of two major phylogroups of human colonic Faecalibacterium prausnitzii can utilize pectin, uronic acids, and host-derived substrates for growth</article-title><source>Applied and Environmental Microbiology</source><volume>78</volume><fpage>420</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1128/AEM.06858-11</pub-id><pub-id pub-id-type="pmid">22101049</pub-id></element-citation></ref><ref id="bib129"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lozupone</surname><given-names>CA</given-names></name><name><surname>Stombaugh</surname><given-names>J</given-names></name><name><surname>Gonzalez</surname><given-names>A</given-names></name><name><surname>Ackermann</surname><given-names>G</given-names></name><name><surname>Wendel</surname><given-names>D</given-names></name><name><surname>Vázquez-Baeza</surname><given-names>Y</given-names></name><name><surname>Jansson</surname><given-names>JK</given-names></name><name><surname>Gordon</surname><given-names>JI</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Meta-analyses of studies of the human microbiota</article-title><source>Genome Research</source><volume>23</volume><fpage>1704</fpage><lpage>1714</lpage><pub-id pub-id-type="doi">10.1101/gr.151803.112</pub-id><pub-id pub-id-type="pmid">23861384</pub-id></element-citation></ref><ref id="bib130"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Machado</surname><given-names>D</given-names></name><name><surname>Andrejev</surname><given-names>S</given-names></name><name><surname>Tramontano</surname><given-names>M</given-names></name><name><surname>Patil</surname><given-names>KR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Fast automated reconstruction of genome-scale metabolic models for microbial species and communities</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>7542</fpage><lpage>7553</lpage><pub-id pub-id-type="doi">10.1093/nar/gky537</pub-id><pub-id pub-id-type="pmid">30192979</pub-id></element-citation></ref><ref id="bib131"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Machiels</surname><given-names>K</given-names></name><name><surname>Joossens</surname><given-names>M</given-names></name><name><surname>Sabino</surname><given-names>J</given-names></name><name><surname>De Preter</surname><given-names>V</given-names></name><name><surname>Arijs</surname><given-names>I</given-names></name><name><surname>Eeckhaut</surname><given-names>V</given-names></name><name><surname>Ballet</surname><given-names>V</given-names></name><name><surname>Claes</surname><given-names>K</given-names></name><name><surname>Van Immerseel</surname><given-names>F</given-names></name><name><surname>Verbeke</surname><given-names>K</given-names></name><name><surname>Ferrante</surname><given-names>M</given-names></name><name><surname>Verhaegen</surname><given-names>J</given-names></name><name><surname>Rutgeerts</surname><given-names>P</given-names></name><name><surname>Vermeire</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A decrease of the butyrate-producing species Roseburia hominis and Faecalibacterium prausnitzii defines dysbiosis in patients with ulcerative colitis</article-title><source>Gut</source><volume>63</volume><fpage>1275</fpage><lpage>1283</lpage><pub-id pub-id-type="doi">10.1136/gutjnl-2013-304833</pub-id><pub-id pub-id-type="pmid">24021287</pub-id></element-citation></ref><ref id="bib132"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Magnúsdóttir</surname><given-names>S</given-names></name><name><surname>Ravcheev</surname><given-names>D</given-names></name><name><surname>de Crécy-Lagard</surname><given-names>V</given-names></name><name><surname>Thiele</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Systematic genome assessment of B-vitamin biosynthesis suggests co-operation among gut microbes</article-title><source>Frontiers in Genetics</source><volume>6</volume><elocation-id>148</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2015.00148</pub-id><pub-id pub-id-type="pmid">25941533</pub-id></element-citation></ref><ref id="bib133"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Marcelino</surname><given-names>VR</given-names></name><name><surname>Welsh</surname><given-names>C</given-names></name><name><surname>Diener</surname><given-names>C</given-names></name><name><surname>Gulliver</surname><given-names>EL</given-names></name><name><surname>Rutten</surname><given-names>EL</given-names></name><name><surname>Young</surname><given-names>RB</given-names></name><name><surname>Giles</surname><given-names>EM</given-names></name><name><surname>Gibbons</surname><given-names>SM</given-names></name><name><surname>Greening</surname><given-names>C</given-names></name><name><surname>Forster</surname><given-names>SC</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Disease-specific loss of microbial cross-feeding interactions in the human gut</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2023.02.17.528570</pub-id></element-citation></ref><ref id="bib134"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martens</surname><given-names>EC</given-names></name><name><surname>Kelly</surname><given-names>AG</given-names></name><name><surname>Tauzin</surname><given-names>AS</given-names></name><name><surname>Brumer</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The devil lies in the details: how variations in polysaccharide fine-structure impact the physiology and evolution of gut microbes</article-title><source>Journal of Molecular Biology</source><volume>426</volume><fpage>3851</fpage><lpage>3865</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2014.06.022</pub-id><pub-id pub-id-type="pmid">25026064</pub-id></element-citation></ref><ref id="bib135"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martín</surname><given-names>R</given-names></name><name><surname>Rios-Covian</surname><given-names>D</given-names></name><name><surname>Huillet</surname><given-names>E</given-names></name><name><surname>Auger</surname><given-names>S</given-names></name><name><surname>Khazaal</surname><given-names>S</given-names></name><name><surname>Bermúdez-Humarán</surname><given-names>LG</given-names></name><name><surname>Sokol</surname><given-names>H</given-names></name><name><surname>Chatel</surname><given-names>JM</given-names></name><name><surname>Langella</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Faecalibacterium: A bacterial genus with promising human health applications</article-title><source>FEMS Microbiology Reviews</source><volume>47</volume><elocation-id>fuad039</elocation-id><pub-id pub-id-type="doi">10.1093/femsre/fuad039</pub-id><pub-id pub-id-type="pmid">37451743</pub-id></element-citation></ref><ref id="bib136"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin-Gallausiaux</surname><given-names>C</given-names></name><name><surname>Marinelli</surname><given-names>L</given-names></name><name><surname>Blottière</surname><given-names>HM</given-names></name><name><surname>Larraufie</surname><given-names>P</given-names></name><name><surname>Lapaque</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>SCFA: mechanisms and functional importance in the gut</article-title><source>The Proceedings of the Nutrition Society</source><volume>80</volume><fpage>37</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1017/S0029665120006916</pub-id><pub-id pub-id-type="pmid">32238208</pub-id></element-citation></ref><ref id="bib137"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maynard</surname><given-names>CL</given-names></name><name><surname>Elson</surname><given-names>CO</given-names></name><name><surname>Hatton</surname><given-names>RD</given-names></name><name><surname>Weaver</surname><given-names>CT</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Reciprocal interactions of the intestinal microbiota and immune system</article-title><source>Nature</source><volume>489</volume><fpage>231</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1038/nature11551</pub-id><pub-id pub-id-type="pmid">22972296</pub-id></element-citation></ref><ref id="bib138"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McCormack</surname><given-names>SA</given-names></name><name><surname>Johnson</surname><given-names>LR</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Role of polyamines in gastrointestinal mucosal growth</article-title><source>The American Journal of Physiology</source><volume>260</volume><fpage>G795</fpage><lpage>G806</lpage><pub-id pub-id-type="doi">10.1152/ajpgi.1991.260.6.G795</pub-id><pub-id pub-id-type="pmid">2058669</pub-id></element-citation></ref><ref id="bib139"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mendoza</surname><given-names>SN</given-names></name><name><surname>Olivier</surname><given-names>BG</given-names></name><name><surname>Molenaar</surname><given-names>D</given-names></name><name><surname>Teusink</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A systematic assessment of current genome-scale metabolic reconstruction tools</article-title><source>Genome Biology</source><volume>20</volume><elocation-id>158</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-019-1769-1</pub-id><pub-id pub-id-type="pmid">31391098</pub-id></element-citation></ref><ref id="bib140"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mesnage</surname><given-names>R</given-names></name><name><surname>Antoniou</surname><given-names>MN</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Computational modelling provides insight into the effects of glyphosate on the shikimate pathway in the human gut microbiome</article-title><source>Current Research in Toxicology</source><volume>1</volume><fpage>25</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1016/j.crtox.2020.04.001</pub-id><pub-id pub-id-type="pmid">34345834</pub-id></element-citation></ref><ref id="bib141"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Metges</surname><given-names>CC</given-names></name><name><surname>El-Khoury</surname><given-names>AE</given-names></name><name><surname>Henneman</surname><given-names>L</given-names></name><name><surname>Petzke</surname><given-names>KJ</given-names></name><name><surname>Grant</surname><given-names>I</given-names></name><name><surname>Bedri</surname><given-names>S</given-names></name><name><surname>Pereira</surname><given-names>PP</given-names></name><name><surname>Ajami</surname><given-names>AM</given-names></name><name><surname>Fuller</surname><given-names>MF</given-names></name><name><surname>Young</surname><given-names>VR</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Availability of intestinal microbial lysine for whole body lysine homeostasis in human subjects</article-title><source>The American Journal of Physiology</source><volume>277</volume><fpage>E597</fpage><lpage>E607</lpage><pub-id pub-id-type="doi">10.1152/ajpendo.1999.277.4.E597</pub-id><pub-id pub-id-type="pmid">10516118</pub-id></element-citation></ref><ref id="bib142"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mikkola</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Nucleotide sugars in chemistry and biology</article-title><source>Molecules</source><volume>25</volume><elocation-id>23</elocation-id><pub-id pub-id-type="doi">10.3390/molecules25235755</pub-id><pub-id pub-id-type="pmid">33291296</pub-id></element-citation></ref><ref id="bib143"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minárik</surname><given-names>P</given-names></name><name><surname>Tomásková</surname><given-names>N</given-names></name><name><surname>Kollárová</surname><given-names>M</given-names></name><name><surname>Antalík</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Malate dehydrogenases--structure and function</article-title><source>General Physiology and Biophysics</source><volume>21</volume><fpage>257</fpage><lpage>265</lpage><pub-id pub-id-type="pmid">12537350</pub-id></element-citation></ref><ref id="bib144"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minh</surname><given-names>BQ</given-names></name><name><surname>Schmidt</surname><given-names>HA</given-names></name><name><surname>Chernomor</surname><given-names>O</given-names></name><name><surname>Schrempf</surname><given-names>D</given-names></name><name><surname>Woodhams</surname><given-names>MD</given-names></name><name><surname>von Haeseler</surname><given-names>A</given-names></name><name><surname>Lanfear</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era</article-title><source>Molecular Biology and Evolution</source><volume>37</volume><fpage>1530</fpage><lpage>1534</lpage><pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id><pub-id pub-id-type="pmid">32011700</pub-id></element-citation></ref><ref id="bib145"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mistry</surname><given-names>J</given-names></name><name><surname>Chuguransky</surname><given-names>S</given-names></name><name><surname>Williams</surname><given-names>L</given-names></name><name><surname>Qureshi</surname><given-names>M</given-names></name><name><surname>Salazar</surname><given-names>GA</given-names></name><name><surname>Sonnhammer</surname><given-names>ELL</given-names></name><name><surname>Tosatto</surname><given-names>SCE</given-names></name><name><surname>Paladin</surname><given-names>L</given-names></name><name><surname>Raj</surname><given-names>S</given-names></name><name><surname>Richardson</surname><given-names>LJ</given-names></name><name><surname>Finn</surname><given-names>RD</given-names></name><name><surname>Bateman</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Pfam: The protein families database in 2021</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>D412</fpage><lpage>D419</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa913</pub-id><pub-id pub-id-type="pmid">33125078</pub-id></element-citation></ref><ref id="bib146"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morgan</surname><given-names>XC</given-names></name><name><surname>Tickle</surname><given-names>TL</given-names></name><name><surname>Sokol</surname><given-names>H</given-names></name><name><surname>Gevers</surname><given-names>D</given-names></name><name><surname>Devaney</surname><given-names>KL</given-names></name><name><surname>Ward</surname><given-names>DV</given-names></name><name><surname>Reyes</surname><given-names>JA</given-names></name><name><surname>Shah</surname><given-names>SA</given-names></name><name><surname>LeLeiko</surname><given-names>N</given-names></name><name><surname>Snapper</surname><given-names>SB</given-names></name><name><surname>Bousvaros</surname><given-names>A</given-names></name><name><surname>Korzenik</surname><given-names>J</given-names></name><name><surname>Sands</surname><given-names>BE</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Dysfunction of the intestinal microbiome in inflammatory bowel disease and treatment</article-title><source>Genome Biology</source><volume>13</volume><elocation-id>R79</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2012-13-9-r79</pub-id><pub-id pub-id-type="pmid">23013615</pub-id></element-citation></ref><ref id="bib147"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moriya</surname><given-names>Y</given-names></name><name><surname>Itoh</surname><given-names>M</given-names></name><name><surname>Okuda</surname><given-names>S</given-names></name><name><surname>Yoshizawa</surname><given-names>AC</given-names></name><name><surname>Kanehisa</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>KAAS: an automatic genome annotation and pathway reconstruction server</article-title><source>Nucleic Acids Research</source><volume>35</volume><fpage>W182</fpage><lpage>W5</lpage><pub-id pub-id-type="doi">10.1093/nar/gkm321</pub-id><pub-id pub-id-type="pmid">17526522</pub-id></element-citation></ref><ref id="bib148"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Musrati</surname><given-names>RA</given-names></name><name><surname>Kollárová</surname><given-names>M</given-names></name><name><surname>Mernik</surname><given-names>N</given-names></name><name><surname>Mikulásová</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Malate dehydrogenase: distribution, function and properties</article-title><source>General Physiology and Biophysics</source><volume>17</volume><fpage>193</fpage><lpage>210</lpage><pub-id pub-id-type="pmid">9834842</pub-id></element-citation></ref><ref id="bib149"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nagalingam</surname><given-names>NA</given-names></name><name><surname>Lynch</surname><given-names>SV</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Role of the microbiota in inflammatory bowel diseases</article-title><source>Inflammatory Bowel Diseases</source><volume>18</volume><fpage>968</fpage><lpage>984</lpage><pub-id pub-id-type="doi">10.1002/ibd.21866</pub-id><pub-id pub-id-type="pmid">21936031</pub-id></element-citation></ref><ref id="bib150"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nikolaus</surname><given-names>S</given-names></name><name><surname>Schulte</surname><given-names>B</given-names></name><name><surname>Al-Massad</surname><given-names>N</given-names></name><name><surname>Thieme</surname><given-names>F</given-names></name><name><surname>Schulte</surname><given-names>DM</given-names></name><name><surname>Bethge</surname><given-names>J</given-names></name><name><surname>Rehman</surname><given-names>A</given-names></name><name><surname>Tran</surname><given-names>F</given-names></name><name><surname>Aden</surname><given-names>K</given-names></name><name><surname>Häsler</surname><given-names>R</given-names></name><name><surname>Moll</surname><given-names>N</given-names></name><name><surname>Schütze</surname><given-names>G</given-names></name><name><surname>Schwarz</surname><given-names>MJ</given-names></name><name><surname>Waetzig</surname><given-names>GH</given-names></name><name><surname>Rosenstiel</surname><given-names>P</given-names></name><name><surname>Krawczak</surname><given-names>M</given-names></name><name><surname>Szymczak</surname><given-names>S</given-names></name><name><surname>Schreiber</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Increased tryptophan metabolism is associated with activity of inflammatory bowel diseases</article-title><source>Gastroenterology</source><volume>153</volume><fpage>1504</fpage><lpage>1516</lpage><pub-id pub-id-type="doi">10.1053/j.gastro.2017.08.028</pub-id><pub-id pub-id-type="pmid">28827067</pub-id></element-citation></ref><ref id="bib151"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nishida</surname><given-names>A</given-names></name><name><surname>Inoue</surname><given-names>R</given-names></name><name><surname>Inatomi</surname><given-names>O</given-names></name><name><surname>Bamba</surname><given-names>S</given-names></name><name><surname>Naito</surname><given-names>Y</given-names></name><name><surname>Andoh</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Gut microbiota in the pathogenesis of inflammatory bowel disease</article-title><source>Clinical Journal of Gastroenterology</source><volume>11</volume><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.1007/s12328-017-0813-5</pub-id><pub-id pub-id-type="pmid">29285689</pub-id></element-citation></ref><ref id="bib152"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nitzan</surname><given-names>O</given-names></name><name><surname>Elias</surname><given-names>M</given-names></name><name><surname>Peretz</surname><given-names>A</given-names></name><name><surname>Saliba</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Role of antibiotics for treatment of inflammatory bowel disease</article-title><source>World Journal of Gastroenterology</source><volume>22</volume><fpage>1078</fpage><lpage>1087</lpage><pub-id pub-id-type="doi">10.3748/wjg.v22.i3.1078</pub-id><pub-id pub-id-type="pmid">26811648</pub-id></element-citation></ref><ref id="bib153"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Nygaard</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2014">2014</year><chapter-title>Purine and pyrimidine salvage pathways</chapter-title><person-group person-group-type="editor"><name><surname>Nygaard</surname><given-names>P</given-names></name></person-group> <source>In Bacillus subtilis and Other Gram-Positive Bacteria</source><publisher-name>ASM Press</publisher-name><fpage>359</fpage><lpage>378</lpage><pub-id pub-id-type="doi">10.1128/9781555818388.ch26</pub-id></element-citation></ref><ref id="bib154"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Orth</surname><given-names>JD</given-names></name><name><surname>Thiele</surname><given-names>I</given-names></name><name><surname>Palsson</surname><given-names>BØ</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>What is flux balance analysis?</article-title><source>Nature Biotechnology</source><volume>28</volume><fpage>245</fpage><lpage>248</lpage><pub-id pub-id-type="doi">10.1038/nbt.1614</pub-id><pub-id pub-id-type="pmid">20212490</pub-id></element-citation></ref><ref id="bib155"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oz</surname><given-names>HS</given-names></name><name><surname>Chen</surname><given-names>TS</given-names></name><name><surname>McClain</surname><given-names>CJ</given-names></name><name><surname>de Villiers</surname><given-names>WJS</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Antioxidants as novel therapy in a murine model of colitis</article-title><source>The Journal of Nutritional Biochemistry</source><volume>16</volume><fpage>297</fpage><lpage>304</lpage><pub-id pub-id-type="doi">10.1016/j.jnutbio.2004.09.007</pub-id><pub-id pub-id-type="pmid">15866230</pub-id></element-citation></ref><ref id="bib156"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pacheco</surname><given-names>AR</given-names></name><name><surname>Moel</surname><given-names>M</given-names></name><name><surname>Segrè</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Costless metabolic secretions as drivers of interspecies interactions in microbial ecosystems</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>103</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-07946-9</pub-id><pub-id pub-id-type="pmid">30626871</pub-id></element-citation></ref><ref id="bib157"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palleja</surname><given-names>A</given-names></name><name><surname>Mikkelsen</surname><given-names>KH</given-names></name><name><surname>Forslund</surname><given-names>SK</given-names></name><name><surname>Kashani</surname><given-names>A</given-names></name><name><surname>Allin</surname><given-names>KH</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>Hansen</surname><given-names>TH</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Zhang</surname><given-names>C</given-names></name><name><surname>Pyl</surname><given-names>PT</given-names></name><name><surname>Coelho</surname><given-names>LP</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Typas</surname><given-names>A</given-names></name><name><surname>Nielsen</surname><given-names>MF</given-names></name><name><surname>Nielsen</surname><given-names>HB</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Vilsbøll</surname><given-names>T</given-names></name><name><surname>Hansen</surname><given-names>T</given-names></name><name><surname>Knop</surname><given-names>FK</given-names></name><name><surname>Arumugam</surname><given-names>M</given-names></name><name><surname>Pedersen</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Recovery of gut microbiota of healthy adults following antibiotic exposure</article-title><source>Nature Microbiology</source><volume>3</volume><fpage>1255</fpage><lpage>1265</lpage><pub-id pub-id-type="doi">10.1038/s41564-018-0257-9</pub-id><pub-id pub-id-type="pmid">30349083</pub-id></element-citation></ref><ref id="bib158"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palù</surname><given-names>M</given-names></name><name><surname>Basile</surname><given-names>A</given-names></name><name><surname>Zampieri</surname><given-names>G</given-names></name><name><surname>Treu</surname><given-names>L</given-names></name><name><surname>Rossi</surname><given-names>A</given-names></name><name><surname>Morlino</surname><given-names>MS</given-names></name><name><surname>Campanaro</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>KEMET - A python tool for KEGG Module evaluation and microbial genome annotation expansion</article-title><source>Computational and Structural Biotechnology Journal</source><volume>20</volume><fpage>1481</fpage><lpage>1486</lpage><pub-id pub-id-type="doi">10.1016/j.csbj.2022.03.015</pub-id><pub-id pub-id-type="pmid">35422973</pub-id></element-citation></ref><ref id="bib159"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Papa</surname><given-names>E</given-names></name><name><surname>Docktor</surname><given-names>M</given-names></name><name><surname>Smillie</surname><given-names>C</given-names></name><name><surname>Weber</surname><given-names>S</given-names></name><name><surname>Preheim</surname><given-names>SP</given-names></name><name><surname>Gevers</surname><given-names>D</given-names></name><name><surname>Giannoukos</surname><given-names>G</given-names></name><name><surname>Ciulla</surname><given-names>D</given-names></name><name><surname>Tabbaa</surname><given-names>D</given-names></name><name><surname>Ingram</surname><given-names>J</given-names></name><name><surname>Schauer</surname><given-names>DB</given-names></name><name><surname>Ward</surname><given-names>DV</given-names></name><name><surname>Korzenik</surname><given-names>JR</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Bousvaros</surname><given-names>A</given-names></name><name><surname>Alm</surname><given-names>EJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Non-invasive mapping of the gastrointestinal microbiota identifies children with inflammatory bowel disease</article-title><source>PLOS ONE</source><volume>7</volume><elocation-id>e39242</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0039242</pub-id><pub-id pub-id-type="pmid">22768065</pub-id></element-citation></ref><ref id="bib160"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname><given-names>DH</given-names></name><name><surname>Chuvochina</surname><given-names>M</given-names></name><name><surname>Waite</surname><given-names>DW</given-names></name><name><surname>Rinke</surname><given-names>C</given-names></name><name><surname>Skarshewski</surname><given-names>A</given-names></name><name><surname>Chaumeil</surname><given-names>P-A</given-names></name><name><surname>Hugenholtz</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A standardized bacterial taxonomy based on genome phylogeny substantially revises the tree of life</article-title><source>Nature Biotechnology</source><volume>36</volume><fpage>996</fpage><lpage>1004</lpage><pub-id pub-id-type="doi">10.1038/nbt.4229</pub-id><pub-id pub-id-type="pmid">30148503</pub-id></element-citation></ref><ref id="bib161"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname><given-names>DH</given-names></name><name><surname>Chuvochina</surname><given-names>M</given-names></name><name><surname>Chaumeil</surname><given-names>PA</given-names></name><name><surname>Rinke</surname><given-names>C</given-names></name><name><surname>Mussig</surname><given-names>AJ</given-names></name><name><surname>Hugenholtz</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A complete domain-to-species taxonomy for Bacteria and Archaea</article-title><source>Nature Biotechnology</source><volume>38</volume><fpage>1079</fpage><lpage>1086</lpage><pub-id pub-id-type="doi">10.1038/s41587-020-0501-8</pub-id><pub-id pub-id-type="pmid">32341564</pub-id></element-citation></ref><ref id="bib162"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname><given-names>DH</given-names></name><name><surname>Chuvochina</surname><given-names>M</given-names></name><name><surname>Rinke</surname><given-names>C</given-names></name><name><surname>Mussig</surname><given-names>AJ</given-names></name><name><surname>Chaumeil</surname><given-names>P-A</given-names></name><name><surname>Hugenholtz</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>GTDB: an ongoing census of bacterial and archaeal diversity through a phylogenetically consistent, rank normalized and complete genome-based taxonomy</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D785</fpage><lpage>D794</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab776</pub-id><pub-id pub-id-type="pmid">34520557</pub-id></element-citation></ref><ref id="bib163"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname><given-names>Y</given-names></name><name><surname>Leung</surname><given-names>HCM</given-names></name><name><surname>Yiu</surname><given-names>SM</given-names></name><name><surname>Chin</surname><given-names>FYL</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>IDBA-UD: A de novo assembler for single-cell and metagenomic sequencing data with highly uneven depth</article-title><source>Bioinformatics</source><volume>28</volume><fpage>1420</fpage><lpage>1428</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts174</pub-id><pub-id pub-id-type="pmid">22495754</pub-id></element-citation></ref><ref id="bib164"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pierzynowski</surname><given-names>S</given-names></name><name><surname>Pierzynowska</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Alpha-ketoglutarate, a key molecule involved in nitrogen circulation in both animals and plants</article-title><source>Adv Med Sci</source><volume>67</volume><fpage>142</fpage><lpage>147</lpage><pub-id pub-id-type="doi">10.1016/j.advms.2022.02.004</pub-id></element-citation></ref><ref id="bib165"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Porrini</surname><given-names>C</given-names></name><name><surname>Guérin</surname><given-names>C</given-names></name><name><surname>Tran</surname><given-names>SL</given-names></name><name><surname>Dervyn</surname><given-names>R</given-names></name><name><surname>Nicolas</surname><given-names>P</given-names></name><name><surname>Ramarao</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Implication of a key region of six <italic>Bacillus cereus</italic> genes involved in siroheme synthesis, nitrite reductase production and iron cluster repair in the bacterial response to nitric oxide stress</article-title><source>International Journal of Molecular Sciences</source><volume>22</volume><elocation-id>5079</elocation-id><pub-id pub-id-type="doi">10.3390/ijms22105079</pub-id><pub-id pub-id-type="pmid">34064887</pub-id></element-citation></ref><ref id="bib166"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Powell</surname><given-names>JE</given-names></name><name><surname>Leonard</surname><given-names>SP</given-names></name><name><surname>Kwong</surname><given-names>WK</given-names></name><name><surname>Engel</surname><given-names>P</given-names></name><name><surname>Moran</surname><given-names>NA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genome-wide screen identifies host colonization determinants in a bacterial gut symbiont</article-title><source>PNAS</source><volume>113</volume><fpage>13887</fpage><lpage>13892</lpage><pub-id pub-id-type="doi">10.1073/pnas.1610856113</pub-id><pub-id pub-id-type="pmid">27849596</pub-id></element-citation></ref><ref id="bib167"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Prindiville</surname><given-names>TP</given-names></name><name><surname>Sheikh</surname><given-names>RA</given-names></name><name><surname>Cohen</surname><given-names>SH</given-names></name><name><surname>Tang</surname><given-names>YJ</given-names></name><name><surname>Cantrell</surname><given-names>MC</given-names></name><name><surname>Silva</surname><given-names>J</given-names><suffix>Jr</suffix></name></person-group><year iso-8601-date="2000">2000</year><article-title>Bacteroides fragilis enterotoxin gene sequences in patients with inflammatory bowel disease</article-title><source>Emerging Infectious Diseases</source><volume>6</volume><fpage>171</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.3201/eid0602.000210</pub-id><pub-id pub-id-type="pmid">10756151</pub-id></element-citation></ref><ref id="bib168"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Cai</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>F</given-names></name><name><surname>Liang</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Guan</surname><given-names>Y</given-names></name><name><surname>Shen</surname><given-names>D</given-names></name><name><surname>Peng</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>D</given-names></name><name><surname>Jie</surname><given-names>Z</given-names></name><name><surname>Wu</surname><given-names>W</given-names></name><name><surname>Qin</surname><given-names>Y</given-names></name><name><surname>Xue</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Han</surname><given-names>L</given-names></name><name><surname>Lu</surname><given-names>D</given-names></name><name><surname>Wu</surname><given-names>P</given-names></name><name><surname>Dai</surname><given-names>Y</given-names></name><name><surname>Sun</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name><name><surname>Tang</surname><given-names>A</given-names></name><name><surname>Zhong</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Chen</surname><given-names>W</given-names></name><name><surname>Xu</surname><given-names>R</given-names></name><name><surname>Wang</surname><given-names>M</given-names></name><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Gong</surname><given-names>M</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>M</given-names></name><name><surname>Hansen</surname><given-names>T</given-names></name><name><surname>Sanchez</surname><given-names>G</given-names></name><name><surname>Raes</surname><given-names>J</given-names></name><name><surname>Falony</surname><given-names>G</given-names></name><name><surname>Okuda</surname><given-names>S</given-names></name><name><surname>Almeida</surname><given-names>M</given-names></name><name><surname>LeChatelier</surname><given-names>E</given-names></name><name><surname>Renault</surname><given-names>P</given-names></name><name><surname>Pons</surname><given-names>N</given-names></name><name><surname>Batto</surname><given-names>J-M</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Chen</surname><given-names>H</given-names></name><name><surname>Yang</surname><given-names>R</given-names></name><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Ehrlich</surname><given-names>SD</given-names></name><name><surname>Nielsen</surname><given-names>R</given-names></name><name><surname>Pedersen</surname><given-names>O</given-names></name><name><surname>Kristiansen</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A metagenome-wide association study of gut microbiota in type 2 diabetes</article-title><source>Nature</source><volume>490</volume><fpage>55</fpage><lpage>60</lpage><pub-id pub-id-type="doi">10.1038/nature11450</pub-id><pub-id pub-id-type="pmid">23023125</pub-id></element-citation></ref><ref id="bib169"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quince</surname><given-names>C</given-names></name><name><surname>Ijaz</surname><given-names>UZ</given-names></name><name><surname>Loman</surname><given-names>N</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Saulnier</surname><given-names>D</given-names></name><name><surname>Russell</surname><given-names>J</given-names></name><name><surname>Haig</surname><given-names>SJ</given-names></name><name><surname>Calus</surname><given-names>ST</given-names></name><name><surname>Quick</surname><given-names>J</given-names></name><name><surname>Barclay</surname><given-names>A</given-names></name><name><surname>Bertz</surname><given-names>M</given-names></name><name><surname>Blaut</surname><given-names>M</given-names></name><name><surname>Hansen</surname><given-names>R</given-names></name><name><surname>McGrogan</surname><given-names>P</given-names></name><name><surname>Russell</surname><given-names>RK</given-names></name><name><surname>Edwards</surname><given-names>CA</given-names></name><name><surname>Gerasimidis</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Extensive modulation of the fecal metagenome in children with crohn’s disease during exclusive enteral nutrition</article-title><source>The American Journal of Gastroenterology</source><volume>110</volume><fpage>1718</fpage><lpage>1729</lpage><pub-id pub-id-type="doi">10.1038/ajg.2015.357</pub-id><pub-id pub-id-type="pmid">26526081</pub-id></element-citation></ref><ref id="bib170"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Ralevic</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2015">2015</year><chapter-title>UDP-glucose</chapter-title><person-group person-group-type="editor"><name><surname>Ralevic</surname><given-names>V</given-names></name></person-group><source>Reference Module in Biomedical Sciences</source><publisher-name>Elsevier</publisher-name><fpage>1</fpage><lpage>4</lpage></element-citation></ref><ref id="bib171"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ramirez</surname><given-names>J</given-names></name><name><surname>Guarner</surname><given-names>F</given-names></name><name><surname>Bustos Fernandez</surname><given-names>L</given-names></name><name><surname>Maruy</surname><given-names>A</given-names></name><name><surname>Sdepanian</surname><given-names>VL</given-names></name><name><surname>Cohen</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Antibiotics as Mmajor disruptors of gut microbiotajor Disruptors of Gut Microbiota</article-title><source>Frontiers in Cellular and Infection Microbiology</source><volume>10</volume><elocation-id>572912</elocation-id><pub-id pub-id-type="doi">10.3389/fcimb.2020.572912</pub-id><pub-id pub-id-type="pmid">33330122</pub-id></element-citation></ref><ref id="bib172"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rampelli</surname><given-names>S</given-names></name><name><surname>Schnorr</surname><given-names>SL</given-names></name><name><surname>Consolandi</surname><given-names>C</given-names></name><name><surname>Turroni</surname><given-names>S</given-names></name><name><surname>Severgnini</surname><given-names>M</given-names></name><name><surname>Peano</surname><given-names>C</given-names></name><name><surname>Brigidi</surname><given-names>P</given-names></name><name><surname>Crittenden</surname><given-names>AN</given-names></name><name><surname>Henry</surname><given-names>AG</given-names></name><name><surname>Candela</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Metagenome sequencing of the hadza hunter-gatherer gut microbiota</article-title><source>Current Biology</source><volume>25</volume><fpage>1682</fpage><lpage>1693</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2015.04.055</pub-id><pub-id pub-id-type="pmid">25981789</pub-id></element-citation></ref><ref id="bib173"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raymond</surname><given-names>F</given-names></name><name><surname>Ouameur</surname><given-names>AA</given-names></name><name><surname>Déraspe</surname><given-names>M</given-names></name><name><surname>Iqbal</surname><given-names>N</given-names></name><name><surname>Gingras</surname><given-names>H</given-names></name><name><surname>Dridi</surname><given-names>B</given-names></name><name><surname>Leprohon</surname><given-names>P</given-names></name><name><surname>Plante</surname><given-names>P-L</given-names></name><name><surname>Giroux</surname><given-names>R</given-names></name><name><surname>Bérubé</surname><given-names>È</given-names></name><name><surname>Frenette</surname><given-names>J</given-names></name><name><surname>Boudreau</surname><given-names>DK</given-names></name><name><surname>Simard</surname><given-names>J-L</given-names></name><name><surname>Chabot</surname><given-names>I</given-names></name><name><surname>Domingo</surname><given-names>M-C</given-names></name><name><surname>Trottier</surname><given-names>S</given-names></name><name><surname>Boissinot</surname><given-names>M</given-names></name><name><surname>Huletsky</surname><given-names>A</given-names></name><name><surname>Roy</surname><given-names>PH</given-names></name><name><surname>Ouellette</surname><given-names>M</given-names></name><name><surname>Bergeron</surname><given-names>MG</given-names></name><name><surname>Corbeil</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The initial state of the human gut microbiome determines its reshaping by antibiotics</article-title><source>The ISME Journal</source><volume>10</volume><fpage>707</fpage><lpage>720</lpage><pub-id pub-id-type="doi">10.1038/ismej.2015.148</pub-id><pub-id pub-id-type="pmid">26359913</pub-id></element-citation></ref><ref id="bib174"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rehman</surname><given-names>T</given-names></name><name><surname>Shabbir</surname><given-names>MA</given-names></name><name><surname>Inam‐Ur‐Raheem</surname><given-names>M</given-names></name><name><surname>Manzoor</surname><given-names>MF</given-names></name><name><surname>Ahmad</surname><given-names>N</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Ahmad</surname><given-names>MH</given-names></name><name><surname>Siddeeg</surname><given-names>A</given-names></name><name><surname>Abid</surname><given-names>M</given-names></name><name><surname>Aadil</surname><given-names>RM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Cysteine and homocysteine as biomarker of various diseases</article-title><source>Food Science &amp; Nutrition</source><volume>8</volume><fpage>4696</fpage><lpage>4707</lpage><pub-id pub-id-type="doi">10.1002/fsn3.1818</pub-id></element-citation></ref><ref id="bib175"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Reiter</surname><given-names>TE</given-names></name><name><surname>Irber</surname><given-names>L</given-names></name><name><surname>Gingrich</surname><given-names>AA</given-names></name><name><surname>Haynes</surname><given-names>D</given-names></name><name><surname>Pierce-Ward</surname><given-names>NT</given-names></name><name><surname>Brooks</surname><given-names>PT</given-names></name><name><surname>Mizutani</surname><given-names>Y</given-names></name><name><surname>Moritz</surname><given-names>D</given-names></name><name><surname>Reidl</surname><given-names>F</given-names></name><name><surname>Willis</surname><given-names>AD</given-names></name><name><surname>Sullivan</surname><given-names>BD</given-names></name><name><surname>Brown</surname><given-names>CT</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Meta-analysis of metagenomes via machine learning and assembly graphs reveals strain switches in Crohn’s disease</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.06.30.498290</pub-id></element-citation></ref><ref id="bib176"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rhodes</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The role of <italic>Escherichia coli</italic> in inflammatory bowel disease</article-title><source>Gut</source><volume>56</volume><fpage>610</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1136/gut.2006.111872</pub-id><pub-id pub-id-type="pmid">17440180</pub-id></element-citation></ref><ref id="bib177"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rigottier-Gois</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Dysbiosis in inflammatory bowel diseases: the oxygen hypothesis</article-title><source>The ISME Journal</source><volume>7</volume><fpage>1256</fpage><lpage>1261</lpage><pub-id pub-id-type="doi">10.1038/ismej.2013.80</pub-id><pub-id pub-id-type="pmid">23677008</pub-id></element-citation></ref><ref id="bib178"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roager</surname><given-names>HM</given-names></name><name><surname>Licht</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Microbial tryptophan catabolites in health and disease</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>3294</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-05470-4</pub-id><pub-id pub-id-type="pmid">30120222</pub-id></element-citation></ref><ref id="bib179"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname><given-names>A</given-names></name><name><surname>Lichtiger</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Clostridium difficile infection: a rarity in patients receiving chronic antibiotic treatment for Crohn’s disease</article-title><source>Inflammatory Bowel Diseases</source><volume>22</volume><fpage>648</fpage><lpage>653</lpage><pub-id pub-id-type="doi">10.1097/MIB.0000000000000641</pub-id><pub-id pub-id-type="pmid">26650148</pub-id></element-citation></ref><ref id="bib180"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ryczko</surname><given-names>MC</given-names></name><name><surname>Pawling</surname><given-names>J</given-names></name><name><surname>Chen</surname><given-names>R</given-names></name><name><surname>Abdel Rahman</surname><given-names>AM</given-names></name><name><surname>Yau</surname><given-names>K</given-names></name><name><surname>Copeland</surname><given-names>JK</given-names></name><name><surname>Zhang</surname><given-names>C</given-names></name><name><surname>Surendra</surname><given-names>A</given-names></name><name><surname>Guttman</surname><given-names>DS</given-names></name><name><surname>Figeys</surname><given-names>D</given-names></name><name><surname>Dennis</surname><given-names>JW</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Metabolic reprogramming by hexosamine biosynthetic and golgi N-glycan branching pathways</article-title><source>Scientific Reports</source><volume>6</volume><elocation-id>23043</elocation-id><pub-id pub-id-type="doi">10.1038/srep23043</pub-id><pub-id pub-id-type="pmid">26972830</pub-id></element-citation></ref><ref id="bib181"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Saitoh</surname><given-names>S</given-names></name><name><surname>Noda</surname><given-names>S</given-names></name><name><surname>Aiba</surname><given-names>Y</given-names></name><name><surname>Takagi</surname><given-names>A</given-names></name><name><surname>Sakamoto</surname><given-names>M</given-names></name><name><surname>Benno</surname><given-names>Y</given-names></name><name><surname>Koga</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Bacteroides ovatus as the predominant commensal intestinal microbe causing a systemic antibody response in inflammatory bowel disease</article-title><source>Clinical and Diagnostic Laboratory Immunology</source><volume>9</volume><fpage>54</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.1128/cdli.9.1.54-59.2002</pub-id><pub-id pub-id-type="pmid">11777829</pub-id></element-citation></ref><ref id="bib182"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sartor</surname><given-names>RB</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Mechanisms of disease: pathogenesis of Crohn’s disease and ulcerative colitis</article-title><source>Nature Clinical Practice. Gastroenterology &amp; Hepatology</source><volume>3</volume><fpage>390</fpage><lpage>407</lpage><pub-id pub-id-type="doi">10.1038/ncpgasthep0528</pub-id><pub-id pub-id-type="pmid">16819502</pub-id></element-citation></ref><ref id="bib183"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schirmer</surname><given-names>M</given-names></name><name><surname>Franzosa</surname><given-names>EA</given-names></name><name><surname>Lloyd-Price</surname><given-names>J</given-names></name><name><surname>McIver</surname><given-names>LJ</given-names></name><name><surname>Schwager</surname><given-names>R</given-names></name><name><surname>Poon</surname><given-names>TW</given-names></name><name><surname>Ananthakrishnan</surname><given-names>AN</given-names></name><name><surname>Andrews</surname><given-names>E</given-names></name><name><surname>Barron</surname><given-names>G</given-names></name><name><surname>Lake</surname><given-names>K</given-names></name><name><surname>Prasad</surname><given-names>M</given-names></name><name><surname>Sauk</surname><given-names>J</given-names></name><name><surname>Stevens</surname><given-names>B</given-names></name><name><surname>Wilson</surname><given-names>RG</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Denson</surname><given-names>LA</given-names></name><name><surname>Kugathasan</surname><given-names>S</given-names></name><name><surname>McGovern</surname><given-names>DPB</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Dynamics of metatranscription in the inflammatory bowel disease gut microbiome</article-title><source>Nature Microbiology</source><volume>3</volume><fpage>337</fpage><lpage>346</lpage><pub-id pub-id-type="doi">10.1038/s41564-017-0089-z</pub-id><pub-id pub-id-type="pmid">29311644</pub-id></element-citation></ref><ref id="bib184"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schirmer</surname><given-names>M</given-names></name><name><surname>Garner</surname><given-names>A</given-names></name><name><surname>Vlamakis</surname><given-names>H</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Microbial genes and pathways in inflammatory bowel disease</article-title><source>Nature Reviews. Microbiology</source><volume>17</volume><fpage>497</fpage><lpage>511</lpage><pub-id pub-id-type="doi">10.1038/s41579-019-0213-6</pub-id><pub-id pub-id-type="pmid">31249397</pub-id></element-citation></ref><ref id="bib185"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Scrivens</surname><given-names>M</given-names></name><name><surname>Dickenson</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Functional expression of the P2Y14 receptor in murine T-lymphocytes</article-title><source>British Journal of Pharmacology</source><volume>146</volume><fpage>435</fpage><lpage>444</lpage><pub-id pub-id-type="doi">10.1038/sj.bjp.0706322</pub-id><pub-id pub-id-type="pmid">15997228</pub-id></element-citation></ref><ref id="bib186"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Seemann</surname><given-names>TN</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>Barrnap: bacterial ribosomal RNA predictor</data-title><version designator="version 3">version 3</version><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/tseemann/barrnap">https://github.com/tseemann/barrnap</ext-link></element-citation></ref><ref id="bib187"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Seetharam</surname><given-names>B</given-names></name><name><surname>Alpers</surname><given-names>DH</given-names></name></person-group><year iso-8601-date="1982">1982</year><article-title>Absorption and transport of cobalamin (vitamin B12)</article-title><source>Annual Review of Nutrition</source><volume>2</volume><fpage>343</fpage><lpage>369</lpage><pub-id pub-id-type="doi">10.1146/annurev.nu.02.070182.002015</pub-id><pub-id pub-id-type="pmid">6313022</pub-id></element-citation></ref><ref id="bib188"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sen</surname><given-names>P</given-names></name><name><surname>Orešič</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Metabolic modeling of human gut microbiota on a genome scale: an overview</article-title><source>Metabolites</source><volume>9</volume><elocation-id>22</elocation-id><pub-id pub-id-type="doi">10.3390/metabo9020022</pub-id><pub-id pub-id-type="pmid">30695998</pub-id></element-citation></ref><ref id="bib189"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shaffer</surname><given-names>M</given-names></name><name><surname>Borton</surname><given-names>MA</given-names></name><name><surname>McGivern</surname><given-names>BB</given-names></name><name><surname>Zayed</surname><given-names>AA</given-names></name><name><surname>La Rosa</surname><given-names>SL</given-names></name><name><surname>Solden</surname><given-names>LM</given-names></name><name><surname>Liu</surname><given-names>P</given-names></name><name><surname>Narrowe</surname><given-names>AB</given-names></name><name><surname>Rodríguez-Ramos</surname><given-names>J</given-names></name><name><surname>Bolduc</surname><given-names>B</given-names></name><name><surname>Gazitúa</surname><given-names>MC</given-names></name><name><surname>Daly</surname><given-names>RA</given-names></name><name><surname>Smith</surname><given-names>GJ</given-names></name><name><surname>Vik</surname><given-names>DR</given-names></name><name><surname>Pope</surname><given-names>PB</given-names></name><name><surname>Sullivan</surname><given-names>MB</given-names></name><name><surname>Roux</surname><given-names>S</given-names></name><name><surname>Wrighton</surname><given-names>KC</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>DRAM for distilling microbial metabolism to automate the curation of microbiome function</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>8883</fpage><lpage>8900</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa621</pub-id><pub-id pub-id-type="pmid">32766782</pub-id></element-citation></ref><ref id="bib190"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname><given-names>YM</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The role of hypoxia in intestinal inflammation</article-title><source>Molecular and Cellular Pediatrics</source><volume>3</volume><elocation-id>301</elocation-id><pub-id pub-id-type="doi">10.1186/s40348-016-0030-1</pub-id><pub-id pub-id-type="pmid">26812949</pub-id></element-citation></ref><ref id="bib191"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shaiber</surname><given-names>A</given-names></name><name><surname>Willis</surname><given-names>AD</given-names></name><name><surname>Delmont</surname><given-names>TO</given-names></name><name><surname>Roux</surname><given-names>S</given-names></name><name><surname>Chen</surname><given-names>L-X</given-names></name><name><surname>Schmid</surname><given-names>AC</given-names></name><name><surname>Yousef</surname><given-names>M</given-names></name><name><surname>Watson</surname><given-names>AR</given-names></name><name><surname>Lolans</surname><given-names>K</given-names></name><name><surname>Esen</surname><given-names>ÖC</given-names></name><name><surname>Lee</surname><given-names>STM</given-names></name><name><surname>Downey</surname><given-names>N</given-names></name><name><surname>Morrison</surname><given-names>HG</given-names></name><name><surname>Dewhirst</surname><given-names>FE</given-names></name><name><surname>Mark Welch</surname><given-names>JL</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Functional and genetic markers of niche partitioning among enigmatic members of the human oral microbiome</article-title><source>Genome Biology</source><volume>21</volume><elocation-id>292</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-020-02195-w</pub-id><pub-id pub-id-type="pmid">33323122</pub-id></element-citation></ref><ref id="bib192"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shan</surname><given-names>Y</given-names></name><name><surname>Lee</surname><given-names>M</given-names></name><name><surname>Chang</surname><given-names>EB</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The gut microbiome and inflammatory bowel diseases</article-title><source>Annual Review of Medicine</source><volume>73</volume><fpage>455</fpage><lpage>468</lpage><pub-id pub-id-type="doi">10.1146/annurev-med-042320-021020</pub-id><pub-id pub-id-type="pmid">34555295</pub-id></element-citation></ref><ref id="bib193"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharpton</surname><given-names>T</given-names></name><name><surname>Lyalina</surname><given-names>S</given-names></name><name><surname>Luong</surname><given-names>J</given-names></name><name><surname>Pham</surname><given-names>J</given-names></name><name><surname>Deal</surname><given-names>EM</given-names></name><name><surname>Armour</surname><given-names>C</given-names></name><name><surname>Gaulke</surname><given-names>C</given-names></name><name><surname>Sanjabi</surname><given-names>S</given-names></name><name><surname>Pollard</surname><given-names>KS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Development of inflammatory bowel disease is linked to a longitudinal restructuring of the gut metagenome in mice</article-title><source>mSystems</source><volume>2</volume><elocation-id>e00036-17</elocation-id><pub-id pub-id-type="doi">10.1128/mSystems.00036-17</pub-id><pub-id pub-id-type="pmid">28904997</pub-id></element-citation></ref><ref id="bib194"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shaw</surname><given-names>SY</given-names></name><name><surname>Blanchard</surname><given-names>JF</given-names></name><name><surname>Bernstein</surname><given-names>CN</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Association between the use of antibiotics and new diagnoses of Crohn’s disease and ulcerative colitis</article-title><source>The American Journal of Gastroenterology</source><volume>106</volume><fpage>2133</fpage><lpage>2142</lpage><pub-id pub-id-type="doi">10.1038/ajg.2011.304</pub-id><pub-id pub-id-type="pmid">21912437</pub-id></element-citation></ref><ref id="bib195"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sherrill</surname><given-names>C</given-names></name><name><surname>Fahey</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Import and metabolism of glutathione by Streptococcus mutans</article-title><source>Journal of Bacteriology</source><volume>180</volume><fpage>1454</fpage><lpage>1459</lpage><pub-id pub-id-type="doi">10.1128/JB.180.6.1454-1459.1998</pub-id><pub-id pub-id-type="pmid">9515913</pub-id></element-citation></ref><ref id="bib196"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname><given-names>K</given-names></name><name><surname>Gobert</surname><given-names>AP</given-names></name><name><surname>Coburn</surname><given-names>LA</given-names></name><name><surname>Barry</surname><given-names>DP</given-names></name><name><surname>Allaman</surname><given-names>M</given-names></name><name><surname>Asim</surname><given-names>M</given-names></name><name><surname>Luis</surname><given-names>PB</given-names></name><name><surname>Schneider</surname><given-names>C</given-names></name><name><surname>Milne</surname><given-names>GL</given-names></name><name><surname>Boone</surname><given-names>HH</given-names></name><name><surname>Shilts</surname><given-names>MH</given-names></name><name><surname>Washington</surname><given-names>MK</given-names></name><name><surname>Das</surname><given-names>SR</given-names></name><name><surname>Piazuelo</surname><given-names>MB</given-names></name><name><surname>Wilson</surname><given-names>KT</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Dietary arginine regulates severity of experimental colitis and affects the colonic microbiome</article-title><source>Frontiers in Cellular and Infection Microbiology</source><volume>9</volume><elocation-id>66</elocation-id><pub-id pub-id-type="doi">10.3389/fcimb.2019.00066</pub-id><pub-id pub-id-type="pmid">30972302</pub-id></element-citation></ref><ref id="bib197"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sinha</surname><given-names>R</given-names></name><name><surname>Abu-Ali</surname><given-names>G</given-names></name><name><surname>Vogtmann</surname><given-names>E</given-names></name><name><surname>Fodor</surname><given-names>AA</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Amir</surname><given-names>A</given-names></name><name><surname>Schwager</surname><given-names>E</given-names></name><name><surname>Crabtree</surname><given-names>J</given-names></name><name><surname>Ma</surname><given-names>S</given-names></name><name><surname>Abnet</surname><given-names>CC</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>White</surname><given-names>O</given-names></name><name><surname>Huttenhower</surname><given-names>C</given-names></name><collab>Microbiome Quality Control Project Consortium</collab></person-group><year iso-8601-date="2017">2017</year><article-title>Assessment of variation in microbial community amplicon sequencing by the Microbiome Quality Control (MBQC) project consortium</article-title><source>Nature Biotechnology</source><volume>35</volume><fpage>1077</fpage><lpage>1086</lpage><pub-id pub-id-type="doi">10.1038/nbt.3981</pub-id><pub-id pub-id-type="pmid">28967885</pub-id></element-citation></ref><ref id="bib198"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Skelton</surname><given-names>L</given-names></name><name><surname>Cooper</surname><given-names>M</given-names></name><name><surname>Murphy</surname><given-names>M</given-names></name><name><surname>Platt</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Human immature monocyte-derived dendritic cells express the G protein-coupled receptor GPR105 (KIAA0001, P2Y14) and increase intracellular calcium in response to its agonist, uridine diphosphoglucose</article-title><source>Journal of Immunology</source><volume>171</volume><fpage>1941</fpage><lpage>1949</lpage><pub-id pub-id-type="doi">10.4049/jimmunol.171.4.1941</pub-id><pub-id pub-id-type="pmid">12902497</pub-id></element-citation></ref><ref id="bib199"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soderberg</surname><given-names>Tim</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Biosynthesis of ribose-5-phosphate and erythrose-4-phosphate in archaea: A phylogenetic analysis of archaeal genomes</article-title><source>Archaea</source><volume>1</volume><fpage>347</fpage><lpage>352</lpage><pub-id pub-id-type="doi">10.1155/2005/314760</pub-id><pub-id pub-id-type="pmid">15876568</pub-id></element-citation></ref><ref id="bib200"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sonnhammer</surname><given-names>EL</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Pfam: A comprehensive database of protein domain families based on seed alignments</article-title><source>Proteins</source><volume>28</volume><fpage>405</fpage><lpage>420</lpage><pub-id pub-id-type="doi">10.1002/(sici)1097-0134(199707)28:3&lt;405::aid-prot10&gt;3.0.co;2-l</pub-id><pub-id pub-id-type="pmid">9223186</pub-id></element-citation></ref><ref id="bib201"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sorbara</surname><given-names>MT</given-names></name><name><surname>Pamer</surname><given-names>EG</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Microbiome-based therapeutics</article-title><source>Nature Reviews. Microbiology</source><volume>20</volume><fpage>365</fpage><lpage>380</lpage><pub-id pub-id-type="doi">10.1038/s41579-021-00667-9</pub-id><pub-id pub-id-type="pmid">34992261</pub-id></element-citation></ref><ref id="bib202"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soufli</surname><given-names>I</given-names></name><name><surname>Toumi</surname><given-names>R</given-names></name><name><surname>Rafa</surname><given-names>H</given-names></name><name><surname>Touil-Boukoffa</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Overview of cytokines and nitric oxide involvement in immuno-pathogenesis of inflammatory bowel diseases</article-title><source>World Journal of Gastrointestinal Pharmacology and Therapeutics</source><volume>7</volume><elocation-id>353</elocation-id><pub-id pub-id-type="doi">10.4292/wjgpt.v7.i3.353</pub-id></element-citation></ref><ref id="bib203"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sprong</surname><given-names>RC</given-names></name><name><surname>Schonewille</surname><given-names>AJ</given-names></name><name><surname>van der Meer</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Dietary cheese whey protein protects rats against mild dextran sulfate sodium-induced colitis: role of mucin and microbiota</article-title><source>Journal of Dairy Science</source><volume>93</volume><fpage>1364</fpage><lpage>1371</lpage><pub-id pub-id-type="doi">10.3168/jds.2009-2397</pub-id><pub-id pub-id-type="pmid">20338413</pub-id></element-citation></ref><ref id="bib204"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spry</surname><given-names>C</given-names></name><name><surname>Kirk</surname><given-names>K</given-names></name><name><surname>Saliba</surname><given-names>KJ</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Coenzyme A biosynthesis: an antimicrobial drug target</article-title><source>FEMS Microbiology Reviews</source><volume>32</volume><fpage>56</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1111/j.1574-6976.2007.00093.x</pub-id><pub-id pub-id-type="pmid">18173393</pub-id></element-citation></ref><ref id="bib205"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets</article-title><source>Nature Biotechnology</source><volume>35</volume><fpage>1026</fpage><lpage>1028</lpage><pub-id pub-id-type="doi">10.1038/nbt.3988</pub-id><pub-id pub-id-type="pmid">29035372</pub-id></element-citation></ref><ref id="bib206"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teng</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Sui</surname><given-names>X</given-names></name><name><surname>Fan</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Lei</surname><given-names>X</given-names></name><name><surname>Shi</surname><given-names>C</given-names></name><name><surname>Sun</surname><given-names>W</given-names></name><name><surname>Song</surname><given-names>M</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Dong</surname><given-names>D</given-names></name><name><surname>Geng</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhu</surname><given-names>X</given-names></name><name><surname>Cai</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>B</given-names></name><name><surname>Min</surname><given-names>Q</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Zhan</surname><given-names>Q</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Gut microbiota-mediated nucleotide synthesis attenuates the response to neoadjuvant chemoradiotherapy in rectal cancer</article-title><source>Cancer Cell</source><volume>41</volume><fpage>124</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.1016/j.ccell.2022.11.013</pub-id><pub-id pub-id-type="pmid">36563680</pub-id></element-citation></ref><ref id="bib207"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tepe</surname><given-names>N</given-names></name><name><surname>Shimko</surname><given-names>LA</given-names></name><name><surname>Duran</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Toxic effects of thiol-reactive compounds on anaerobic biomass</article-title><source>Bioresource Technology</source><volume>97</volume><fpage>592</fpage><lpage>598</lpage><pub-id pub-id-type="doi">10.1016/j.biortech.2005.03.029</pub-id><pub-id pub-id-type="pmid">15913993</pub-id></element-citation></ref><ref id="bib208"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tong</surname><given-names>BC</given-names></name><name><surname>Barbul</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Cellular and physiological effects of arginine</article-title><source>Mini Reviews in Medicinal Chemistry</source><volume>4</volume><fpage>823</fpage><lpage>832</lpage><pub-id pub-id-type="doi">10.2174/1389557043403305</pub-id><pub-id pub-id-type="pmid">15544543</pub-id></element-citation></ref><ref id="bib209"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Treem</surname><given-names>WR</given-names></name><name><surname>Ahsan</surname><given-names>N</given-names></name><name><surname>Shoup</surname><given-names>M</given-names></name><name><surname>Hyams</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Fecal short-chain fatty acids in children with inflammatory bowel disease</article-title><source>Journal of Pediatric Gastroenterology and Nutrition</source><volume>18</volume><fpage>159</fpage><lpage>164</lpage><pub-id pub-id-type="doi">10.1097/00005176-199402000-00007</pub-id><pub-id pub-id-type="pmid">8014762</pub-id></element-citation></ref><ref id="bib210"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Turnbaugh</surname><given-names>PJ</given-names></name><name><surname>Hamady</surname><given-names>M</given-names></name><name><surname>Yatsunenko</surname><given-names>T</given-names></name><name><surname>Cantarel</surname><given-names>BL</given-names></name><name><surname>Duncan</surname><given-names>A</given-names></name><name><surname>Ley</surname><given-names>RE</given-names></name><name><surname>Sogin</surname><given-names>ML</given-names></name><name><surname>Jones</surname><given-names>WJ</given-names></name><name><surname>Roe</surname><given-names>BA</given-names></name><name><surname>Affourtit</surname><given-names>JP</given-names></name><name><surname>Egholm</surname><given-names>M</given-names></name><name><surname>Henrissat</surname><given-names>B</given-names></name><name><surname>Heath</surname><given-names>AC</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Gordon</surname><given-names>JI</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>A core gut microbiome in obese and lean twins</article-title><source>Nature</source><volume>457</volume><fpage>480</fpage><lpage>484</lpage><pub-id pub-id-type="doi">10.1038/nature07540</pub-id><pub-id pub-id-type="pmid">19043404</pub-id></element-citation></ref><ref id="bib211"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ungaro</surname><given-names>R</given-names></name><name><surname>Bernstein</surname><given-names>CN</given-names></name><name><surname>Gearry</surname><given-names>R</given-names></name><name><surname>Hviid</surname><given-names>A</given-names></name><name><surname>Kolho</surname><given-names>K-L</given-names></name><name><surname>Kronman</surname><given-names>MP</given-names></name><name><surname>Shaw</surname><given-names>S</given-names></name><name><surname>Van Kruiningen</surname><given-names>H</given-names></name><name><surname>Colombel</surname><given-names>J-F</given-names></name><name><surname>Atreja</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Antibiotics associated with increased risk of new-onset Crohn’s disease but not ulcerative colitis: A meta-analysis</article-title><source>The American Journal of Gastroenterology</source><volume>109</volume><fpage>1728</fpage><lpage>1738</lpage><pub-id pub-id-type="doi">10.1038/ajg.2014.246</pub-id><pub-id pub-id-type="pmid">25223575</pub-id></element-citation></ref><ref id="bib212"><element-citation publication-type="web"><person-group person-group-type="author"><collab>University of Sydney</collab></person-group><year iso-8601-date="2022">2022</year><article-title>metagenome fecal microbiota, Ilumina seq reads of 12 individuals at 2 timepoints</article-title><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJEB6092">https://www.ncbi.nlm.nih.gov/bioproject/PRJEB6092</ext-link><date-in-citation iso-8601-date="2022-09-23">September 23, 2022</date-in-citation></element-citation></ref><ref id="bib213"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vacca</surname><given-names>M</given-names></name><name><surname>Celano</surname><given-names>G</given-names></name><name><surname>Calabrese</surname><given-names>FM</given-names></name><name><surname>Portincasa</surname><given-names>P</given-names></name><name><surname>Gobbetti</surname><given-names>M</given-names></name><name><surname>De Angelis</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The controversial role of human gut lachnospiraceae</article-title><source>Microorganisms</source><volume>8</volume><elocation-id>573</elocation-id><pub-id pub-id-type="doi">10.3390/microorganisms8040573</pub-id><pub-id pub-id-type="pmid">32326636</pub-id></element-citation></ref><ref id="bib214"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vaisman</surname><given-names>A</given-names></name><name><surname>Pivovarov</surname><given-names>K</given-names></name><name><surname>McGeer</surname><given-names>A</given-names></name><name><surname>Willey</surname><given-names>B</given-names></name><name><surname>Borgundvaag</surname><given-names>B</given-names></name><name><surname>Porter</surname><given-names>V</given-names></name><name><surname>Gnanasuntharam</surname><given-names>P</given-names></name><name><surname>Wei</surname><given-names>Y</given-names></name><name><surname>Nguyen</surname><given-names>GC</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Prevalence and incidence of antimicrobial-resistant organisms among hospitalized inflammatory bowel disease patients</article-title><source>The Canadian Journal of Infectious Diseases &amp; Medical Microbiology = Journal Canadien Des Maladies Infectieuses et de La Microbiologie Medicale</source><volume>24</volume><fpage>e117</fpage><lpage>e21</lpage><pub-id pub-id-type="doi">10.1155/2013/609230</pub-id><pub-id pub-id-type="pmid">24489571</pub-id></element-citation></ref><ref id="bib215"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Dam</surname><given-names>V</given-names></name><name><surname>Olrichs</surname><given-names>N</given-names></name><name><surname>Breukink</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Specific labeling of peptidoglycan precursors as a tool for bacterial cell wall studies</article-title><source>Chembiochem</source><volume>10</volume><fpage>617</fpage><lpage>624</lpage><pub-id pub-id-type="doi">10.1002/cbic.200800678</pub-id><pub-id pub-id-type="pmid">19173317</pub-id></element-citation></ref><ref id="bib216"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vanni</surname><given-names>C</given-names></name><name><surname>Schechter</surname><given-names>MS</given-names></name><name><surname>Acinas</surname><given-names>SG</given-names></name><name><surname>Barberán</surname><given-names>A</given-names></name><name><surname>Buttigieg</surname><given-names>PL</given-names></name><name><surname>Casamayor</surname><given-names>EO</given-names></name><name><surname>Delmont</surname><given-names>TO</given-names></name><name><surname>Duarte</surname><given-names>CM</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Finn</surname><given-names>RD</given-names></name><name><surname>Kottmann</surname><given-names>R</given-names></name><name><surname>Mitchell</surname><given-names>A</given-names></name><name><surname>Sánchez</surname><given-names>P</given-names></name><name><surname>Siren</surname><given-names>K</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Gloeckner</surname><given-names>FO</given-names></name><name><surname>Fernàndez-Guerra</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Unifying the known and unknown microbial coding sequence space</article-title><source>eLife</source><volume>11</volume><elocation-id>e67667</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.67667</pub-id><pub-id pub-id-type="pmid">35356891</pub-id></element-citation></ref><ref id="bib217"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vich Vila</surname><given-names>A</given-names></name><name><surname>Imhann</surname><given-names>F</given-names></name><name><surname>Collij</surname><given-names>V</given-names></name><name><surname>Jankipersadsing</surname><given-names>SA</given-names></name><name><surname>Gurry</surname><given-names>T</given-names></name><name><surname>Mujagic</surname><given-names>Z</given-names></name><name><surname>Kurilshikov</surname><given-names>A</given-names></name><name><surname>Bonder</surname><given-names>MJ</given-names></name><name><surname>Jiang</surname><given-names>X</given-names></name><name><surname>Tigchelaar</surname><given-names>EF</given-names></name><name><surname>Dekens</surname><given-names>J</given-names></name><name><surname>Peters</surname><given-names>V</given-names></name><name><surname>Voskuil</surname><given-names>MD</given-names></name><name><surname>Visschedijk</surname><given-names>MC</given-names></name><name><surname>van Dullemen</surname><given-names>HM</given-names></name><name><surname>Keszthelyi</surname><given-names>D</given-names></name><name><surname>Swertz</surname><given-names>MA</given-names></name><name><surname>Franke</surname><given-names>L</given-names></name><name><surname>Alberts</surname><given-names>R</given-names></name><name><surname>Festen</surname><given-names>EAM</given-names></name><name><surname>Dijkstra</surname><given-names>G</given-names></name><name><surname>Masclee</surname><given-names>AAM</given-names></name><name><surname>Hofker</surname><given-names>MH</given-names></name><name><surname>Xavier</surname><given-names>RJ</given-names></name><name><surname>Alm</surname><given-names>EJ</given-names></name><name><surname>Fu</surname><given-names>J</given-names></name><name><surname>Wijmenga</surname><given-names>C</given-names></name><name><surname>Jonkers</surname><given-names>D</given-names></name><name><surname>Zhernakova</surname><given-names>A</given-names></name><name><surname>Weersma</surname><given-names>RK</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Gut microbiota composition and functional changes in inflammatory bowel disease and irritable bowel syndrome</article-title><source>Science Translational Medicine</source><volume>10</volume><elocation-id>eaap8914</elocation-id><pub-id pub-id-type="doi">10.1126/scitranslmed.aap8914</pub-id><pub-id pub-id-type="pmid">30567928</pub-id></element-citation></ref><ref id="bib218"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vineis</surname><given-names>JH</given-names></name><name><surname>Ringus</surname><given-names>DL</given-names></name><name><surname>Morrison</surname><given-names>HG</given-names></name><name><surname>Delmont</surname><given-names>TO</given-names></name><name><surname>Dalal</surname><given-names>S</given-names></name><name><surname>Raffals</surname><given-names>LH</given-names></name><name><surname>Antonopoulos</surname><given-names>DA</given-names></name><name><surname>Rubin</surname><given-names>DT</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name><name><surname>Chang</surname><given-names>EB</given-names></name><name><surname>Sogin</surname><given-names>ML</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Patient-specific bacteroides genome variants in pouchitis</article-title><source>mBio</source><volume>7</volume><elocation-id>e01713-16</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.01713-16</pub-id><pub-id pub-id-type="pmid">27935837</pub-id></element-citation></ref><ref id="bib219"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname><given-names>MY</given-names></name><name><surname>Pratap</surname><given-names>S</given-names></name><name><surname>Southerland</surname><given-names>JH</given-names></name><name><surname>Farmer-Dixon</surname><given-names>CM</given-names></name><name><surname>Lakshmyya</surname><given-names>K</given-names></name><name><surname>Gangula</surname><given-names>PR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Role of oral and gut microbiome in nitric oxide-mediated colon motility</article-title><source>Nitric Oxide</source><volume>73</volume><fpage>81</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1016/j.niox.2017.06.003</pub-id></element-citation></ref><ref id="bib220"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Watson</surname><given-names>AR</given-names></name><name><surname>Füssel</surname><given-names>J</given-names></name><name><surname>Veseli</surname><given-names>I</given-names></name><name><surname>DeLongchamp</surname><given-names>JZ</given-names></name><name><surname>Silva</surname><given-names>M</given-names></name><name><surname>Trigodet</surname><given-names>F</given-names></name><name><surname>Lolans</surname><given-names>K</given-names></name><name><surname>Shaiber</surname><given-names>A</given-names></name><name><surname>Fogarty</surname><given-names>E</given-names></name><name><surname>Runde</surname><given-names>JM</given-names></name><name><surname>Quince</surname><given-names>C</given-names></name><name><surname>Yu</surname><given-names>MK</given-names></name><name><surname>Söylev</surname><given-names>A</given-names></name><name><surname>Morrison</surname><given-names>HG</given-names></name><name><surname>Lee</surname><given-names>STM</given-names></name><name><surname>Kao</surname><given-names>D</given-names></name><name><surname>Rubin</surname><given-names>DT</given-names></name><name><surname>Jabri</surname><given-names>B</given-names></name><name><surname>Louie</surname><given-names>T</given-names></name><name><surname>Eren</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Metabolic independence drives gut microbial colonization and resilience in health and disease</article-title><source>Genome Biology</source><volume>24</volume><elocation-id>78</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-023-02924-x</pub-id><pub-id pub-id-type="pmid">37069665</pub-id></element-citation></ref><ref id="bib221"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname><given-names>C</given-names></name><name><surname>Zheng</surname><given-names>Z</given-names></name><name><surname>Shao</surname><given-names>T</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Xie</surname><given-names>Z</given-names></name><name><surname>Le Chatelier</surname><given-names>E</given-names></name><name><surname>He</surname><given-names>Z</given-names></name><name><surname>Zhong</surname><given-names>W</given-names></name><name><surname>Fan</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Wu</surname><given-names>C</given-names></name><name><surname>Hu</surname><given-names>C</given-names></name><name><surname>Xu</surname><given-names>Q</given-names></name><name><surname>Zhou</surname><given-names>J</given-names></name><name><surname>Cai</surname><given-names>S</given-names></name><name><surname>Wang</surname><given-names>D</given-names></name><name><surname>Huang</surname><given-names>Y</given-names></name><name><surname>Breban</surname><given-names>M</given-names></name><name><surname>Qin</surname><given-names>N</given-names></name><name><surname>Ehrlich</surname><given-names>SD</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Quantitative metagenomics reveals unique gut microbiome biomarkers in ankylosing spondylitis</article-title><source>Genome Biology</source><volume>18</volume><elocation-id>142</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-017-1271-6</pub-id><pub-id pub-id-type="pmid">28750650</pub-id></element-citation></ref><ref id="bib222"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wexler</surname><given-names>HM</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Bacteroides: the good, the bad, and the nitty-gritty</article-title><source>Clinical Microbiology Reviews</source><volume>20</volume><fpage>593</fpage><lpage>621</lpage><pub-id pub-id-type="doi">10.1128/CMR.00008-07</pub-id><pub-id pub-id-type="pmid">17934076</pub-id></element-citation></ref><ref id="bib223"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><source>Ggplot2: Elegant Graphics for Data Analysis</source><publisher-name>Springer</publisher-name></element-citation></ref><ref id="bib224"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wolfe</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The acetate switch</article-title><source>Microbiology and Molecular Biology Reviews</source><volume>69</volume><fpage>12</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1128/MMBR.69.1.12-50.2005</pub-id><pub-id pub-id-type="pmid">15755952</pub-id></element-citation></ref><ref id="bib225"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname><given-names>CC</given-names></name><name><surname>Fong</surname><given-names>W</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Gut microbes promote chemoradiotherapy resistance via metabolic cross-feeding</article-title><source>Cancer Cell</source><volume>41</volume><fpage>12</fpage><lpage>14</lpage><pub-id pub-id-type="doi">10.1016/j.ccell.2022.11.017</pub-id><pub-id pub-id-type="pmid">36563683</pub-id></element-citation></ref><ref id="bib226"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Woting</surname><given-names>A</given-names></name><name><surname>Blaut</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The Intestinal microbiota in metabolic disease</article-title><source>Nutrients</source><volume>8</volume><elocation-id>202</elocation-id><pub-id pub-id-type="doi">10.3390/nu8040202</pub-id><pub-id pub-id-type="pmid">27058556</pub-id></element-citation></ref><ref id="bib227"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>W</given-names></name><name><surname>Sun</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>F</given-names></name><name><surname>Cao</surname><given-names>AT</given-names></name><name><surname>Liu</surname><given-names>H</given-names></name><name><surname>Zhao</surname><given-names>Y</given-names></name><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Xiao</surname><given-names>Y</given-names></name><name><surname>Yao</surname><given-names>S</given-names></name><name><surname>Zhao</surname><given-names>Q</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Cong</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Microbiota metabolite short-chain fatty acid acetate promotes intestinal IgA response to microbiota which is mediated by GPR43</article-title><source>Mucosal Immunology</source><volume>10</volume><fpage>946</fpage><lpage>956</lpage><pub-id pub-id-type="doi">10.1038/mi.2016.114</pub-id><pub-id pub-id-type="pmid">27966553</pub-id></element-citation></ref><ref id="bib228"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>L</given-names></name><name><surname>Tang</surname><given-names>Z</given-names></name><name><surname>Chen</surname><given-names>H</given-names></name><name><surname>Ren</surname><given-names>Z</given-names></name><name><surname>Ding</surname><given-names>Q</given-names></name><name><surname>Liang</surname><given-names>K</given-names></name><name><surname>Sun</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Mutual interaction between gut microbiota and protein/amino acid metabolism for host mucosal immunity and health</article-title><source>Animal Nutrition</source><volume>7</volume><fpage>11</fpage><lpage>16</lpage><pub-id pub-id-type="doi">10.1016/j.aninu.2020.11.003</pub-id><pub-id pub-id-type="pmid">33997326</pub-id></element-citation></ref><ref id="bib229"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>H</given-names></name><name><surname>Guo</surname><given-names>R</given-names></name><name><surname>Zhong</surname><given-names>H</given-names></name><name><surname>Feng</surname><given-names>Q</given-names></name><name><surname>Lan</surname><given-names>Z</given-names></name><name><surname>Qin</surname><given-names>B</given-names></name><name><surname>Ward</surname><given-names>KJ</given-names></name><name><surname>Jackson</surname><given-names>MA</given-names></name><name><surname>Xia</surname><given-names>Y</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Chen</surname><given-names>B</given-names></name><name><surname>Xia</surname><given-names>H</given-names></name><name><surname>Xu</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>F</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Al-Aama</surname><given-names>JY</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Kristiansen</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Steves</surname><given-names>CJ</given-names></name><name><surname>Bell</surname><given-names>JT</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Spector</surname><given-names>TD</given-names></name><name><surname>Jia</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Shotgun Metagenomics of 250 adult twins reveals genetic and environmental impacts on the gut microbiome</article-title><source>Cell Systems</source><volume>3</volume><fpage>572</fpage><lpage>584</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2016.10.004</pub-id><pub-id pub-id-type="pmid">27818083</pub-id></element-citation></ref><ref id="bib230"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>J</given-names></name><name><surname>Kalhan</surname><given-names>SC</given-names></name><name><surname>Hanson</surname><given-names>RW</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>What is the metabolic role of phosphoenolpyruvate carboxykinase?</article-title><source>The Journal of Biological Chemistry</source><volume>284</volume><fpage>27025</fpage><lpage>27029</lpage><pub-id pub-id-type="doi">10.1074/jbc.R109.040543</pub-id><pub-id pub-id-type="pmid">19636077</pub-id></element-citation></ref><ref id="bib231"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Xu</surname><given-names>Y</given-names></name><name><surname>Luo</surname><given-names>T</given-names></name><name><surname>Ge</surname><given-names>Y</given-names></name><name><surname>Jiang</surname><given-names>Y</given-names></name><name><surname>Shi</surname><given-names>Y</given-names></name><name><surname>Sun</surname><given-names>J</given-names></name><name><surname>Le</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Dietary methionine restriction improves the gut microbiota and reduces intestinal permeability and inflammation in high-fat-fed mice</article-title><source>Food &amp; Function</source><volume>10</volume><fpage>5952</fpage><lpage>5968</lpage><pub-id pub-id-type="doi">10.1039/C9FO00766K</pub-id></element-citation></ref><ref id="bib232"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname><given-names>Y</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>A parsimony approach to biological pathway reconstruction/inference for genomes and metagenomes</article-title><source>PLOS Computational Biology</source><volume>5</volume><elocation-id>e1000465</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1000465</pub-id><pub-id pub-id-type="pmid">19680427</pub-id></element-citation></ref><ref id="bib233"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yusufu</surname><given-names>I</given-names></name><name><surname>Ding</surname><given-names>K</given-names></name><name><surname>Smith</surname><given-names>K</given-names></name><name><surname>Wankhade</surname><given-names>UD</given-names></name><name><surname>Sahay</surname><given-names>B</given-names></name><name><surname>Patterson</surname><given-names>GT</given-names></name><name><surname>Pacholczyk</surname><given-names>R</given-names></name><name><surname>Adusumilli</surname><given-names>S</given-names></name><name><surname>Hamrick</surname><given-names>MW</given-names></name><name><surname>Hill</surname><given-names>WD</given-names></name><name><surname>Isales</surname><given-names>CM</given-names></name><name><surname>Fulzele</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>A tryptophan-deficient diet induces gut microbiota dysbiosis and increases systemic inflammation in aged mice</article-title><source>International Journal of Molecular Sciences</source><volume>22</volume><elocation-id>5005</elocation-id><pub-id pub-id-type="doi">10.3390/ijms22095005</pub-id><pub-id pub-id-type="pmid">34066870</pub-id></element-citation></ref><ref id="bib234"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>H</given-names></name><name><surname>Chen</surname><given-names>T</given-names></name><name><surname>Shi</surname><given-names>L</given-names></name><name><surname>Wang</surname><given-names>D</given-names></name><name><surname>Tang</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Regulatory role of short-chain fatty acids in inflammatory bowel disease</article-title><source>Cell Communication and Signaling</source><volume>20</volume><elocation-id>64</elocation-id><pub-id pub-id-type="doi">10.1186/s12964-022-00869-5</pub-id><pub-id pub-id-type="pmid">35546404</pub-id></element-citation></ref><ref id="bib235"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Sun</surname><given-names>Q</given-names></name><name><surname>Qin</surname><given-names>P</given-names></name><name><surname>Wang</surname><given-names>Q</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Compositional and functional features of the female premenopausal and postmenopausal gut microbiota</article-title><source>FEBS Letters</source><volume>593</volume><fpage>2655</fpage><lpage>2664</lpage><pub-id pub-id-type="doi">10.1002/1873-3468.13527</pub-id><pub-id pub-id-type="pmid">31273779</pub-id></element-citation></ref><ref id="bib236"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname><given-names>Z</given-names></name><name><surname>Tran</surname><given-names>PQ</given-names></name><name><surname>Breister</surname><given-names>AM</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Kieft</surname><given-names>K</given-names></name><name><surname>Cowley</surname><given-names>ES</given-names></name><name><surname>Karaoz</surname><given-names>U</given-names></name><name><surname>Anantharaman</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>METABOLIC: high-throughput profiling of microbial genomes for functional traits, metabolism, biogeochemistry, and community-scale functional networks</article-title><source>Microbiome</source><volume>10</volume><elocation-id>33</elocation-id><pub-id pub-id-type="doi">10.1186/s40168-021-01213-8</pub-id><pub-id pub-id-type="pmid">35172890</pub-id></element-citation></ref><ref id="bib237"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zimmermann</surname><given-names>M</given-names></name><name><surname>Zimmermann-Kogadeeva</surname><given-names>M</given-names></name><name><surname>Wegmann</surname><given-names>R</given-names></name><name><surname>Goodman</surname><given-names>AL</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Mapping human microbiome drug metabolism by gut bacteria and their genes</article-title><source>Nature</source><volume>570</volume><fpage>462</fpage><lpage>467</lpage><pub-id pub-id-type="doi">10.1038/s41586-019-1291-3</pub-id></element-citation></ref><ref id="bib238"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zimmermann</surname><given-names>J</given-names></name><name><surname>Kaleta</surname><given-names>C</given-names></name><name><surname>Waschina</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>gapseq: informed prediction of bacterial metabolic pathways and reconstruction of accurate metabolic models</article-title><source>Genome Biology</source><volume>22</volume><elocation-id>81</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-021-02295-1</pub-id><pub-id pub-id-type="pmid">33691770</pub-id></element-citation></ref><ref id="bib239"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zong</surname><given-names>X</given-names></name><name><surname>Fu</surname><given-names>J</given-names></name><name><surname>Xu</surname><given-names>B</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Jin</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Interplay between gut microbiota and antimicrobial peptides</article-title><source>Animal Nutrition</source><volume>6</volume><fpage>389</fpage><lpage>396</lpage><pub-id pub-id-type="doi">10.1016/j.aninu.2020.09.002</pub-id><pub-id pub-id-type="pmid">33364454</pub-id></element-citation></ref><ref id="bib240"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zorrilla</surname><given-names>F</given-names></name><name><surname>Buric</surname><given-names>F</given-names></name><name><surname>Patil</surname><given-names>KR</given-names></name><name><surname>Zelezniak</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>metaGEM: reconstruction of genome scale metabolic models directly from metagenomes</article-title><source>Nucleic Acids Research</source><volume>49</volume><elocation-id>e126</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkab815</pub-id><pub-id pub-id-type="pmid">34614189</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s7"><title>Low sequencing depth results in poor characterization of community richness in assembled metagenomes</title><p>Within our dataset, we observed a correlation between the estimated number of distinct populations in assembled sequences and sequencing depth, i.e., the number of short reads generated from a given sample. In shallow sequencing of metagenomes, short reads may not cover the entirety of population genomes, thus decreasing the rate of recovery of SCGs in the assembly and resulting in an underestimation of the number of populations present. Indeed, the linear relationship between sequencing depth and the number of observed microbial populations we observe in lower depths of sequencing (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>) starts to plateau once the sequencing depth exceeds approximately 25 million reads, suggesting that our strategy to estimate the number of distinct microbial populations within these samples serves as a good approximation of the true number of genomes only at relatively higher depths of sequencing. Since an incomplete recovery of population genomes in metagenomic samples also interferes with a meaningful quantification of metabolic potential in a given sample, we set a minimum sequencing depth threshold of 25 million sequencing reads (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>). A set of 408 samples (101 IBD, 229 healthy, and 78 non-IBD) from 10 different studies passed our quality threshold to be utilized for further analysis.</p><p>Low sequencing depth disproportionately affects IBD metagenomes, thereby reducing our ability to effectively study this disease model in comparison with healthy controls. It also disproportionately affects some studies over others, which could allow cohort or study-specific effects to influence the differential signal between the groups. However, we concluded that the benefits of stringent thresholding outweigh the potential complications arising from imbalanced cohort sizes in our sample subset.</p><fig id="app1fig1" position="float"><label>Appendix 1—figure 1.</label><caption><title>Scatterplot of sequencing depth vs estimated number of microbial populations in each of 2893 stool metagenomes.</title><p>Sequencing depth is represented by the number of R1 reads, except for (<xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>) samples, in which case it is the number of merged paired-end reads. The vertical line indicates our sequencing depth threshold of 25 million reads. Per-group Spearman’s correlation coefficients and p-values are shown for the subset of samples with depth &lt; 25 million reads (top left) and for the subset with depth ≥ 25 million reads (top right). Regression lines are shown for each group in each subset, with standard error indicated by the colored background.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig1-v1.tif"/></fig></sec><sec sec-type="appendix" id="s8"><title>Technical details of metabolism estimation in anvi’o</title><p>This section describes technical details of the program ‘anvi-estimate-metabolism’, which is the main program in the metabolism reconstruction framework in anvi’o (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2A</xref>). Documentation for this program, including an extended and more up-to-date version of these technical details, can be found at <ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/anvi-estimate-metabolism">https://anvio.org/m/anvi-estimate-metabolism</ext-link>.</p><fig id="app1fig2" position="float"><label>Appendix 1—figure 2.</label><caption><title>Technical details of the metabolism reconstruction software framework in anvi’o.</title><p>(<bold>A</bold>) Workflow of metabolism reconstruction programs and their inputs/outputs. Dark arrows indicate the primary analysis path utilized in this study. Blue background indicates optional features in the framework. A demonstration of completeness score and copy number calculations for metabolic pathways performed by the program ‘anvi-estimate-metabolism’ is shown using example enzyme annotation data in panels <bold>B–E</bold> (for a theoretical pathway) and <bold>F–I</bold> (for a real pathway). (<bold>B</bold>) Theoretical metabolic pathway, where hexagons represent metabolites, arrows represent chemical reactions, letters represent enzymes (subscripts indicate enzyme components), and the example number of gene annotation hits for each enzyme is written in gray. (<bold>C</bold>) The definition of the theoretical pathway from panel B, written in terms of the required enzymes. (<bold>D</bold>) Table showing the major steps in the pathway and example calculations for step presence and copy number. Step presence is calculated by evaluating a Boolean expression created from the step definition in which enzymes with &gt; 0 hits are replaced with True (<bold>T</bold>) and the others with False (<bold>F</bold>). Step copy number is calculated by evaluating the corresponding arithmetic expression in which the enzymes are replaced with their annotation counts. (<bold>E</bold>) Final calculations of completeness score (fraction of present steps) and copy number for the theoretical metabolic pathway. (<bold>F–I</bold>) Same as panels <bold>B–E</bold>, but for KEGG module M00043. A high-resolution version of this figure is available at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.22851173">https://doi.org/10.6084/m9.figshare.22851173</ext-link>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig2-v1.tif"/></fig><sec sec-type="appendix" id="s8-1"><title>Summary of program usage</title><p>The program ‘anvi-estimate-metabolism’ predicts the metabolic capabilities of organisms based on their genetic content. It relies upon enzyme annotations and metabolism information from KEGG, specifically using metabolic modules from the KEGG MODULE (<xref ref-type="bibr" rid="bib93">Kanehisa et al., 2023</xref>) database, which are defined in terms of KEGG Orthologs (KOs) that can be annotated via the KOfam database of hidden Markov model (HMM) profiles (<xref ref-type="bibr" rid="bib9">Aramaki et al., 2020</xref>). It can also work with user-defined metabolic pathways, as described in the documentation page <ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/user-modules-data">https://anvio.org/m/user-modules-data</ext-link>. In our analysis, we used the program ‘anvi-run-kegg-kofams’ to annotate the KOs used for downstream metabolism estimation; this software implements a heuristic to obtain more accurate and comprehensive annotation results than other contemporary tools for KOfam annotation, as described in <xref ref-type="bibr" rid="bib90">Kananen et al., 2025</xref>.</p><p>The program ‘anvi-estimate-metabolism’ determines which enzymes are annotated in an input sample and uses these functions to compute the completeness and copy number of each metabolic module within the sample. Input samples can be individual genomes, binned or unbinned metagenomes, or ad hoc lists of enzyme accessions. The output of ‘anvi-estimate-metabolism’ is one or more tabular text files detailing the completeness and copy number scores per module as well as (customizable) information such as pathway metadata; shared/unique enzymes; gene coverage data; and pathway substrates, intermediates, and products. A detailed output description and examples can be found at <ext-link ext-link-type="uri" xlink:href="https://anvio.org/m/kegg-metabolism/">https://anvio.org/m/kegg-metabolism/</ext-link>.</p></sec><sec sec-type="appendix" id="s8-2"><title>Module definitions and interpretation strategies</title><p>Metabolic pathways are defined by the enzymes responsible for each reaction in the pathway, using the convention established by the KEGG MODULE database. In these definitions, commas separate alternative enzymes that can catalyze the same reaction, spaces separate subsequent reactions, plus signs indicate essential components of enzyme complexes, minus signs indicate nonessential components of complexes, and parentheses indicate the order of operations. These definitions can also be written in terms of the logical relationships between reactions, such that spaces and plus signs are converted into ‘AND’ relationships and commas are converted into ‘OR’ relationships (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2B–C and F–G</xref>).</p><p>‘anvi-estimate-metabolism’ has two strategies for interpreting module definition strings that treat alternative enzymes and pathway branches differently. One is the ‘pathwise’ strategy, which considers all possible combinations of enzymes. In this method, each alternative set of enzymes that could be used together to catalyze every reaction in the metabolic pathway is called a ‘path’ through the module. The program computes completeness and copy number metrics for each path separately, and then identifies the most complete path(s) as the most biologically relevant representative of the module as a whole. Alternatively, with the ‘stepwise’ strategy the module definition is parsed into high-level ‘steps’ that each encompasses a set of alternative enzymes for a particular reaction or branch point. The presence and copy numbers of each step are respectively combined into a completeness score and copy number for the entire module.</p></sec><sec sec-type="appendix" id="s8-3"><title>Calculation of stepwise completeness and copy number</title><p>The analyses in this paper rely on the ‘stepwise’ metrics of module completeness and copy number, which are calculated as demonstrated in <xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2D–E and H–I</xref>. We divide each module into steps by splitting the definition string on the outermost ‘AND’ relationships (spaces not within parentheses). To determine whether each step is present, we convert the step definition into a Boolean expression in which ‘True’ represents annotated enzymes and ‘False’ represents enzymes without annotations. If the Boolean expression evaluates to ‘True’, then the step is considered present. The module completeness score is the number of present steps divided by the total number of steps. To determine the step copy number, we convert the step definition into an arithmetic expression wherein ‘AND’ relationships become minimum operations and ‘OR’ relationships become addition operations. We take the minimum of all per-step copy numbers obtained by evaluating these arithmetic operations to get the overall module copy number.</p><fig id="app1fig3" position="float"><label>Appendix 1—figure 3.</label><caption><title>Comparison of unnormalized copy number data and normalized per-population copy number (PPCN) data for the inflammatory bowel disease (IBD)-enriched modules.</title><p>(<bold>A</bold>) Boxplot of median copy numbers for each module in the healthy samples (blue) and IBD samples (red). (<bold>B</bold>) Boxplots of median PPCN for each module in the healthy samples (blue) and IBD samples (red). Lines connect data points for the same module in each plot. The gray dashed line in each plot indicates the overall median value.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig3-v1.tif"/></fig></sec></sec><sec sec-type="appendix" id="s9"><title>Differential annotation efficiency between IBD and healthy samples</title><p>We observed that the proportion of predicted genes with functional annotations was markedly less in healthy metagenomes than in IBD samples, for both sequence homology-based annotation methods (NCBI COGs) and annotation with probabilistic models (KEGG KOfams and Pfams) (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1d</xref>). One possible interpretation of this that aligns with our metabolic competency hypothesis is that the populations with LMI that thrive in the healthy gut environment are relatively less well characterized than the HMI populations that are more likely to survive in the stressful conditions of IBD, resulting in an annotation bias against healthy samples. This interpretation is congruent with our observation that most uncharacterized gut microbial genomes from the GTDB, which have temporary code names in place of taxonomic assignments, were identified as non-HMI (<xref ref-type="fig" rid="fig3">Figure 3A</xref>).</p><fig id="app1fig4" position="float"><label>Appendix 1—figure 4.</label><caption><title>Histograms of annotations per gene call from (<bold>A, B</bold>) NCBI Clusters of Orthologous Groups (COGs); (<bold>C, D</bold>) KEGG KOfams; and (<bold>E, F</bold>) Pfams.</title><p>Panels A, C, and E show data for metagenomes in the subset of 330 deeply sequenced samples from healthy people and people with inflammatory bowel disease (IBD), and panels B, D, and F show data for all 2893 samples including those from non-IBD controls. (<bold>G</bold>) Proportion of genes with each classification from AGNOSTOS (<xref ref-type="bibr" rid="bib216">Vanni et al., 2022</xref>) in the subset of 330 deeply sequenced samples. (<bold>H</bold>) Proportion of genes with at least one annotation from KEGG KOfams (<xref ref-type="bibr" rid="bib9">Aramaki et al., 2020</xref>), NCBI COGs (<xref ref-type="bibr" rid="bib58">Galperin et al., 2015</xref>), or Pfams (<xref ref-type="bibr" rid="bib145">Mistry et al., 2021</xref>) (green) and proportion without any annotation (brown) in the subset of 330 deeply sequenced samples.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig4-v1.tif"/></fig><p>The reduced metabolic capacity of LMI microbes and their resulting reliance on robust community interactions (i.e. cross-feeding) may increase their resistance to cultivation that typically aims to isolate individual populations rather than communities. Microbes auxotrophic for key metabolites rely on metabolic interactions with their surrounding community, and current cultivation practices may not sufficiently account for the lack of such interactions. As a result, the available genomes of such populations would be limited to sequences from metagenomic surveys, which are often incomplete and/or composite and are therefore typically not included in efforts to generate models and nonredundant sequence databases for gene annotation (<xref ref-type="bibr" rid="bib9">Aramaki et al., 2020</xref>; <xref ref-type="bibr" rid="bib58">Galperin et al., 2015</xref>; <xref ref-type="bibr" rid="bib200">Sonnhammer et al., 1997</xref>). Thus, the reduced proportion of annotated genes in healthy metagenomes may reflect missing annotations due to lack of sufficiently homologous sequences in state-of-the-art databases. The true reduction in metabolic potential in the healthy sample group may not be as extensive as we have observed in this study.</p><p>The discrepancy in annotation efficiency between the healthy and IBD groups disappeared when analyzing all 2893 samples (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4</xref>). This suggests that the observed annotation bias does not strongly affect microbial populations that are readily assembled via shallow sequencing – likely, these are populations of high relative abundance in both healthy and IBD samples. Populations of lower abundance, which are less likely to be assembled from shallow metagenomes due to lack of sufficient coverage, are probably also less well characterized as a result. For this to contribute to fewer annotations per gene in healthy samples would necessitate that healthy samples contain relatively more low-abundance populations than IBD samples. Indeed, this is the case: healthy samples contain an average of 86 detected genomes from our set of GTDB gut microbes, and those genomes have a low average percent abundance of 0.61% across these samples. Non-IBD samples are similar, having an average of 77 detected genomes per sample with an average percent abundance of 0.79%. IBD samples, meanwhile, contain 30 detected genomes on average, with a higher average percent abundance of 2.24%. Therefore, the lack of characterization of low-abundance populations may be another factor that contributes to the relative reduction in gene annotations in the healthy samples.</p><p>To understand the potential origins of the reduced annotation rate in healthy metagenomes, we ran AGNOSTOS (<xref ref-type="bibr" rid="bib216">Vanni et al., 2022</xref>) to classify known and unknown genes within the healthy and IBD sample groups. AGNOSTOS clusters genes to contextualize them within an extensive reference dataset and then categorizes each gene as ‘known’ (has homology to genes annotated with Pfam domains of known function), ‘genomic unknown’ (has homology to genes in genomic reference databases that do not have known functional domains), or ‘environmental unknown’ (has homology to genes from metagenomes or MAGs that do not have known functional domains). The resulting classifications confirm that healthy metagenomes contain fewer ‘known’ genes than metagenomes in the IBD sample group – the proportion of ‘known’ genes classified by AGNOSTOS is about 3.0% less in the healthy metagenomes than in the IBD sample group, which is similar to the ~3.5% decrease in the proportion of ‘unannotated’ genes observed by simply counting the number of genes with at least one functional annotation (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4G and H</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1e</xref>). Furthermore, the majority of the unannotated genes in either sample group were categorized by AGNOSTOS as ‘genomic unknown’ (<xref ref-type="fig" rid="app1fig4">Appendix 1—figure 4G</xref>), suggesting that the unannotated sequences are genes without biochemically characterized functions currently associated with them and are thus legitimately lacking a functional annotation in our analysis, rather than representing distant homologs of known protein families that we failed to annotate. Based upon the classifications, a systematic technical bias is unlikely driving the annotation discrepancy between the sample groups.</p><p>Our observations into the annotation efficiency of microbial genes in different contexts, and our speculations regarding the sources of such bias and its implications warrant further investigations.</p></sec><sec sec-type="appendix" id="s10"><title>‘Non-IBD’ samples are intermediate to IBD and healthy samples</title><p>While so far we divided samples into two groups, our dataset also includes individuals who do not suffer from IBD, yet are not healthy either. A recent study using flux balance analysis to model metabolite secretion potential in the dysbiotic, non-dysbiotic, and control gut communities of Crohn’s disease patients has shown that several predicted microbial metabolic activities align with gradients of host health (<xref ref-type="bibr" rid="bib71">Heinken et al., 2021</xref>). To test whether the HMI signal captures gradients in host health, we included the ‘non-IBD’ group of patients that suffer from GI conditions other than IBD in our analysis. The set of 78 samples classified as ‘non-IBD’ indeed represent an intermediate group between healthy individuals and those diagnosed with IBD (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5B</xref>). While the HMI signal was reduced in ‘non-IBD’ patients, 75% of the pathways enriched in IBD patients were also enriched in the ‘non-IBD’ group compared to healthy individuals. Similarly, when sorting each individual cohort along a health gradient based on cohort descriptions in their respective studies (see ‘Characterizing cohort-specific metabolic capacity across the gradient of health and disease’), the relative proportion of metabolic pathways indicative of HMI increased as a function of increasing disease severity (<xref ref-type="fig" rid="app1fig6">Appendix 1—figure 6A</xref>). These findings suggest that the HMI signal is sufficiently sensitive to resolve gradients in host health and could serve as a diagnostic tool to monitor changing stress levels in a single individual over time.</p><fig id="app1fig5" position="float"><label>Appendix 1—figure 5.</label><caption><title>Additional boxplots of median per-population copy number for various subsets of metabolic pathways and metagenome samples.</title><p>(<bold>A</bold>) 33 modules enriched in high metabolic independence (HMI) populations from <xref ref-type="bibr" rid="bib220">Watson et al., 2023</xref>, compared to the 33 inflammatory bowel disease (IBD)-enriched modules from this study, with medians computed in the set of deeply sequenced healthy (n = 229) and IBD (n = 101) samples. (<bold>B</bold>) The 33 IBD-enriched modules from this study, with medians computed in the set of deeply sequenced healthy (n = 229), non-IBD (n = 78), and IBD (n = 101) samples. (<bold>C</bold>) All KEGG modules (n = 117) with nonzero copy number in at least one sample, with medians computed in the set of deeply sequenced healthy (n = 229) and IBD (n = 101) samples. (<bold>D</bold>) All biosynthesis modules (n = 88) from the KEGG MODULE database, with medians computed in the set of deeply sequenced healthy (n = 229) and IBD (n = 101) samples. Where applicable, dashed lines indicate the overall median for all modules, and solid lines connect the points for the same module in each sample group. The IBD sample group is highlighted in red, the NON-IBD group in pink, and the HEALTHY group in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig5-v1.tif"/></fig></sec><sec sec-type="appendix" id="s11"><title>Module enrichment without consideration of effect size leads to nonspecific results</title><p>Our analyses indicate that the majority of KEGG modules had higher PPCN in IBD metagenomes (<xref ref-type="fig" rid="fig2">Figure 2E</xref>, <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2b</xref>). Indeed, when we examine all modules with nonzero median PPCN in at least one group of samples (n = 117), their median normalized copy number is systematically higher in the IBD group than in the healthy group (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5C</xref>; 98 out of 117 modules have a higher normalized copy number in the IBD group than in the healthy group at 5% FDR-adjusted significance level using a one-sided Wilcoxon test). This result is likely a natural outcome of the differential distribution of HMI and non-HMI genomes in the two sample groups, as seen in our analysis of reference genomes (<xref ref-type="fig" rid="fig3">Figure 3B and C</xref>), where the overrepresentation of HMI populations with larger genomes that encode many more complete pathways in IBD samples leads to higher PPCNs computed at the metagenome level. The consistent elevation in PPCN of metabolic modules in IBD could also be attributed, at least in part, to the aforementioned functional annotation biases that seem to disproportionally affect the characterization of healthy metagenomes. The lower annotation efficiency in healthy metagenomes could result in partial copies of pathways, which are ignored by our stringent copy number calculation that only counts complete copies. Therefore, to narrow down our results and identify which pathways are <italic>particularly</italic> important for microbial resilience in the IBD gut environment, we considered only those pathways with the largest difference in normalized copy number (‘effect size’) between the two groups to identify metabolic modules that are truly elevated in IBD metagenomes (see Methods).</p><p>With similar considerations we also investigated whether biosynthetic capacity in general was enriched in IBD samples. For this, we expanded our analysis to also consider biosynthesis pathways that did not meet the enrichment criteria we have used for inclusion in the final set of 33 IBD-enriched modules. As expected, we found that the majority of all biosynthesis pathways in the KEGG MODULE database (n = 88) have significantly higher normalized copy numbers in IBD samples (<xref ref-type="fig" rid="app1fig5">Appendix 1—figure 5D</xref>; at a 5% FDR-adjusted significance level, 62 out of 88 (70%) biosynthesis pathways have a higher normalized copy number using a one-sided Wilcoxon test). This analysis also showed a similar increase for non-biosynthetic pathways: 63 out of 91 (69%) non-biosynthetic pathways showed significant increase in IBD samples (two-sample test for equality of proportion: 0.88). Overall, these data indicate that without the consideration of effect size, both biosynthetic and non-biosynthetic capacity appear to be increased in the IBD gut microbiome. In contrast, maintenance of a higher metabolic capacity for the biosynthesis of essential nutrients emerges as an important factor for microbial resilience in IBD through a strict enrichment criteria in addition to statistical significance scores calculated for differential abundance.</p></sec><sec sec-type="appendix" id="s12"><title>A review of HMI-associated modules in the context of gut microbiome literature</title><p>The 33 pathways that are enriched in microbial communities associated with individuals with IBD likely provide competencies that are critical for survival in the stressed gut environment. In this section, we offer a review of IBD-enriched modules with existing gut microbiome scientific literature.</p><sec sec-type="appendix" id="s12-1"><title>Amino acid pathways</title><p>Of the eight proteinogenic amino acids that can be synthesized with IBD-enriched pathways, <italic>leucine</italic>, <italic>tryptophan</italic>, <italic>threonine</italic>, <italic>isoleucine</italic>, and <italic>methionine</italic> are essential amino acids for humans (<xref ref-type="bibr" rid="bib127">Lopez and Mohiuddin, 2023</xref>), while <italic>cysteine</italic> and <italic>arginine</italic> are semi-essential (<xref ref-type="bibr" rid="bib174">Rehman et al., 2020</xref>; <xref ref-type="bibr" rid="bib208">Tong and Barbul, 2004</xref>). Furthermore, leucine, tryptophan, isoleucine, and cysteine can reduce oxidative stress for intestinal epithelial cells (<xref ref-type="bibr" rid="bib97">Katayama and Mine, 2007</xref>), which may be beneficial given the increased GI oxygen levels associated with IBD (<xref ref-type="bibr" rid="bib177">Rigottier-Gois, 2013</xref>). Several of these amino acids have been analyzed for their potential therapeutic effects in IBD (<xref ref-type="bibr" rid="bib124">Liu et al., 2017</xref>). Regardless, it is unknown if the depleted gut microbiome in IBD would produce these amino acids in sufficient quantity to promote health benefits to the host, especially considering that the microbes themselves require these molecules for protein production and as energy sources – for instance, in proteolytic fermentation (<xref ref-type="bibr" rid="bib228">Wu et al., 2021</xref>; <xref ref-type="bibr" rid="bib120">Lin et al., 2017</xref>).</p><p>The <italic>Shikimate pathway</italic>, which converts phosphoenolpyruvate and erythrose 4-phosphate (E4P) to chorismate, is a prerequisite for <italic>tryptophan biosynthesis</italic>. This pathway is only present in microorganisms and plants, and it also produces intermediates for other metabolic pathways such as quinate degradation and antibiotic synthesis (<xref ref-type="bibr" rid="bib75">Herrmann and Weaver, 1999</xref>). A recent analysis of paired fecal metagenomes and metatranscriptomes from the HMP using reference genome-based functional inference demonstrated that the Shikimate pathway is typically incomplete in gut microbes and only transcriptionally active in a few, suggesting that most gut microbes are auxotrophic for aromatic amino acids and therefore rely on dietary sources and potentially cross-feeding to obtain these molecules or their precursors (<xref ref-type="bibr" rid="bib140">Mesnage and Antoniou, 2020</xref>). This offers a potential explanation for the enrichment of the Shikimate and tryptophan biosynthesis pathways in the IBD gut microbiome, where a depleted community may restrict the availability of cross-fed metabolites. Indeed, a tryptophan-deficient diet alters the composition of the gut microbiota in aged mice (<xref ref-type="bibr" rid="bib233">Yusufu et al., 2021</xref>), providing auxiliary evidence that the loss of this amino acid impacts microbial survival. Furthermore, the serum levels of tryptophan are reduced in individuals with IBD due to high host metabolism rates (<xref ref-type="bibr" rid="bib150">Nikolaus et al., 2017</xref>), which may exacerbate the lack of bioavailable tryptophan for gut microbes. On the host side, tryptophan and its derivatives influence a number of physiological processes, though it is unclear how much the microbial production of tryptophan contributes to these effects (<xref ref-type="bibr" rid="bib2">Agus et al., 2018</xref>). Nevertheless, the lack of tryptophan appears to worsen intestinal inflammation while supplementation can attenuate it (<xref ref-type="bibr" rid="bib102">Kim et al., 2010</xref>; <xref ref-type="bibr" rid="bib70">Hashimoto et al., 2012</xref>).</p><p><italic>Cysteine biosynthesis</italic> was previously found to be enriched in the IBD gut microbiome based on reference genome analysis of 16S ribosomal RNA gene amplicons (<xref ref-type="bibr" rid="bib146">Morgan et al., 2012</xref>), in which the authors propose that cysteine metabolism could be important to microbial management of oxidative stress via the production of glutathione, which is protective against reactive oxygen species (<xref ref-type="bibr" rid="bib195">Sherrill and Fahey, 1998</xref>; <xref ref-type="bibr" rid="bib207">Tepe et al., 2006</xref>), from cysteine and glutamate. Cysteine can also be converted into hydrogen sulfide (H<sub>2</sub>S) by host colonocytes and some intestinal microbes. Though H<sub>2</sub>S produced by colonocytes can help support their energy production, excess microbially derived H<sub>2</sub>S in the lumen is a risk factor for gut mucosal inflammation and H<sub>2</sub>S may play a role in colorectal carcinogenesis (<xref ref-type="bibr" rid="bib18">Blachier et al., 2019</xref>). Interestingly, cysteine biosynthesis is also enriched in the gut microbiomes of postmenopausal women, where it is thought to contribute to elevated homocysteine levels and therefore to increased risk of cardiovascular disease (<xref ref-type="bibr" rid="bib235">Zhao et al., 2019</xref>).</p><p><italic>Leucine</italic> and <italic>isoleucine</italic>, as branched-chain amino acids (BCAAs), are important nutrients and signaling molecules in humans (<xref ref-type="bibr" rid="bib64">Gojda and Cahova, 2021</xref>). Gut microbial synthesis of these compounds does contribute to human BCAA pools, as evidenced by experiments with heavy isotope labeling and correlations between serum and fecal BCAA levels (<xref ref-type="bibr" rid="bib141">Metges et al., 1999</xref>; <xref ref-type="bibr" rid="bib42">Dhakan et al., 2019</xref>). The extent of this exchange has not been characterized in individuals with IBD. However, a study of individuals receiving anti-integrin therapy for Crohn’s disease demonstrated that pathways for biosynthesis of L-isoleucine and arginine were enriched at baseline in the gut microbiomes of responders to the therapy (<xref ref-type="bibr" rid="bib8">Ananthakrishnan et al., 2017</xref>). In a longitudinal study of mice, biosynthesis pathways for leucine and proline were more abundant in animals modeling IBD (<xref ref-type="bibr" rid="bib193">Sharpton et al., 2017</xref>).</p><p><italic>Threonine</italic> and <italic>proline</italic> are both important components of intestinal mucins (<xref ref-type="bibr" rid="bib88">Johansson and Hansson, 2016</xref>; <xref ref-type="bibr" rid="bib52">Faure et al., 2005</xref>) and thus contribute to mucosal barrier integrity, which is typically impaired in IBD (<xref ref-type="bibr" rid="bib87">Johansson et al., 2010</xref>). For instance, threonine, proline, and cysteine supplementation has been shown to reduce symptoms and restore lactobacilli and bifidobacteria counts in rats with DSS-induced inflammation (<xref ref-type="bibr" rid="bib203">Sprong et al., 2010</xref>; <xref ref-type="bibr" rid="bib53">Faure et al., 2006</xref>). The latter observation suggests the importance of an external source of these three amino acids to the fitness of the lactobacilli and bifidobacterial populations and thereby supports the idea that they are community metabolites.</p><p>We also found <italic>methionine</italic> biosynthesis to be enriched in the IBD gut microbiome. In individuals with quiescent IBD, reduced serum levels of methionine, proline, and tryptophan are correlated with changes in the gut microbiome that are associated with increased symptoms of fatigue (<xref ref-type="bibr" rid="bib19">Borren et al., 2021</xref>), demonstrating a putative link between methionine bioavailability, microbial abundances, and host wellbeing. Indeed, L-methionine supplementation in piglets results in improved mucosal integrity and villus architecture (<xref ref-type="bibr" rid="bib30">Chen et al., 2014</xref>), and the activated form of methionine, <italic>S</italic>-adenosylmethionine, can reverse colon lesions and cytoskeletal damage in intestinal cells in DSS-treated mice (<xref ref-type="bibr" rid="bib155">Oz et al., 2005</xref>). Yet, reducing methionine in high-fat diets given to mice was shown to improve intestinal barrier function, reduce inflammation, and increase the abundance of short-chain fatty acid (SCFA)-producing microbes (<xref ref-type="bibr" rid="bib231">Yang et al., 2019</xref>), so the net impact of methionine on host health and microbial fitness remains unclear.</p><p><italic>Arginine</italic> has been well studied in the context of IBD. It has been shown to reduce cytokine production, promote intestinal healing, and improve intestinal barrier function in DSS-treated mice, perhaps by enhancing production of nitric oxide (NO) (<xref ref-type="bibr" rid="bib34">Coburn et al., 2012</xref>; <xref ref-type="bibr" rid="bib63">Gobert et al., 2004</xref>; <xref ref-type="bibr" rid="bib196">Singh et al., 2019</xref>). NO is a free radical that has been implicated in regulating mucosal barrier integrity, GI motility, and protection against oxidative stress, though overproduction of this compound can have detrimental effects (<xref ref-type="bibr" rid="bib106">Kolios et al., 2004</xref>; <xref ref-type="bibr" rid="bib219">Walker et al., 2018</xref>). Biosynthesis of <italic>ornithine</italic>, which is both a precursor and a derivative of arginine, was enriched in the IBD gut microbiome as well in agreement with another study that reported an increase in ornithine biosynthesis in the gut microbiome of individuals with active UC (<xref ref-type="bibr" rid="bib72">Hellmann et al., 2023</xref>). Finally, <italic>polyamines</italic> – which are derived from arginine and were also represented in the enriched pathways – promote intestinal barrier function <xref ref-type="bibr" rid="bib122">Liu et al., 2009</xref>; for instance, by regulating the growth of intestinal epithelial cells (<xref ref-type="bibr" rid="bib138">McCormack and Johnson, 1991</xref>).</p></sec><sec sec-type="appendix" id="s12-2"><title>Carbohydrate pathways</title><p>Three KEGG modules describing the <italic>pentose phosphate pathway</italic> (PPP) were enriched in IBD samples – the entire pentose phosphate cycle (M00004), the oxidative phase (M00006), and the non-oxidative phase (M00007). We removed the oxidative phase (M00006) from our set of IBD-enriched modules because it was an exact copy of the initial steps in M00004; however, we kept the non-oxidative phase (M00007) in our set because it is defined using slightly different enzymes than the non-oxidative portion of M00004. M00007 is defined in four steps and utilizes a ribulose-phosphate 3-epimerase and a ribose 5-phosphate isomerase in the last two steps, while the non-oxidative phase in M00004 is defined in three steps and utilizes a glucose-6-phosphate isomerase in the last step. The PPP is a ubiquitous pathway in most bacteria and eukaryotes, as it plays a central role in cellular metabolism. It produces the important cellular intermediates ribose 5-phosphate and E4P, which are used for synthesis of nucleotides and aromatic amino acids, respectively (<xref ref-type="bibr" rid="bib199">Soderberg, 2005</xref>). In fact, E4P is one of the inputs to the Shikimate pathway, another IBD-enriched module discussed above. The PPP also produces NADPH, a reducing equivalent important for reductive reactions and prevention of oxidative stress (<xref ref-type="bibr" rid="bib111">Kruger and von Schaewen, 2003</xref>; <xref ref-type="bibr" rid="bib31">Christodoulou et al., 2018</xref>). Beyond its link to other enriched amino acid biosynthesis pathways, it is unusual that such a central pathway would have an increased copy number in the IBD gut microbiome rather than being equally distributed across all samples. Some gut microbes are known to lack the transaldolase gene in this pathway and may instead encode an alternative pathway for pentose degradation called the sedoheptulose 1,7-bisphosphate pathway (SBPP) (<xref ref-type="bibr" rid="bib60">Garschagen et al., 2021</xref>); it is therefore possible that the enrichment of the more common PPP in IBD is related to an increased ratio of microbial populations that use the PPP rather than the SBPP in the less-diverse microbiome of IBD patients, though this requires further investigation to verify.</p><p>The first carbon oxidation of the <italic>citric acid cycle</italic> (TCA cycle), which is a three-step conversion from oxaloacetate to 2-oxoglutarate (alpha-ketoglutarate), is enriched in the IBD samples. Similar to the PPP, the citric acid cycle is a central metabolic pathway, especially with regard to generation of energy and key metabolites for other pathways (<xref ref-type="bibr" rid="bib4">Akram, 2014</xref>). It is unclear why only this particular portion of the cycle would be enriched, though this could perhaps be attributed to the role of alpha-ketoglutarate in the production of glutamate, the precursor to proline, ornithine, and arginine (three amino acids with enriched biosynthesis pathways in the IBD sample group, as discussed above). It has been said that 2-oxoglutarate is the most fundamental compound of this cycle, serving as the link between carbon and nitrogen metabolism and also as a critical element in the recovery of amine groups for amino acid and protein production (<xref ref-type="bibr" rid="bib164">Pierzynowski and Pierzynowska, 2022</xref>; <xref ref-type="bibr" rid="bib79">Huergo and Dixon, 2015</xref>). Thus, the enrichment of 2-oxoglutarate production capacity in the IBD gut environment could be related to the enrichment of amino acid biosynthesis pathways.</p><p>Two nucleotide sugar biosynthesis pathways are enriched in the IBD gut microbiome. One of these is <italic>synthesis of UDP-glucose</italic>, which is an important molecule implicated in a variety of key cellular metabolisms. It is an intermediate in polysaccharide biosynthesis and pyrimidine metabolism, a precursor of lipopolysaccharides in the outer cell membrane of Gram-negative bacteria, and an extracellular signaling molecule (<xref ref-type="bibr" rid="bib170">Ralevic, 2015</xref>). Additionally, as an agonist for P2Y-14 receptors, it could play a role in modulating host GI functions like muscular contraction (<xref ref-type="bibr" rid="bib14">Bassil et al., 2009</xref>), and in modulating host inflammatory responses by activating this receptor specifically in T-lymphocytes (<xref ref-type="bibr" rid="bib185">Scrivens and Dickenson, 2005</xref>) and in immature monocyte-derived dendritic cells (<xref ref-type="bibr" rid="bib198">Skelton et al., 2003</xref>). The other enriched nucleotide sugar pathway is <italic>UDP-GlcNAc biosynthesis</italic>. Flux through this pathway is linked to a multitude of other central metabolisms, including amino acid and fatty acid metabolism (<xref ref-type="bibr" rid="bib69">Hardivillé and Hart, 2014</xref>). Furthermore, UDP-GlcNAc is an important substrate in protein glycosylation pathways (<xref ref-type="bibr" rid="bib69">Hardivillé and Hart, 2014</xref>; <xref ref-type="bibr" rid="bib180">Ryczko et al., 2016</xref>), and a precursor to critical cell wall components in bacteria (<xref ref-type="bibr" rid="bib123">Liu and Breukink, 2016</xref>; <xref ref-type="bibr" rid="bib142">Mikkola, 2020</xref>; <xref ref-type="bibr" rid="bib215">van Dam et al., 2009</xref>). In the gut, this molecule has been implicated in regulation of nutrient uptake by the host (<xref ref-type="bibr" rid="bib180">Ryczko et al., 2016</xref>).</p><p><italic>D-Glucuronate (glucuronic acid) degradation</italic> into pyruvate and D-glyceraldehyde 3-phosphate is also enriched in the IBD gut microbiome. Some gut microbes are capable of growth on host-derived uronic acids (<xref ref-type="bibr" rid="bib128">Lopez-Siles et al., 2012</xref>), so this pathway may serve as a source of energy to microbes living in the IBD gut environment. In mice, there is evidence that derivatives of glycosaminoglycan degradation such as D-glucuronate can worsen colitis (<xref ref-type="bibr" rid="bib114">Lee et al., 2009</xref>).</p><p>Finally, the <italic>phosphoribosyl diphosphate (PRPP) biosynthesis pathway</italic> is important because PRPP is used in the formation of glycosidic bonds as well as in the biosynthesis of a number of cofactors, amino acids, and nucleotides (<xref ref-type="bibr" rid="bib78">Hove-Jensen et al., 2017</xref>). It is discussed further below in the context of nucleotide metabolism.</p></sec><sec sec-type="appendix" id="s12-3"><title>Cofactor and vitamin pathways</title><p>Biosynthesis or salvage pathways for the following five cofactors and vitamins are enriched in IBD: <italic>heme</italic>, <italic>siroheme</italic>, <italic>thiamine (vitamin B1)</italic>, <italic>cobalamin (vitamin B12)</italic>, and <italic>coenzyme A (CoA)</italic>. <italic>Heme</italic> is required for aerobic respiration (<xref ref-type="bibr" rid="bib66">Gruss et al., 2012</xref>) and the increase in this pathway may be related to elevated oxygen levels in the gut as a result of inflammation, which promotes the growth of aerotolerant microbes (<xref ref-type="bibr" rid="bib190">Shah, 2016</xref>; <xref ref-type="bibr" rid="bib28">Cevallos et al., 2019</xref>). Dietary heme has also been associated with gut dysbiosis, aggravated colitis, and increased cytotoxicity in the colon (<xref ref-type="bibr" rid="bib35">Constante et al., 2017</xref>; <xref ref-type="bibr" rid="bib83">Ijssennagger et al., 2015</xref>); and genes related to heme and siroheme biosynthesis have also been found with high abundance in infants with neonatal necrotizing enterocolitis (<xref ref-type="bibr" rid="bib32">Claud et al., 2013</xref>).</p><p>Both <italic>thiamine</italic> and <italic>cobalamin</italic> are important cofactors that are commonly shared between gut microbes (<xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>), suggesting that microbes incapable of synthesizing these cofactors are unlikely to thrive in the low-diversity microbial communities of the IBD gut environment. Neither of these vitamins is produced by host cells but they are typically acquired from dietary sources (cobalamin, in particular, is absorbed in the small intestine) (<xref ref-type="bibr" rid="bib187">Seetharam and Alpers, 1982</xref>; <xref ref-type="bibr" rid="bib39">Degnan et al., 2014b</xref>; <xref ref-type="bibr" rid="bib77">Hossain et al., 2022</xref>), so the enrichment of these pathways is unlikely to have a large impact on host health.</p><p><italic>Coenzyme A</italic> can be produced from pantothenate (vitamin B5) by most gut microbes (<xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>) and its biosynthesis has been described as ‘essential’ considering that CoA is required for a large number of enzymatic reactions (<xref ref-type="bibr" rid="bib204">Spry et al., 2008</xref>; <xref ref-type="bibr" rid="bib116">Leonardi et al., 2005</xref>). It is therefore interesting that this pathway appears to be enriched in the IBD gut microbiome, which implies a relative deficiency of CoA biosynthesis in the healthy gut microbiome. It is possible that the module is spuriously enriched, despite its low p-value of 5.7e-21, given the short length of this pathway – it has three major steps when the KEGG module definition is interpreted in a ‘stepwise’ fashion by anvi-estimate-metabolism, though there are in fact five chemical conversions (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2a</xref>). An alternative possibility is that the KO HMMs for the required enzymes do not sufficiently represent the diversity of these proteins across the gut microbiota, which could cause this pathway to be undercounted due to lack of proper annotations.</p></sec><sec sec-type="appendix" id="s12-4"><title>Nucleotide pathways</title><p>The IBD-enriched modules include pathways for synthesis of the first complete <italic>purine</italic>, <italic>inosine monophosphate</italic>, as well as a series of <italic>pyrimidine biosynthesis pathways</italic> encoding the conversion from uridine monophosphate to ribonucleotides (UDP/UTP, CDP/CTP) and finally to the cytosine deoxyribonucleotide (dCTP). The <italic>PRPP</italic> <italic>biosynthesis pathway</italic> is also included in this list; though it is classified as central carbohydrate metabolism in KEGG due to its role in glycosidic bond formation, this molecule is an important precursor for nucleotide biosynthesis (both purines and pyrimidines) and synthesis of the amino acids tryptophan and histidine (<xref ref-type="bibr" rid="bib78">Hove-Jensen et al., 2017</xref>). Though many microbes are capable of producing their own nucleotides, some – especially lactic acid bacteria – are not and rely on uptake of exogenous nucleosides and bases, which are converted to nucleotides via salvage pathways (<xref ref-type="bibr" rid="bib153">Nygaard, 2014</xref>; <xref ref-type="bibr" rid="bib101">Kilstrup et al., 2005</xref>). Notably, these salvage pathways are not enriched in the IBD gut microbiome, suggesting that self-sufficiency in nucleotide biosynthesis (especially in the early stages in this process) is selected for in these communities. This also implies the importance of pyrimidine and purine cross-feeding in the healthy gut environment, which is supported by evidence that some gut microbes (e.g. <italic>Bacteroides vulgatus</italic>) actively secrete nucleosides in the colon (<xref ref-type="bibr" rid="bib225">Wong et al., 2023</xref>; <xref ref-type="bibr" rid="bib206">Teng et al., 2023</xref>).</p></sec><sec sec-type="appendix" id="s12-5"><title>Lipid pathways</title><p>Two lipid biosynthesis pathways – <italic>initiation and elongation of fatty acids</italic> – are enriched in IBD. Fatty acids are essential components of cell membranes and also serve as signaling molecules (<xref ref-type="bibr" rid="bib22">Brown et al., 2023</xref>); thus, the ability to synthesize them is an important fitness determinant. For example, gut <italic>Bacteroides</italic> species that are deficient in sphingolipid production capabilities are much less resilient to oxidative stress than wild-type species (<xref ref-type="bibr" rid="bib7">An et al., 2011</xref>). Since oxidative stress is a hallmark of IBD, it is possible that this environment selects for microbes capable of fatty acid biosynthesis.</p></sec><sec sec-type="appendix" id="s12-6"><title>Energy pathways</title><p>The <italic>Pta-Ack pathway</italic> is important for microbial energy production and adaptation to different growth conditions via the ‘acetate switch’, which enables either production or consumption of acetate depending on available nutrients (<xref ref-type="bibr" rid="bib224">Wolfe, 2005</xref>). SCFAs such as acetate serve as important energy sources to intestinal epithelial cells. They also play a role in regulating gut barrier function and host immune responses (<xref ref-type="bibr" rid="bib136">Martin-Gallausiaux et al., 2021</xref>; <xref ref-type="bibr" rid="bib234">Zhang et al., 2022</xref>), and impaired absorption and oxidation of SCFAs can contribute to the development of IBD (<xref ref-type="bibr" rid="bib234">Zhang et al., 2022</xref>). Acetate promotes host intestinal IgA production and thereby has a protective effect against gut inflammation (<xref ref-type="bibr" rid="bib227">Wu et al., 2017</xref>), but acetate levels are reduced in children with IBD (<xref ref-type="bibr" rid="bib209">Treem et al., 1994</xref>). Further study is required to determine the flux direction of the Pta-Ack pathway and whether it contributes to the reduction of acetate in the IBD gut environment.</p><p><italic>CAM metabolism</italic> is categorized as a carbon fixation pathway in the KEGG MODULE database, yet is a short (two-step) pathway utilizing enzymes required in other common metabolisms. Its first step is catalyzed by PEP carboxylase, an enzyme that is involved in gluconeogenesis, serine biosynthesis, and carbon skeleton conversions in the citric acid cycle (<xref ref-type="bibr" rid="bib230">Yang et al., 2009</xref>). Its second step is catalyzed by malate dehydrogenases, a ubiquitous class of enzymes that convert 2-hydroxy acids to 2-keto acids and are involved in gluconeogenesis, the TCA cycle, glyoxylate bypass, and amino acid synthesis (<xref ref-type="bibr" rid="bib143">Minárik et al., 2002</xref>; <xref ref-type="bibr" rid="bib148">Musrati et al., 1998</xref>). The increase in this pathway in IBD gut microbiomes could be attributed in part to the increase in aerobic respiration due to elevated oxygen levels (<xref ref-type="bibr" rid="bib190">Shah, 2016</xref>; <xref ref-type="bibr" rid="bib28">Cevallos et al., 2019</xref>) and in part to the increase in amino acid biosynthesis capacity as evidenced by the multiple amino acid pathways that are also enriched.</p></sec><sec sec-type="appendix" id="s12-7"><title>Drug resistance pathways</title><p>The use of antibiotics to treat IBD and its complications is known to increase antibiotic resistance in the gut microbiome (<xref ref-type="bibr" rid="bib152">Nitzan et al., 2016</xref>; <xref ref-type="bibr" rid="bib113">Ledder, 2019</xref>) and several studies have noted that individuals exposed to antibiotics are more likely to develop IBD (<xref ref-type="bibr" rid="bib110">Kronman et al., 2012</xref>; <xref ref-type="bibr" rid="bib211">Ungaro et al., 2014</xref>; <xref ref-type="bibr" rid="bib113">Ledder, 2019</xref>; <xref ref-type="bibr" rid="bib194">Shaw et al., 2011</xref>). This potentially explains the enrichment of two drug resistance pathways in the IBD microbiome: <italic>efflux pump MepA</italic> (conferring multidrug resistance) and the <italic>bla system</italic> (conferring beta-lactam resistance), as higher rates of antibiotic exposure in this sample group naturally leads to selection for resistance phenotypes (<xref ref-type="bibr" rid="bib118">Levy, 2000</xref>; <xref ref-type="bibr" rid="bib5">Alekshun and Levy, 2007</xref>). Beta-lactamases in particular have been found with higher frequency in people with IBD (<xref ref-type="bibr" rid="bib217">Vich Vila et al., 2018</xref>; <xref ref-type="bibr" rid="bib117">Leung et al., 2012</xref>; <xref ref-type="bibr" rid="bib214">Vaisman et al., 2013</xref>). Increased microbial drug resistance can heighten the risk of a severe infection such as <italic>Clostridium difficile</italic> infection (CDI) (<xref ref-type="bibr" rid="bib125">Llor and Bjerrum, 2014</xref>). CDI already occurs with higher frequency in individuals with IBD (<xref ref-type="bibr" rid="bib86">Jodorkovsky et al., 2010</xref>), though the higher incidence of CDI is not necessarily linked to chronic antibiotic use in these individuals (at least in one retrospective study of Crohn’s disease) (<xref ref-type="bibr" rid="bib179">Roy and Lichtiger, 2016</xref>). Regardless, antibiotic resistance is a global health problem that affects everyone, not just those with IBD.</p></sec><sec sec-type="appendix" id="s12-8"><title>Most enriched pathways in HMI reference genomes</title><p>The three HMI-associated pathways with the largest difference in average completion (&gt;40%) between HMI and non-HMI reference genomes were <italic>siroheme biosynthesis</italic>, <italic>cobalamin biosynthesis</italic>, and <italic>tryptophan biosynthesis</italic> (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3g</xref>). Siroheme and cobalamin biosynthesis represent complex pathways that require 6–8 and 11–13 enzymatic steps, respectively, and both compounds belong to the tetrapyrroles that are involved in various essential biological functions (<xref ref-type="bibr" rid="bib23">Bryant et al., 2020</xref>). Siroheme is a cofactor required for nitrite and sulfite reduction and its biosynthetic pathway provides the precursors required for cobalamin biosynthesis. Genes belonging to biosynthetic pathways of siroheme and cobalamin had higher average relative abundance in infants diagnosed with neonatal necrotizing enterocolitis (<xref ref-type="bibr" rid="bib32">Claud et al., 2013</xref>), an inflammatory bowel condition affecting premature newborns. The siroheme biosynthesis pathway is upregulated in some human pathogens in response to high NO levels likely in relation to the NO detoxification function of nitrite reductase (<xref ref-type="bibr" rid="bib165">Porrini et al., 2021</xref>). Increased NO levels are commonly associated with active inflammation in IBD (<xref ref-type="bibr" rid="bib202">Soufli et al., 2016</xref>).</p><p>While siroheme is central to sulfite and nitrite reduction in prokaryotes, <italic>cobalamin</italic> (vitamin B12) is essential not only for the majority of gut microbes (~80%) (<xref ref-type="bibr" rid="bib98">Kelly et al., 2019</xref>; <xref ref-type="bibr" rid="bib77">Hossain et al., 2022</xref>; <xref ref-type="bibr" rid="bib38">Degnan et al., 2014a</xref>) but also for the human host, and functions as a coenzyme in key metabolic pathways in humans and bacteria. However, only relatively few gut microbes (~20–40%) encode the metabolic pathway for its synthesis (<xref ref-type="bibr" rid="bib39">Degnan et al., 2014b</xref>; <xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>; <xref ref-type="bibr" rid="bib98">Kelly et al., 2019</xref>) and humans largely rely on cobalamin supplied via their diet. B12 deficiency in humans leads to reduced villi length (<xref ref-type="bibr" rid="bib17">Berg et al., 1972</xref>) and may affect intestinal barrier functioning (<xref ref-type="bibr" rid="bib21">Bressenot et al., 2013</xref>). However, microbially produced cobalamin alone is insufficient to sustain the host’s requirements (<xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>). The high average completion of this complex pathway in reference genomes classified as HMI (86%) in contrast to non-HMI reference genomes (40%) demonstrates the importance of metabolic independence for the survival of microorganisms in stressed gut environments, whereas in a healthy gut environment cross-feeding of B-vitamins likely supports those microbes that do not have metabolic means to synthesize them (<xref ref-type="bibr" rid="bib132">Magnúsdóttir et al., 2015</xref>).</p><p><italic>Tryptophan</italic> is an essential amino acid that serves as a precursor for a variety of microbial (<xref ref-type="bibr" rid="bib6">Alkhalaf and Ryan, 2015</xref>) and human metabolites that play a potential role in IBD pathogenesis (<xref ref-type="bibr" rid="bib2">Agus et al., 2018</xref>). Tryptophan metabolites mediate a variety of host microbe interactions in the human gut (<xref ref-type="bibr" rid="bib2">Agus et al., 2018</xref>), contribute to gut barrier integrity, and exert anti-inflammatory functions (<xref ref-type="bibr" rid="bib13">Bansal et al., 2010</xref>; <xref ref-type="bibr" rid="bib178">Roager and Licht, 2018</xref>). While fecal tryptophan concentrations can be elevated in IBD patients (<xref ref-type="bibr" rid="bib85">Jansson et al., 2009</xref>), tryptophan host metabolism via the Kynurenine pathway also appears to be elevated in disease, resulting in decreased serum levels of the amino acid (<xref ref-type="bibr" rid="bib150">Nikolaus et al., 2017</xref>). At the same time, a tryptophan-deficient diet in mice is linked to intestinal inflammation and alterations of the microbial community composition (<xref ref-type="bibr" rid="bib70">Hashimoto et al., 2012</xref>; <xref ref-type="bibr" rid="bib233">Yusufu et al., 2021</xref>).</p><p>Overall, our data contains no evidence or indication for any direct links between the increased representation of microbial metabolic modules in IBD and the role of the products these metabolic activities yield in human disease states.</p></sec></sec><sec sec-type="appendix" id="s13"><title>Characterizing cohort-specific metabolic capacity across the gradient of health and disease</title><p>We sought to evaluate the cohort-specific trends in metabolic capacity by computing the median PPCN of the 33 IBD-enriched modules within each sample from each study. Considering the heterogeneity within each sample group, we ordered the studies from most healthy to least healthy, using the cohort description from each publication to approximate relative healthiness based on the number and types of exclusions listed for healthy or non-IBD controls, or on the diagnostic criteria for people with IBD (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1a</xref>).</p><p>The amount of detail provided as well as the GI conditions considered varied between studies. To overcome this challenge, we placed more emphasis on exclusionary conditions to sort samples based on host health status. For instance, <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib173">Raymond et al., 2016</xref> used the most stringent criteria in the selection of healthy individuals. Both studies excluded patients with GI-related conditions like disease, surgery, and medication; medications affecting the immune system; or antibiotics. Additionally, <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> also excluded individuals diagnosed with type 2 diabetes while <xref ref-type="bibr" rid="bib173">Raymond et al., 2016</xref> did not, and we assigned a higher ‘health score’ to samples classified as healthy by <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>. Similarly, <xref ref-type="bibr" rid="bib183">Schirmer et al., 2018</xref> applied more stringent exclusion criteria for samples classified as ‘non-IBD’ controls than <xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref>, and was therefore considered a healthier cohort within that group.</p><p>To arrange samples classified as ‘IBD’ along a gradient of host health status, we considered similar cohorts to be more unhealthy if their diagnosis was supported by several lines of evidence. For example, <xref ref-type="bibr" rid="bib183">Schirmer et al., 2018</xref> diagnosed IBD based on a screening colonoscopy and included existing patients with consistent diagnosis over the past 5+ years, <xref ref-type="bibr" rid="bib126">Lloyd-Price et al., 2019</xref> required a combination of endoscopic and histopathologic evidence for diagnosis, and <xref ref-type="bibr" rid="bib57">Franzosa et al., 2019</xref> only considered patients diagnosed via endoscopic, histopathologic, and radiographic approaches. Regardless, these cohorts are likely extremely similar in healthiness. Patients with the lowest health score were described by <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> – with a cohort composed of total proctocolectomy patients with ileal pouches, some of which developed pouchitis.</p><p>Ordering the per-sample median PPCN values along this gradient of cohort health indicates that the HMI metric for gut microbial metabolic capacity increases as host health decreases (<xref ref-type="fig" rid="app1fig6">Appendix 1—figure 6A</xref>). Therefore, HMI adequately captures the variability in gut environment conditions that challenge microbial survival.</p><fig id="app1fig6" position="float"><label>Appendix 1—figure 6.</label><caption><title>Boxplots of median per-population copy number of 33 inflammatory bowel disease (IBD)-enriched modules for samples from each individual cohort.</title><p>(<bold>A</bold>) with medians computed within each sample (i.e. one point per sample) and (<bold>B</bold>) with medians computed for each IBD-enriched module (i.e. one point per module). The x-axis indicates study of origin. (<bold>C</bold>) Boxplots of median per-population copy number of 33 IBD-enriched modules for the 115 samples in the deeply sequenced set that are not from <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>, or <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>. The dashed line indicates the overall median for all 33 modules, and solid lines connect the points for the same module in each sample group.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig6-v1.tif"/></fig></sec><sec sec-type="appendix" id="s14"><title>Considerations of batch effect</title><p>One concern in comparing samples from multiple studies is that differential sample processing strategies could contribute to the signal between groups; in other words, batch effects could partially explain the observed trends between different cohorts. However, by including samples from a variety of studies in each group (healthy, non-IBD, and IBD) for our meta-analysis, we can mitigate the impact of batch effects on our observations. The similar distribution of the median normalized copy number for each of the 33 IBD-enriched metabolic modules (summarized across all samples within a given study), across all studies within a given sample group (<xref ref-type="fig" rid="app1fig6">Appendix 1—figure 6B</xref>), confirms that the sample group explains more of the trend than the study of origin.</p><p>Two studies dominate our sample set: <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> contributes 151 (52.8%) of the healthy samples, and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> contributes 64 (63.4%) of the IBD samples (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1b</xref>). To exclude that a cohort effect between these studies influences our observations, we repeated the IBD enrichment analysis on (1) <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> only; as well as on (2) the remaining samples. While the results obtained from the two larger studies tend to have smaller p-values, the top IBD-enriched modules are broadly similar (Kendall correlation of Wilcoxon test p-values computed on two subsets: 0.59; see <xref ref-type="fig" rid="app1fig7">Appendix 1—figure 7</xref>), demonstrating that we are capturing generic signals across studies in our sample set.</p><fig id="app1fig7" position="float"><label>Appendix 1—figure 7.</label><caption><title>Assessing batch effect of the inflammatory bowel disease (IBD) enrichment study.</title><p>(<bold>A</bold>) Scatter plot comparing the module ranks of Wilcoxon-Mann-Whitney p-values comparing IBD and healthy subjects on <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> (y-axis) and the rest of our dataset (x-axis). (<bold>B</bold>) Venn diagram displaying the overlap of IBD-enriched modules identified by the 33 smallest p-values in <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> (left) and the rest of our dataset (right). There is good agreement (20 out of 33) between the two sets of modules, indicating generalizability of the signals across studies used in our sample set.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig7-v1.tif"/></fig></sec><sec sec-type="appendix" id="s15"><title>Testing the generalizability of the metagenome classifier</title><p>To check whether performance of our logistic regression classifier was similar across the different studies in our sample set, we tested the model’s performance using a leave-two-studies-out cross-validation strategy, whereby we trained the classifier on all samples except for those from one IBD study and one healthy study, and then tested it using samples from the two studies that were left out, for a total of 24 folds. Performance was quite variable across the different folds, as expected considering the large range of sample sizes from each study and the variability in health status of each cohort. The best overall performance occurred when testing on healthy samples from <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>, with average accuracy of 89.9% across 3 folds. The worst performance occurred when testing on healthy samples from <xref ref-type="bibr" rid="bib54">Feng et al., 2015</xref>, with average accuracy of 43.1% across 4 folds. In the fold leaving out healthy samples from <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and IBD samples from <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>, no IBD-enriched modules had p-values below our FDR-adjusted significance threshold of 2e-10 and therefore no classifier was trained. As these two studies contributed the largest number of samples to our deeply sequenced subset (<xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref>: n = 151 out of 330 or 45.8%, all of which were healthy samples; <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>: n = 64 out of 330 or 19.4%, all of which were IBD samples), we considered that cohort-specific or study-specific effects could be driving the differential signal between healthy and IBD samples. To test this, we removed the samples from <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>, and ran 10-fold cross-validation using an 80–20 train-test split of the remaining 115 samples (37 IBD, 78 healthy), using the 33 IBD-enriched modules (computed from the full sample set) as features. We found that the model performed better than a naive classifier, with an average fold accuracy of 66.5%, average true healthy rate of 69.4%, and an average true IBD rate of 61%. Therefore, while a portion of the signal in our initial analysis is indeed attributable to the differences between samples from <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> and <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>, the classifier still captures an IBD-specific signal across the other studies using this set of IBD-enriched pathways.</p><p>Furthermore, we note that the two dominating studies represent individuals at the extremes of the health gradient across our sample set, as described previously. The <xref ref-type="bibr" rid="bib112">Le Chatelier et al., 2013</xref> cohort, with its numerous exclusionary conditions, contains the healthiest individuals, while the <xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref> cohort of proctocolectomy and pouchitis patients contains the unhealthiest. It is therefore unsurprising that there is a large contrast in the metabolic potential of the gut microbiome in these individuals, considering the biological differences in their respective gut environments. This is also supported by the aforementioned ability of HMI to resolve the variability in host health, as demonstrated in <xref ref-type="fig" rid="app1fig5">Appendix 1—figures 5B</xref> and <xref ref-type="fig" rid="app1fig6">6A</xref>.</p><fig id="app1fig8" position="float"><label>Appendix 1—figure 8.</label><caption><title>Identification of gut-associated genomes.</title><p>(<bold>A</bold>) Histogram of Ribosomal Protein S6 gene clusters (94% ANI) for which at least 50% of the representative gene sequence is covered by at least 1 read (≥50% ‘detection’) in fecal metagenomes from the Human Microbiome Project (HMP) (<xref ref-type="bibr" rid="bib80">Human Microbiome Project Consortium, 2012</xref>). The dashed line indicates our threshold for reaching at least 50% detection in at least 10% of the HMP samples; gray bars indicate the 11,145 gene clusters that do not meet this threshold while purple bars indicate the 836 clusters that do. (<bold>B</bold>) Data for the 836 genomes whose Ribosomal Protein S6 sequences belonged to one of the passing (purple) gene clusters. The y-axis indicates the number of healthy/inflammatory bowel disease (IBD) gut metagenomes from our set of 330 in which the full genome sequence has at least 50% detection, and the x-axis indicates the genome’s maximum detection across all 330 samples. The dashed line indicates our threshold for reaching at least 50% genome detection in at least 2% of samples; the 338 genomes that pass this threshold are tan and those that do not are purple.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app1-fig8-v1.tif"/></fig></sec><sec sec-type="appendix" id="s16"><title>Examining the impact of different HMI score thresholds on genome-level results</title><p>Determining the HMI status of a given genome required us to set a threshold for the HMI score above which a genome would be considered to have HMI. We tested several different thresholds by varying the average percent completeness of the 33 IBD-enriched metabolic modules that we expected from the ‘HMI’ genomes from ≥ 75% (corresponding to an HMI score of ≥ 24.75) to ≥ 85% (corresponding to an HMI score of ≥ 28.05). For each threshold, we computed the same statistics and ran the same statistical tests as those reported in our main manuscript to assess the impact of these thresholds on the results (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3h</xref>). At the highest threshold we tested (HMI score ≥ 28.05), a small proportion of the reference genomes (7%, or n = 24) were classified as HMI, so we did not test higher thresholds.</p><p>We found that the results from comparing HMI genomes to non-HMI genomes are similar regardless of which HMI score threshold is used to classify genomes into either group. No matter which HMI score threshold was used, the mean genome size and mean number of genes were higher for HMI genomes than for non-HMI genomes. On average, the HMI genomes were about 1 Mb larger and had 1032 more gene calls than non-HMI genomes. We ran two Wilcoxon rank-sum statistical tests to assess the following null hypotheses: (1) HMI genomes do not have higher detection in IBD samples than non-HMI genomes, and (2) HMI genomes do not have higher detection in healthy samples than non-HMI genomes. For both tests, the p-values decreased (grew more significant) as the HMI score threshold decreased due to the inclusion of more genomes in the HMI bin. The first test for higher detection of HMI genomes than non-HMI genomes in IBD samples yielded p-values less than α=0.05 at all HMI score thresholds. The second test for higher detection of HMI genomes than non-HMI genomes in healthy samples yielded p-values less than α=0.05 for the three lowest HMI score thresholds (HMI score ≥ 24.75, ≥25.08, or ≥ 25.41). However, irrespective of significance threshold and HMI score threshold, there was always far stronger evidence to reject the first null hypothesis than the second, given that the p-value for the first test in IBD samples was 1–5 orders of magnitude lower (more significant) than the p-value for the second test in healthy samples.</p><p>IBD samples harbored a significantly higher fraction of genomes classified as HMI than healthy or non-IBD samples, regardless of HMI score threshold (p &lt; 1e-15, Kruskal-Wallis rank-sum test). The p-values for this test increased (grew less significant) as the HMI score threshold decreased. This suggests that, at higher thresholds, relatively more genomes drop out of the HMI fraction in healthy/non-IBD samples than in IBD samples, thereby leading to larger differences and more significant p-values. Consequently, the HMI scores of genomes detected in IBD samples must be higher than the HMI scores of genomes detected in the other sample groups – indeed, the average HMI score of genomes detected within at least one IBD sample is 24.75, while the average score of genomes detected within at least one healthy sample is 22.78. Within a given sample, the mean HMI score of genomes detected within that sample is higher for the IBD group than in the healthy group: the average per-sample mean HMI score is 25.14 across IBD samples compared to the average of 23.00 across healthy samples.</p><p>We also assessed how many <italic>Bacteroides</italic> genomes were classified as HMI with each HMI score threshold. Given the prevalence of <italic>Bacteroides</italic> populations in individuals with IBD (<xref ref-type="bibr" rid="bib218">Vineis et al., 2016</xref>), there is an a priori expectation that several genomes of microbes in this genus would be classified as HMI. However, because the maximum HMI score of <italic>Bacteroides</italic> genomes is 25.73 (average score: 24.17), only the three lowest HMI score thresholds enabled any Bacteroides to be classified as HMI, and even then the proportion of Bacteroides genomes classified as HMI was quite low (≤30%; <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3h</xref>). In addition, several members of other groups of microbes that are not empirically associated with IBD – such as Lachnospiraceae and Ruminococcaceae species (<xref ref-type="bibr" rid="bib213">Vacca et al., 2020</xref>; <xref ref-type="bibr" rid="bib135">Martín et al., 2023</xref>) – were classified as HMI at these lower thresholds. Thus, to avoid introducing false positive genomes into the HMI group, we elected to set the HMI score threshold to at least 26.4, or 80% average completeness of the 33 IBD-enriched modules, despite the fact that none of the expected <italic>Bacteroides</italic> are classified as HMI at this level of stringency. There are likely multiple ways for a genome to be metabolically independent that are not captured by the simple classification strategy (and the specific IBD-enriched metabolic modules) utilized in this study.</p></sec></app><app id="appendix-2"><title>Appendix 2</title><sec sec-type="appendix" id="s17"><title>Comparison of anvi-estimate-metabolism to existing tools for metabolism reconstruction</title><p>There are two main strategies for estimation of metabolic potential from sequencing data. The first is metabolic modeling, in which genome-scale metabolic models (GSMMs) are built to computationally represent the network of available metabolic reactions for a particular organism (<xref ref-type="bibr" rid="bib49">Fang et al., 2020</xref>; <xref ref-type="bibr" rid="bib67">Gu et al., 2019</xref>). This strategy enables mathematical modeling of metabolic fluxes, typically with the linear programming technique known as flux-balance analysis (FBA) (<xref ref-type="bibr" rid="bib154">Orth et al., 2010</xref>), which contextualizes the metabolic network within a set of constraints and thereby enables simulation of particular physiological conditions (<xref ref-type="bibr" rid="bib188">Sen and Orešič, 2019</xref>). The second strategy is pathway prediction, which estimates the presence/absence and/or completeness of metabolic pathways to produce a summary of the metabolic capacity encoded in the input sequences. This technique has received less attention than metabolic modeling, but its results are more readily interpretable than models, and it is critical for understanding microbial functional roles without the need for a parameterized, in silico environment (<xref ref-type="bibr" rid="bib236">Zhou et al., 2022</xref>). Both methods can be integrated with auxiliary information such as gene expression data or growth kinetics for validation of predicted metabolisms (<xref ref-type="bibr" rid="bib67">Gu et al., 2019</xref>).</p><p>A variety of software tools exist for both types of metabolism reconstruction. Two early examples with basic approaches are the web-based server platforms KAAS (<xref ref-type="bibr" rid="bib147">Moriya et al., 2007</xref>) and RAST (<xref ref-type="bibr" rid="bib12">Aziz et al., 2008</xref>). KAAS simply highlights annotated enzymes within pathway maps from the KEGG database (<xref ref-type="bibr" rid="bib91">Kanehisa et al., 2006</xref>), without producing any quantitative estimates. RAST similarly produces a limited summary of metabolism by categorizing enzymes into metabolic ‘subsystems’, but is also able to produce a metabolic model using the SEED infrastructure (<xref ref-type="bibr" rid="bib40">DeJongh et al., 2007</xref>). There are a plethora of more contemporary modeling tools that generate GSMMs, including ModelSEED (<xref ref-type="bibr" rid="bib74">Henry et al., 2010</xref>), RAVEN (<xref ref-type="bibr" rid="bib1">Agren et al., 2013</xref>), merlin (<xref ref-type="bibr" rid="bib43">Dias et al., 2015</xref>), CarveMe (<xref ref-type="bibr" rid="bib130">Machado et al., 2018</xref>), and AuReMe (<xref ref-type="bibr" rid="bib3">Aite et al., 2018</xref>). For comprehensive reviews about these tools, we refer the reader to several previous publications (<xref ref-type="bibr" rid="bib51">Faria et al., 2018</xref>; <xref ref-type="bibr" rid="bib139">Mendoza et al., 2019</xref>; <xref ref-type="bibr" rid="bib67">Gu et al., 2019</xref>).</p><p>Software for pathway prediction include MinPath, DRAM, METABOLIC, and metaPathPredict. MinPath (<xref ref-type="bibr" rid="bib232">Ye and Doak, 2009</xref>) uses integer programming to determine the minimum set of pathways that explain an input set of annotations. DRAM (<xref ref-type="bibr" rid="bib189">Shaffer et al., 2020</xref>) and METABOLIC (<xref ref-type="bibr" rid="bib236">Zhou et al., 2022</xref>) both integrate annotation of genes from various enzyme databases with estimation of pathway completeness; DRAM is specialized for working with metagenome-assembled genomes (MAGs) while METABOLIC focuses on biogeochemical cycles. The goal of metaPathPredict (<xref ref-type="bibr" rid="bib61">Geller-McGrath et al., 2023</xref>) is to produce better estimations for incomplete genomes (especially MAGs reconstructed from environmental samples) using machine learning models trained on reference databases.</p><p>Though most of these tools specialize in one method of metabolism reconstruction, some software – such as Pathway Tools, KBase, gapseq, and KEMET – have the capacity for both reconstruction strategies. Pathway Tools (<xref ref-type="bibr" rid="bib95">Karp et al., 2016</xref>) is a primarily web-based platform for numerous functional analyses based upon a custom ‘omics data format called a Pathway/Genome Database, which can be used for both FBA and querying available metabolic capacity. KBase (<xref ref-type="bibr" rid="bib10">Arkin et al., 2018</xref>) is an online workspace for hosting scientific analyses on ‘omics datasets, and it contains apps for running existing metabolism software (such as DRAM, ModelSeed, and Rast) on uploaded data. Both gapseq (<xref ref-type="bibr" rid="bib238">Zimmermann et al., 2021</xref>) and KEMET (<xref ref-type="bibr" rid="bib158">Palù et al., 2022</xref>) were designed to produce more accurate metabolic models by incorporating a gap-filling process into their model generation workflows, and their pathway prediction capabilities are a side effect of this strategy. Gapseq achieves this via a novel linear programming algorithm and by utilizing a highly curated reaction database, while KEMET uses pathway prediction results for updating the metabolic models that it creates by internally running CarveMe (<xref ref-type="bibr" rid="bib130">Machado et al., 2018</xref>).</p><p>Within the landscape of these current tools, ‘anvi-estimate-metabolism’ represents a software for pathway prediction, providing quantitative predictions of metabolic capacity by computing completeness scores for a predefined set of metabolic pathways. Anvi-estimate-metabolism distinguishes itself from the existing pathway prediction tools in several important ways. First, it is the only current tool that calculates a pathway redundancy metric, to the best of our knowledge. It achieves this primarily by computing pathway copy numbers, which is an essential strategy for community-level analysis of metabolic capabilities. This program also generates an alternative metric for pathway redundancy by providing gene-level coverage values for enzymes within each pathway through its integration with the wider anvi’o codebase and data structures, assuming that read recruitment results are available. Second, ‘anvi-estimate-metabolism’ offers two distinct interpretation strategies for metabolic pathway definitions – a ‘pathwise’ strategy which considers all possible enzyme combinations (‘paths’) that would yield a complete pathway, and a ‘stepwise’ strategy that equally weighs alternative enzymes for the same reaction step. In other words, the specific enzymes used for a given metabolic conversion matter for the ‘pathwise’ metrics, but not for the ‘stepwise’ metrics. Each strategy is suitable for a different type of analysis – for instance, ‘pathwise’ metrics can be advantageous for studies of individual genomes while ‘stepwise’ metrics are appropriate for metagenomic analysis. Thus, the combinations of pathway interpretation strategies and metric type allow for the application of this tool to a variety of different research questions and input data types. In contrast, most of the other pathway prediction tools explicitly target genomic data.</p><p>Finally, ‘anvi-estimate-metabolism’ is one of the only tools that supports user-defined metabolic pathways rather than exclusively relying on reference pathways (i.e. from KEGG). The program enables users to create their own pathway files, using enzyme annotations from any functional annotation source (including from various standard databases such as NCBI COGs, Pfam, and CAZymes as well as from custom annotations imported by the user into their anvi’o databases). The only other software with a similar feature is DRAM, which offers estimation from <ext-link ext-link-type="uri" xlink:href="https://github.com/WrightonLabCSU/DRAM/wiki/3a.-Running-DRAM#using-custom-distillate-files">‘</ext-link><ext-link ext-link-type="uri" xlink:href="https://github.com/WrightonLabCSU/DRAM/wiki/3a.-Running-DRAM#using-custom-distillate-files">custom distillate’</ext-link> files. However, DRAM’s custom pathways entirely rely on enzymes from the KO database.</p></sec><sec sec-type="appendix" id="s18"><title>Validation of PPCN approach on simulated metagenomic data</title><p>Our novel approach for normalizing metabolic pathway copy numbers by the estimated number of populations within a community to get PPCNs required validation. We used simulated metagenomes to test the robustness of our approach to the following common parameters of microbial communities that could potentially influence our comparison between healthy and IBD gut metagenomes: genome size, community size (e.g. number of distinct microbial populations within a metagenome), and diversity level (e.g. number of distinct phyla). We generated these synthetic metagenomes by randomly combining bacterial and archaeal representative genomes from different species clusters in the GTDB v95 (<xref ref-type="bibr" rid="bib162">Parks et al., 2022</xref>) according to which parameter we wanted to test (see Supplementary Methods). These representative genomes included both isolate genomes and MAGs that are not necessarily complete or well studied, but represent a wide diversity of microbial taxa. Most of the samples we created were synthetic assemblies, generated by concatenating the contig sequences from each selected genome’s FASTA into one file. To validate the full process starting from assembly, we also generated a test case starting from synthetic short reads (see Supplementary Methods).</p><p>We applied our PPCN approach to the synthetic metagenome assemblies to mimic our analysis of pathway completeness in gut metagenomes. This approach included gene annotation with KEGG KOfams and microbial SCGs, estimation of KEGG module copy numbers with ‘anvi-estimate-metabolism’, estimation of the number of populations in the community based on SCGs, and calculation of the normalized PPCN values from the resulting data. We then analyzed the accuracy of the PPCN values and their correlation with sample parameters.</p><sec sec-type="appendix" id="s18-1"><title>Two metrics for PPCN accuracy relative to genomic values</title><p>Validating the accuracy of the PPCN calculation required comparison of computed PPCN values to the true PPCN within a given metagenome. Obtaining this ‘true’ value for each synthetic community is difficult without expert knowledge of each microbe’s metabolic capacities and extensive manual calculation. Therefore, we approximated the ‘true’ PPCN in a high-throughput manner by averaging the genomic pathway metrics within a given sample. We used ‘anvi-estimate-metabolism’ to compute the stepwise completeness or copy number for each metabolic pathway within each genome in the synthetic community, then averaged these values. Though our ability to predict metabolic capacity from individual genomes is in itself limited by genome (in)completeness and missing annotations, these values can serve as a reference point for how well we summarize community-level metabolism given our current genome-level knowledge.</p><p>We used both average genomic completeness and average genomic copy number values because each metric has advantages and limitations when approximating the true PPCN value. Average genomic copy number could be considered the most direct analog to metagenomic PPCN, yet the copy number calculation in ‘anvi-estimate-metabolism’ very conservatively does not take into account partial copies of a pathway. Even when a pathway is highly complete in a given genome, a copy is not counted unless 100% of the steps are present; hence, the average genomic copy number value can underestimate the true PPCN. Genomic completeness scores can capture these partial versions of a pathway, making them a better approximation when the synthetic community harbors multiple incomplete genomes. However, completeness scores cannot resolve multiple copies of the same pathway encoded in an individual genome and can underestimate the true PPCN value, especially for short or simple pathways. Given these limitations, we assessed the accuracy of our computed PPCN values using both metrics individually as reference points. We computed ‘PPCN error’ by subtracting either average genomic completeness or average genomic copy number from the metagenomic PPCN value.</p></sec><sec sec-type="appendix" id="s18-2"><title>The PPCN calculation is generally accurate relative to genomic values but can slightly overestimate community-level metabolic capacity</title><p>Across all our simulated test cases, we observed that metagenomic PPCN was typically very close to the average genomic metrics. Relative to average completeness scores, the distribution of PPCN error was centered close to 0 and somewhat right-skewed, with a mean error ranging from –0.13 to –0.10 and a standard deviation ranging from 0.18 to 0.22. Relative to average copy number, PPCN error was centered closer to 0 and left-skewed, with a mean error ranging from 0.04 to 0.06 and a standard deviation ranging from 0.10 to 0.11 (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6a</xref>; <xref ref-type="fig" rid="app2fig1">Appendix 2—figure 1</xref>). The error range was limited in both cases, but was much smaller for error computed relative to average genomic copy number. Thus, while overall quite accurate, metagenomic PPCN has a slight tendency to underestimate average genomic completeness and overestimate average genomic copy number. For the subset of IBD-enriched pathways, the PPCN error distributions showed similar trends but were centered at a slightly higher point, with a mean of –0.06 to –0.02 relative to average completeness and a mean of 0.14–0.17 relative to average copy number.</p><p>Examining the outliers in the underlying data revealed explanations for these trends. A primary reason for the overestimation of average genomic copy number was a phenomenon we term the ‘pathway complementarity effect’. When multiple members of a synthetic community contained partial yet complementary portions of a given metabolic pathway, these enzymes combined at the metagenomic level to produce additional ‘complete’ copies of the pathway. This effect was not observed when PPCN is compared to average genomic completeness scores because the latter metric takes partial copies of the pathway into account, thus reducing the observed error. Although cross-feeding is known to occur within microbial communities (<xref ref-type="bibr" rid="bib37">Culp and Goodman, 2023</xref>; <xref ref-type="bibr" rid="bib156">Pacheco et al., 2019</xref>) and in some cases pathway complementarity at the metagenome level could capture a legitimate biological signal, for the most part this is a technical artifact yielding a degree of error in the PPCN calculation, and is a natural outcome of the decision to consider the metagenome as one large pot of genes for the purposes of pathway prediction. That said, the magnitude of this error is typically small.</p><p>Metagenomic PPCN tends to slightly underestimate average genomic completeness due to the conservative nature of the copy number calculation. In cases when multiple members of a synthetic community have incomplete pathways (with nonzero completeness scores) and their respective portions of a given pathway are noncomplementary, the average completeness scores will always be higher than the PPCN value, which does not count partial copies.</p><p>A few specific pathways had systematically overestimated PPCNs relative to one of the genomic metrics. For example, PPCN values for the beta-oxidation pathway (M00086) had the highest average error (0.946) relative to genomic completeness scores. This is an extremely short pathway, with only one reaction that can be catalyzed by one of two alternative enzymes. Thus, it represents an extreme case in which the copy number of the pathway is directly equivalent to the number of annotations for these two enzymes, which can lead to an extremely high copy number at the metagenome level. Within a given genome, the completeness of this pathway can never increase beyond 100% regardless of the number of annotations, and this limitation translates into a maximum average completeness score of 100% at the metagenome level. The systematic overestimation of PPCN for M00086 in this case is therefore due to the nature of the pathway itself and the limitation of average completeness score as a reference point. Relative to average genomic copy number, a module for glycolysis (M00001) had the highest average PPCN error (0.426) across all test cases. This is an extremely common metabolic pathway expected to occur in the majority of microbial genomes, which likely increases the pathway complementarity effect.</p></sec><sec sec-type="appendix" id="s18-3"><title>Accuracy of estimating the number of microbial populations within a metagenome</title><p>We also explored the accuracy of our method for estimating the number of populations from SCGs. In general, this method has a slight tendency to underestimate the true number of populations, and errors are more likely when the actual community size is larger (<xref ref-type="fig" rid="app2fig3">Appendix 2—figure 3</xref>). Regardless, the estimates were within 2 of the correct value over 90% of the time in all test cases for which we combined genomic contigs to create a synthetic metagenomic assembly. In the more realistic test case, when we generated synthetic short reads and assembled those reads de novo, the estimation accuracy dropped and was only within 2 of the correct value 67% of the time (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6a</xref>). Accuracy increased with greater sequencing depth (<xref ref-type="fig" rid="app2fig2">Appendix 2—figure 2</xref>), similar to what we observed in the gut metagenomes used in our main analysis (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1</xref>). This suggests that most errors in estimation were due to missing SCGs from incomplete genomes.</p><p>Since the estimated number of populations is the denominator in the PPCN calculation, underestimating these values can contribute to overestimation of the PPCN. This effect was not as strong as the pathway complementarity effect in the samples that we manually checked, which all belonged to the ideal test cases.</p></sec><sec sec-type="appendix" id="s18-4"><title>The impact of genome size in an ‘ideal’ scenario</title><p>To explore how genome size influences the PPCN approach, we ‘binned’ genomes according to genome length to obtain the following size categories: small genomes (&lt;2 Mb), medium genomes (2 Mb up to 5 Mb), and large genomes (5 Mb up to 20 Mb). Genomes larger than 20 Mb in size were excluded. We then generated 189 random communities each containing 20 genomes from two size categories (S vs M, S vs L, and M vs L), such that each sample included a different proportion of genomes from each size category on a gradient from 0 to 1 (see Supplementary Methods). We independently analyzed each group of samples with the same size category pair with a Spearman’s correlation test to identify the relationship between genome size and PPCN, PPCN accuracy, and the estimated number of populations in the metagenome.</p><p>Across all pathways in the KEGG MODULE database, proportion of small genomes in a given sample had a weak negative correlation (–0.09 &lt; R&lt;–0.02) with PPCN values (<xref ref-type="fig" rid="app2fig4">Appendix 2—figure 4</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>). The correlation became moderate (–0.48 ≤ R≤–0.21) for the subset of 33 IBD-enriched modules identified in our comparison of healthy and IBD gut metagenomes (<xref ref-type="fig" rid="app2fig5">Appendix 2—figure 5</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>). All correlations were significant with a p-value threshold of p &lt; 0.05, and the correlations were strongest for the S vs L group (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>). Thus, communities with more large genomes tend to have higher PPCN values, which makes sense considering that larger microbial genomes encode more genes and therefore more metabolic pathways. The stronger correlation for the subset of IBD-enriched modules mirrors our observation that IBD gut metagenomes harbor microbes with larger genomes with increased metabolic capacity (see ‘Reference genomes with higher metabolic independence are overrepresented in the gut metagenomes of individuals with IBD’ in the main manuscript), and suggests that this subset of pathways is particularly likely to be found in large genomes regardless of environmental or taxonomic context.</p><p>The accuracy of the PPCN calculation had a weak correlation (–0.02 &lt; R &lt; 0.07) with proportion of small genomes, regardless of which genomic metric was used to approximate the error (<xref ref-type="fig" rid="app2fig4">Appendix 2—figure 4</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>). The correlations became weakly negative for the subset of IBD-enriched pathways (–0.13 &lt; R &lt; 0.04), indicating that PPCN values are slightly more accurate for these pathways in communities of larger genomes, although several of the latter correlations were nonsignificant (<xref ref-type="fig" rid="app2fig5">Appendix 2—figure 5</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>).</p><p>Proportion of small genomes had significant, moderate to strong negative correlations with the accuracy of community size estimates (–0.71 ≤ R≤–0.36; p &lt; 1e-02), indicating that these estimates are more accurate when genomes in the community are larger (<xref ref-type="fig" rid="app2fig6">Appendix 2—figure 6</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>). This might reflect a general tendency of larger genomes to contain more complete sets of SCGs.</p></sec><sec sec-type="appendix" id="s18-5"><title>The impact of genome size in a ‘realistic’ scenario</title><p>The prior test, which is based on the combination of pre-assembled genomic contigs into a synthetic metagenomic ‘assembly’, validates how our approach works for an ‘ideal’ scenario in which all community members can be assembled. However, a full metagenomic analysis workflow starts from highly fragmented sequencing reads, which must be assembled into contigs before gene annotation and other downstream analyses can be run. The assembly process can result in data loss if some microbial populations in the community cannot be fully assembled, which can impact downstream results. To simulate this entire process, we took the same 189 synthetic communities generated for the genome size test case and generated simulated short reads from each genome to create synthetic metagenome sequencing samples (see Supplementary Methods). We randomly assigned each genome in the community a relative abundance value from a normalized relative abundance curve of the top 20 most abundant populations in a healthy human gut metagenome (<xref ref-type="fig" rid="app2fig7">Appendix 2—figure 7</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>, Supplementary Methods). We then converted the relative abundance values into coverage values ranging from 20× to 420×, and generated synthetic sequencing reads from each genome to produce its corresponding coverage value. Thus, each of the synthetic samples had the same coverage distribution across its 20 community members. After assembling the synthetic samples, we ran the PPCN workflow and performed the same validation described above.</p><p>The validation results for this more realistic scenario showed largely the same trends as the ideal genome size case (<xref ref-type="fig" rid="app2fig8">Appendix 2—figure 8</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6a and b</xref>) with two exceptions: (1) the accuracy of the population size estimates was lower, as mentioned previously. (2) The correlation between genome size proportion and population size estimation accuracy was weakly positive and nonsignificant for the M vs L group of genomes (<xref ref-type="fig" rid="app2fig9">Appendix 2—figure 9</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6a and b</xref>). Given the similarity between the results for the ‘ideal’ case and the ‘realistic’ case, we decided to only test the less computationally intensive ‘ideal’ case for the remaining parameters.</p></sec><sec sec-type="appendix" id="s18-6"><title>The impact of community size</title><p>To analyze the impact of community size on our approach, we generated 120 synthetic metagenome assemblies each containing 5, 10, 15, or 20 randomly selected genomes from the same size category (S, M, or L; see Supplementary Methods). We independently analyzed each size category as described above.</p><p>As expected, we observed a significant positive correlation between metagenomic copy number (the numerator of PPCN) and community size in each group, likely driven by the increase in the copy number of core metabolic pathways in larger communities (<xref ref-type="fig" rid="app2fig10">Appendix 2—figure 10</xref>). Interestingly, this correlation was much stronger for the subset of IBD-enriched pathways (0.49 ≤ R ≤ 0.67) than for all modules (0.12 ≤ R ≤ 0.13).</p><p>However, the correlation was much weaker and often nonsignificant for the normalized PPCN data in both groups of modules (all modules: 0.01 &lt; R &lt; 0.04, enriched modules: 0.04 &lt; R &lt; 0.09, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>; <xref ref-type="fig" rid="app2fig11">Appendix 2—figure 11</xref>), which demonstrates the suitability of our normalization method to remove the effect of community size in comparisons of metagenome-level metabolic capacity.</p><p>The correlations between community size and PPCN accuracy were weakly yet significantly positive (0.07 &lt; R ≤ 0.16), were slightly stronger for the subset of enriched modules (0.11 ≤ R ≤ 0.25), and indicate that PPCN values are slightly more accurate for smaller communities (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>, <xref ref-type="fig" rid="app2fig11">Appendix 2—figure 11</xref>).</p><p>Finally, the accuracy of estimating community size was negatively correlated with community size for communities of small and medium-sized genomes (R≤–0.61, p &lt; 1e-04), meaning our method more accurately predicts the size of smaller communities. However, this trend disappeared for communities of large genomes, in which community sizes were predicted with 100% accuracy (<xref ref-type="fig" rid="app2fig12">Appendix 2—figure 12</xref>). Similar to the observation in the genome size test case where the presence of more large genomes increased the accuracy of community size estimates, this trend might result from a tendency of larger genomes to contain more complete sets of SCGs.</p></sec><sec sec-type="appendix" id="s18-7"><title>The impact of diversity</title><p>Our final test case assessed how diversity level influences the PPCN approach. We generated 100 synthetic metagenome assemblies, each containing 20 randomly selected genomes from a different number of phyla (see Supplementary Methods). The number of genomes representing a given phylum was approximately equal in a given sample; however, genome size was not necessarily consistent. Due to the lack of genome size groups in this test case, we analyzed all 100 samples collectively.</p><p>The number of phyla represented in a given community was weakly correlated with both PPCN and PPCN accuracy (0.02 &lt; R &lt; 0.04, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>, <xref ref-type="fig" rid="app2fig13">Appendix 2—figure 13</xref>), with only slightly stronger correlations for the subset of IBD-enriched modules (0.04 &lt; R &lt; 0.07, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>, <xref ref-type="fig" rid="app2fig13">Appendix 2—figure 13</xref>). The accuracy of community size estimates were not significantly correlated with diversity level (<xref ref-type="fig" rid="app2fig14">Appendix 2—figure 14</xref>).</p></sec><sec sec-type="appendix" id="s18-8"><title>Overall impact on the comparison between healthy and IBD gut metagenomes</title><p>In summary, our validation strategy revealed good accuracy at estimating metagenome-level metabolic capacity relative to our genome-level knowledge in the simulated data. While it often underestimated average genomic completeness by ignoring partial copies of metabolic pathways and often overestimated average genomic copy number due to the effect of pathway complementarity between different community members, the magnitude of error was overall limited in range and the error distributions were centered at or near 0. Furthermore, we observed these broad error trends in all cases we tested, and therefore we expect that they would also apply to both sample groups in our comparative analysis. Thus, we next considered how the PPCN approach might have influenced our analyses that considered metagenomes from healthy individuals and from those who have IBD – two groups that differed from one another with respect to some of the variables considered in our tests.</p><p>Most of the correlations between PPCN or PPCN accuracy and sample parameters were weak, yet significant (<xref ref-type="table" rid="app2table1">Appendix 2—table 1</xref>). They showed that community size and diversity level have limited influence on the PPCN calculation, while genome size does not influence its accuracy. The only exception was the moderate correlation between PPCN and genome size, particularly for the subset of IBD-enriched pathways. It was a negative correlation with the proportion of small genomes in a metagenome, indicating that PPCN values for these pathways are larger when there are more large genomes in the community and suggesting that these pathways tend to occur frequently in larger genomes. This is in line with our observation that IBD communities contain more large genomes and therefore confirms our interpretation that the populations surviving in the IBD gut microbiome are those with the genomic space to encode more metabolic capacities.</p><table-wrap id="app2table1" position="float"><label>Appendix 2—table 1.</label><caption><title>Summary of correlation relationships between per-population copy number (PPCN), PPCN accuracy, and sample parameters from <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>.</title><p>We labeled each pair according to the strength of correlation indicated with the R value: NO, very weak with |R| ≤ 0.19; SOME, weak with 0.19 &lt; |R| ≤ 0.39; YES, moderate to strong with |R| &gt; 0.39. A ‘+’ sign in front of the R value indicates that all R values were positive, a ‘-’ sign indicates that all were negative, and the absolute value sign indicates that they had mixed signs. An asterisk (*) indicates that some of the correlations were nonsignificant (p &gt; 0.05).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Correlation?</th><th align="left" valign="bottom">PPCN (all)</th><th align="left" valign="bottom">PPCN error (all)</th><th align="left" valign="bottom">PPCN (enriched)</th><th align="left" valign="bottom">PPCN error (enriched)</th></tr></thead><tbody><tr><td align="left" valign="bottom">Genome size<break/>(proportion of smaller genomes)</td><td align="left" valign="bottom">NO<break/>(-R &lt; 0.09)</td><td align="left" valign="bottom">NO<break/>(+R ≤ 0.11)</td><td align="left" valign="bottom">YES<break/>(-R ≥ 0.48)</td><td align="left" valign="bottom">NO*<break/>(|R| &lt; 0.13)</td></tr><tr><td align="left" valign="bottom">Community size<break/>(# of populations)</td><td align="left" valign="bottom">NO*<break/>(+R &lt; 0.04)</td><td align="left" valign="bottom">SOME<break/>(+R ≤ 0.16)</td><td align="left" valign="bottom">NO*<break/>(+R &lt; 0.085)</td><td align="left" valign="bottom">SOME<break/>(+R ≤ 0.25)</td></tr><tr><td align="left" valign="bottom">Diversity<break/>(# of phyla)</td><td align="left" valign="bottom">NO<break/>(+R = 0.033)</td><td align="left" valign="bottom">NO<break/>(+R &lt; 0.04)</td><td align="left" valign="bottom">NO<break/>(+R = 0.069)</td><td align="left" valign="bottom">NO<break/>(+R &lt; 0.065)</td></tr></tbody></table></table-wrap><p>If we consider even the weak correlations, two of those relationships indicate that our approach would be more accurate for IBD metagenomes than for healthy metagenomes. For instance, PPCN accuracy was slightly higher for smaller communities (as in IBD samples), with a weakly positive correlation between PPCN error and community size. It was also slightly more accurate for less diverse communities (as in IBD samples), with a weakly positive correlation between PPCN error and number of phyla. The only opposing trend was the weakly positive correlation between PPCN error and proportion of smaller genomes, which favors higher accuracy in communities with smaller genomes (as in healthy samples). Given that our analysis focuses on the pathways enriched in IBD samples, an overall higher accuracy in IBD samples would increase the confidence in our enrichment results.</p><p>We also examined the accuracy of our method to predict the number of populations within a metagenome based on the distribution and frequency of SCGs (i.e. the denominator in the calculation of PPCN). Our benchmarks show that the estimates are overall accurate, where most errors reflect a negligible amount of underestimations of the actual number of populations. Errors occurred more frequently for the realistic synthetic assemblies generated from simulated short read data than for the ideal synthetic assemblies generated from the combination of genomic contigs. The correlations between estimation accuracy and sample parameters indicated that the population estimates are more accurate for smaller communities and communities with more large genomes, as in IBD samples (<xref ref-type="table" rid="app2table2">Appendix 2—table 2</xref>). Thus, this method is more likely to underestimate the community size in healthy samples, and these errors could lead to overestimation of PPCN in healthy samples relative to IBD samples. Thus, the enrichment of a given pathway in the IBD samples would have to overcome its relative overestimation in the healthy sample group, making it more likely that we identified pathways that were truly enriched in the IBD communities.</p><table-wrap id="app2table2" position="float"><label>Appendix 2—table 2.</label><caption><title>Summary of correlation relationships between the accuracy of community size estimates and sample parameters from <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6b</xref>.</title><p>We labeled each pair according to the strength of correlation indicated with the R value: NO, very weak with |R| ≤ 0.19; SOME, weak with 0.19 &lt; |R| ≤ 0.39; YES, moderate to strong with |R| &gt; 0.39. A ‘+’ sign in front of the R value indicates that all R values were positive, a ‘-’ sign indicates that all were negative, and the absolute value sign indicates that they had mixed signs. An asterisk (*) indicates that some of the correlations were nonsignificant (p &gt; 0.05).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Correlation?</th><th align="left" valign="bottom">Error in community size estimate</th></tr></thead><tbody><tr><td align="left" valign="bottom">Genome size<break/>(proportion of smaller genomes)</td><td align="left" valign="bottom">YES* (0.14&lt;|R| &lt; 0.72)</td></tr><tr><td align="left" valign="bottom">Community size<break/>(# of populations)</td><td align="left" valign="bottom">YES (0.6 &lt; -R &lt; 0.7)<break/>(except for large genomes)</td></tr><tr><td align="left" valign="bottom">Diversity<break/>(# of phyla)</td><td align="left" valign="bottom">NO* (+R = 0.071)</td></tr></tbody></table></table-wrap><p>Overall, the consideration of our simulations in the context of healthy vs IBD metagenomes suggest that slight biases in our estimates as a function of unequal diversity with sample groups should have driven PPCN calculations toward a conclusion that is opposite of our observations under neutral conditions. Thus, clear differences between healthy vs IBD metagenomes that overcome these biases suggest that biology, and not potential bioinformatics artifacts, is the primary driver of our observations.</p></sec></sec><sec sec-type="appendix" id="s19"><title>Supplementary methods</title><p>Below are the methods used for validation of our approach on simulated metagenomic data. All custom scripts and auxiliary data files are available on Figshare at DOI:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.26038018">10.6084/m9.figshare.26038018</ext-link>. A reproducible workflow for this section is also available at <ext-link ext-link-type="uri" xlink:href="https://merenlab.org/data/ibd-gut-metabolism/">https://merenlab.org/data/ibd-gut-metabolism/</ext-link>.</p><sec sec-type="appendix" id="s19-1"><title>Generation of synthetic communities</title><p>We generated synthetic metagenomic communities using species cluster representative genomes from the GTDB v95 (<xref ref-type="bibr" rid="bib162">Parks et al., 2022</xref>), the same database utilized in our main analysis. We used genome length to group each genome into different size categories: 7238 ‘small’ genomes of &lt; 2 Mbp; 18,416 ‘medium’-sized genomes of 2 Mbp to &lt; 5 Mbp; and 6254 ‘large’ genomes of 5 Mbp to 20 Mbp. Two especially large cyanobacterial genomes with over 20 Mbp were excluded from further analysis.</p><p>We then used custom scripts to randomly combine these genomes into synthetic communities for each simulated test case. To test the robustness of our approach to genome size, we combined 20 genomes of two size categories (small and medium; small and large; medium and large) in different proportions in each community, varying the number of genomes from the first size group from 0 to 20 and varying the number from the second size group from 20 to 0. We randomly selected genomes from each size group without replacement, and generated 3 samples per proportion value for each size category pair. This produced a total of 189 synthetic communities representing a gradient of mixed genome sizes for the genome size test case. To test the robustness of our approach to community size, we combined 5, 10, 15, or 20 randomly selected genomes (without replacement) of a single size category (small, medium, or large) in each community. We generated 40 synthetic communities (10 for each community size value) for each genome size category to obtain a total of 120 samples in the population size test case. Finally, to test the robustness of our approach to different diversity levels, we generated communities containing a total of 20 random bacterial genomes, varying the number of phyla represented in the community from 1 to 20 and ensuring that there was an approximately equal number of genomes from each phylum in the sample. Each phylum was selected randomly without replacement from the set of unique bacterial phyla represented in the GTDB. Phyla containing less than 20 genomes were excluded. We randomly selected genomes from each phylum without replacement, and we generated 5 synthetic communities for each diversity level to obtain 100 samples representing a gradient of diversity for the diversity test case.</p></sec><sec sec-type="appendix" id="s19-2"><title>Generation of ‘ideal’ synthetic metagenomes</title><p>The ‘ideal’ synthetic metagenomes represent the scenario in which all microbial populations in a metagenome are fully represented in the assembly (at least to the level of completion for each genome in the GTDB). For each of the aforementioned synthetic communities (in each test case), we concatenated the individual genome FASTA files into one FASTA containing the contig sequences of all genomes in the community. We then applied the approach (see main Methods) used in our main analysis to estimate population sizes, calculate metagenomic copy number, and compute PPCN for metabolic pathways in each synthetic metagenome. We used the same version of the KEGG database (from December 2020) as in our main analysis for consistency with those results.</p></sec><sec sec-type="appendix" id="s19-3"><title>Generation of ‘realistic’ synthetic metagenomes</title><p>We generated ‘realistic’ synthetic metagenomes for the 189 samples in the genome size test case by simulating paired-end short reads from genomic data. To ensure that the populations in these samples had realistic relative abundances, we first computed a relative abundance curve that is typical of healthy human gut metagenomes. For this task, we used the relative abundance data computed by MetaPhlAn 3 for publicly available gut metagenomes (<xref ref-type="bibr" rid="bib15">Beghini et al., 2021</xref>). We filtered this dataset to keep only species-level relative abundance values in 662 healthy control metagenomes, sorted the values in descending order, and kept only the top 20 relative abundances in each sample. We then scaled each sample’s data to have the same maximum relative abundance value, averaged the values at each rank, scaled the resulting averages to obtain a minimum average relative abundance value of 1, rounded the averages into integer coverage values, and multiplied the coverages by 20 to obtain high enough sequencing depth for proper assembly of the synthetic metagenomes. This process yielded coverage values of 20–420× for individual populations within a ‘typical’ healthy human gut metagenome (<xref ref-type="fig" rid="app2fig7">Appendix 2—figure 7</xref>).</p><p>We used a custom script to randomly select one of these typical coverage values for each genome in a given synthetic community. With the program ‘gen-paired-end-reads’ (<ext-link ext-link-type="uri" xlink:href="https://github.com/merenlab/reads-for-assembly">https://github.com/merenlab/reads-for-assembly</ext-link> copy archived at <xref ref-type="bibr" rid="bib47">Eren, 2025</xref>) we generated synthetic short reads from each input genome with the specified target coverage value, only simulating from contigs with length ≥ 1000 bp. The simulation parameters for the paired-end reads were as follows: read length of 150, inner distances with a mean of 100 bp and standard deviation of 10 bp, and a 0.05% sequencing error rate. After simulating the short reads, we assembled each individual synthetic metagenome with MEGAHIT (<xref ref-type="bibr" rid="bib119">Li et al., 2015</xref>), specifying ‘--min-contig-len 1000’. From there we followed the same approach used in our main analysis and for the ‘ideal’ synthetic metagenomes to obtain PPCNs for metabolic pathways.</p></sec><sec sec-type="appendix" id="s19-4"><title>Metabolism estimation for individual genomes</title><p>To obtain genomic pathway prediction metrics for comparison to the metagenomic values, we ran ‘anvi-estimate-metabolism’ on each genome used to generate the synthetic communities. We used the following parameters: ‘--include-zeros’, ‘--matrix-format’, and ‘--add-copy-number’ (which were the same parameters used for estimating metabolism in the synthetic metagenomes).</p></sec><sec sec-type="appendix" id="s19-5"><title>Analysis of estimation accuracy and robustness to sample parameters</title><p>We wrote custom scripts to compare the PPCN values from metagenomes to the metabolism estimation data from individual genomes, and to compare the estimated number of populations to the true community size. We then used a custom R script for analysis and visualization of these comparisons, which include statistical analysis, Spearman’s correlation with the sample parameters, and figure generation.</p><fig id="app2fig1" position="float"><label>Appendix 2—figure 1.</label><caption><title>Distribution of per-population copy number (PPCN) error relative to average genomic completeness.</title><p>(<bold>A</bold>, <bold>B</bold>) or average genomic copy number (<bold>C</bold>, <bold>D</bold>) for all modules (<bold>A, C</bold>) or just the inflammatory bowel disease (IBD)-enriched modules (<bold>B, D</bold>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig1-v1.tif"/></fig><fig id="app2fig2" position="float"><label>Appendix 2—figure 2.</label><caption><title>Scatterplot of sequencing depth vs estimated number of microbial populations in each of 189 ‘realistic’ synthetic metagenome assemblies.</title><p>The blue line shows the actual number of genomes in each synthetic community (n = 20) and the black line shows the sequencing depth threshold used in our main analysis.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig2-v1.tif"/></fig><fig id="app2fig3" position="float"><label>Appendix 2—figure 3.</label><caption><title>Distribution of error for the estimated number of populations in the synthetic metagenomes.</title><p>(<bold>A</bold>) Histogram of the difference between estimated and actual community size. (<bold>B</bold>) Distribution of estimates (y-axis) for each actual community size (x-axis).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig3-v1.tif"/></fig><fig id="app2fig4" position="float"><label>Appendix 2—figure 4.</label><caption><title>Correlations between proportion of genomes in smaller size category and (<bold>A–C</bold>) per-population copy number (PPCN) or (<bold>D–F</bold>) PPCN error relative to average genomic copy number for each size category pair (<bold>A/D</bold>) small vs medium genomes; (<bold>B/E</bold>) small vs large genomes; (<bold>C/F</bold>) medium vs large genomes across all modules.</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig4-v1.tif"/></fig><fig id="app2fig5" position="float"><label>Appendix 2—figure 5.</label><caption><title>Correlations between proportion of genomes in smaller size category and (<bold>A–C</bold>) per-population copy number (PPCN) or (<bold>D–F</bold>) PPCN error relative to average genomic copy number for each size category pair (<bold>A/D</bold>: small vs medium genomes; <bold>B/E</bold>: small vs large genomes; <bold>C/F</bold>: medium vs large genomes) across inflammatory bowel disease (IBD)-enriched modules (n = 33).</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig5-v1.tif"/></fig><fig id="app2fig6" position="float"><label>Appendix 2—figure 6.</label><caption><title>Correlation between proportion of genomes in smaller size category and error in community size estimate (relative to actual community size) for each size category pair (<bold>A</bold>: small vs medium genomes; <bold>B</bold>: small vs large genomes; <bold>C</bold>: medium vs large genomes).</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig6-v1.tif"/></fig><fig id="app2fig7" position="float"><label>Appendix 2—figure 7.</label><caption><title>Normalized average relative abundance curve.</title><p>(<bold>A</bold>) for the top 20 most abundant populations in a typical healthy human gut metagenome and (<bold>B</bold>) their corresponding coverage values in our synthetic metagenomes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig7-v1.tif"/></fig><fig id="app2fig8" position="float"><label>Appendix 2—figure 8.</label><caption><title>Correlations between proportion of genomes in smaller size category and (<bold>A–C, G–I</bold>) per-population copy number (PPCN) or (<bold>D–F, J–L</bold>) PPCN error relative to average genomic copy number for each size category pair across all modules (<bold>A–F</bold>) or the subset of inflammatory bowel disease (IBD)-enriched modules (<bold>G–L</bold>) in the realistic genome size test case.</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig8-v1.tif"/></fig><fig id="app2fig9" position="float"><label>Appendix 2—figure 9.</label><caption><title>Correlations between proportion of genomes in smaller size category and error in community size estimate (relative to actual community size) for each size category pair.</title><p>(<bold>A</bold>) small vs medium genomes; (<bold>B</bold>) small vs large genomes; (<bold>C</bold>) medium vs large genomes in the realistic genome size test case. The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig9-v1.tif"/></fig><fig id="app2fig10" position="float"><label>Appendix 2—figure 10.</label><caption><title>Correlation between community size and metagenomic copy number across all modules.</title><p>(<bold>A–C</bold>) and across the subset of enriched modules (<bold>D–F</bold>) for each genome size category (<bold>A/D</bold>: small genomes; <bold>B/E</bold>: medium genomes; <bold>C/F</bold>: large genomes) in the community size test case. The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig10-v1.tif"/></fig><fig id="app2fig11" position="float"><label>Appendix 2—figure 11.</label><caption><title>Correlations between community size and (<bold>A–C, G–I</bold>) per-population copy number (PPCN) or (<bold>D–F, J–L</bold>) PPCN error relative to average genomic copy number for each genome size category (<bold>A/D/G/J</bold>: small genomes; <bold>B/E/H/K</bold>: medium genomes; <bold>C/F/I/L</bold>: largegenomes), across all modules (<bold>A–F</bold>) or the subset of inflammatory bowel disease (IBD)-enriched modules (<bold>G–L</bold>) in the community size test case.</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted inblue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig11-v1.tif"/></fig><fig id="app2fig12" position="float"><label>Appendix 2—figure 12.</label><caption><title>Correlations between community size and error in community size estimate (relative to actual community size) for each genome size category.</title><p>(<bold>A</bold>) small; (<bold>B</bold>) medium; (<bold>C</bold>) large in the community size test case. The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig12-v1.tif"/></fig><fig id="app2fig13" position="float"><label>Appendix 2—figure 13.</label><caption><title>Correlations between number of phyla and (<bold>A/C</bold>) per-population copy number (PPCN) or PPCN accuracy relative to average genomic copy number (<bold>B/D</bold>), for all modules (<bold>A/B</bold>) or the subset of inflammatory bowel disease (IBD)-enriched modules (<bold>C/D</bold>) in the diversity test case.</title><p>The Spearman’s correlation coefficients and p-values are shown in the top-right corner of each plot, and regression lines are plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig13-v1.tif"/></fig><fig id="app2fig14" position="float"><label>Appendix 2—figure 14.</label><caption><title>Correlation between number of phyla and accuracy of community size estimates (relative to actual community size) in the diversity test case.</title><p>The Spearman’s correlation coefficient and p-value are shown in the top-right corner of each plot, and the regression line is plotted in blue.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-app2-fig14-v1.tif"/></fig></sec></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89862.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Turnbaugh</surname><given-names>Peter J</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>University of California, San Francisco</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Compelling</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Important</kwd></kwd-group></front-stub><body><p>This study presents an <bold>important</bold> new bioinformatics tool for normalizing gene copy number from metagenomic assemblies and applies it to gain functional insights into the loss of microbial diversity during conditions of stress. The inclusion of extensive computational validation makes this a <bold>compelling</bold> study that raises intriguing new hypotheses regarding the impact of disease states on the gut microbiome. This paper will likely be of broad interest to researchers studying the role of complex microbial communities in host health and disease.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89862.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>In this work, Veseli et al. present a computational framework to infer the functional diversity of microbiomes in relation to microbial diversity directly from metagenomic data. The framework reconstructs metabolic modules form metagenomes and calculates the per-population copy number of each module, resulting in the proportion of microbes in the sample carrying certain genes. They applied this framework to a dataset of gut microbiomes from 109 inflammatory bowel disease (IBD) patients, 78 patients with other gastrointestinal conditions, and 229 healthy controls. The found that the microbiomes of IBD patients were enriched in a high fraction of metabolic pathways, including biosynthesis pathways such as those for amino acids, vitamins, nucleotides, and lipids. Hence, they had higher metabolic independence compared with healthy controls. To an extent, the authors also found a pathway enrichment suggesting higher metabolic independence in patients with gastrointestinal conditions other than IBD indicating this could be a signal for a general loss in host health. Finally, a machine learning classifier using high metabolic independence in microbiomes could predict IBD with good accuracy. Overall, this is an interesting and well-written article and presents a novel workflow that enables a comprehensive characterization of microbiome cohorts.</p><p>Comments on revisions:</p><p>I believe that after the second round of revisions, the Reviewers sufficiently addressed the comments and improved the manuscript. Open questions have been answered. I have no further comments.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89862.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>This study builds upon the team's recent discovery that antibiotic treatment and other disturbances favours the persistence of bacteria with genomes that encode complete modules for the synthesis of essential metabolites (Watson et al. 2023). Veseli and collaborators now provide an in-depth analysis of metabolic pathway completeness within microbiomes, finding strong evidence for an enrichment of bacteria with high metabolic independence in the microbiomes associated with IBD and other gastrointestinal disorders. Importantly, this study provides a new open-source software to facilitate the reconstruction of metabolic pathways, estimate their completeness and normalize their results according to species diversity. Finally, this study also shows that metabolic independence of microbial communities can be used as a marker of dysbiosis. The function-based health index proposed here is more robust to individual's lifestyles and geographic origin than previously proposed methods based on bacterial taxonomy.</p><p>The implications of this study have the potential to spur a paradigm shift in the field. It shows that certain bacterial taxa that have been consistently associated with disease might not be harmful to their host as previously thought. These bacteria seem to be the only species that are able to survive in a stressed gut environment. They might even be important to rebuild a healthy microbiome (although the authors are careful in not making this speculation).</p><p>This paper provides an in-depth discussion of the results, and limitations are clearly addressed throughout the manuscript (see also the supplementary files for an in-depth assessment of the robustness of the methods). Some of the potential limitations relate to the use of large publicly available datasets, where sample processing and the definition of healthy status varies between studies. The authors have recognised these issues and their results were robust to analyses performed at a per-cohort basis. The potential limitations therefore are unlikely to have affected the conclusions of this study.</p><p>Overall, this is manuscript is a magnificent contribution to the field, likely to inspire many other studies to come.</p><p>Comments on revisions:</p><p>The authors have performed a detailed assessment of the accuracy and robustness of their new methods, and included an informative session comparing their new approach with existing ones. The new analyses have strengthened the manuscript, and the results support the biological interpretations of the study.</p><p>I commend the authors for the effort and the excellent research.</p></body></sub-article><sub-article article-type="referee-report" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89862.3.sa3</article-id><title-group><article-title>Reviewer #3 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>The major strength of this manuscript is the &quot;anvi-estimate-metabolism' tool, which is already accessible online, extensively documented, and potentially broadly useful to microbial ecologists. Inclusion of extensive benchmarking and validation on simulated metagenomes has further increased confidence in this approach. Further, the conceptual insights raise interesting hypotheses that could be pursued in follow-on experimental work.</p><p>Comments on revisions:</p><p>Thank you for the very thorough response and congratulations!</p></body></sub-article><sub-article article-type="author-comment" id="sa4"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89862.3.sa4</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Veseli</surname><given-names>Iva</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Chen</surname><given-names>Yiqun T</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Schechter</surname><given-names>Matthew S</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Vanni</surname><given-names>Chiara</given-names></name><role specific-use="author">Author</role><aff><institution>University of Bremen</institution><addr-line><named-content content-type="city">Bremen</named-content></addr-line><country>Germany</country></aff></contrib><contrib contrib-type="author"><name><surname>Fogarty</surname><given-names>Emily C</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Watson</surname><given-names>Andrea R</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Jabri</surname><given-names>Bana</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Blekhman</surname><given-names>Ran</given-names></name><role specific-use="author">Author</role><aff><institution>University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Willis</surname><given-names>Amy D</given-names></name><role specific-use="author">Author</role><aff><institution>University of Washington</institution><addr-line><named-content content-type="city">Seattle</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Yu</surname><given-names>Michael K</given-names></name><role specific-use="author">Author</role><aff><institution>Toyota Technological Institute at Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Fernàndez-Guerra</surname><given-names>Antonio</given-names></name><role specific-use="author">Author</role><aff><institution>University of Copenhagen</institution><addr-line><named-content content-type="city">Copenhagen</named-content></addr-line><country>Denmark</country></aff></contrib><contrib contrib-type="author"><name><surname>Füssel</surname><given-names>Jessika</given-names></name><role specific-use="author">Author</role><aff><institution>Carl von Ossietzky University of Oldenburg</institution><addr-line><named-content content-type="city">Oldenburg</named-content></addr-line><country>Germany</country></aff></contrib><contrib contrib-type="author"><name><surname>Eren</surname><given-names>A Murat</given-names></name><role specific-use="author">Author</role><aff><institution>Helmholtz Institute for Functional Marine Biodiversity</institution><addr-line><named-content content-type="city">Oldenburg</named-content></addr-line><country>Germany</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p><bold>Response to Public Reviewer Comments:</bold></p><p><bold>Reviewer 1:</bold></p><p>In this work, Veseli et al. present a computational framework to infer the functional diversity of microbiomes in relation to microbial diversity directly from metagenomic data. The framework reconstructs metabolic modules from metagenomes and calculates the per-population copy number of each module, resulting in the proportion of microbes in the sample carrying certain genes. They applied this framework to a dataset of gut microbiomes from 109 inflammatory bowel disease (IBD) patients, 78 patients with other gastrointestinal conditions, and 229 healthy controls. They found that the microbiomes of IBD patients were enriched in a high fraction of metabolic pathways, including biosynthesis pathways such as those for amino acids, vitamins, nucleotides, and lipids. Hence, they had higher metabolic independence compared with healthy controls. To an extent, the authors also found a pathway enrichment suggesting higher metabolic independence in patients with gastrointestinal conditions other than IBD indicating this could be a signal for a general loss in host health. Finally, a machine learning classifier using high metabolic independence in microbiomes could predict IBD with good accuracy. Overall, this is an interesting and well-written article and presents a novel workflow that enables a comprehensive characterization of microbiome cohorts.</p></disp-quote><p>We thank the reviewer for their interest in our study, their summary of its findings, and their kind words about the manuscript quality.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer 2:</bold></p><p>This study builds upon the team's recent discovery that antibiotic treatment and other disturbances favour the persistence of bacteria with genomes that encode complete modules for the synthesis of essential metabolites (Watson et al. 2023). Veseli and collaborators now provide an in-depth analysis of metabolic pathway completeness within microbiomes, finding strong evidence for an enrichment of bacteria with high metabolic independence in the microbiomes associated with IBD and other gastrointestinal disorders. Importantly, this study provides new open-source software to facilitate the reconstruction of metabolic pathways, estimate their completeness and normalize their results according to species diversity. Finally, this study also shows that the metabolic independence of microbial communities can be used as a marker of dysbiosis. The function-based health index proposed here is more robust to individuals' lifestyles and geographic origin than previously proposed methods based on bacterial taxonomy.</p><p>The implications of this study have the potential to spur a paradigm shift in the field. It shows that certain bacterial taxa that have been consistently associated with disease might not be harmful to their host as previously thought. These bacteria seem to be the only species that are able to survive in a stressed gut environment. They might even be important to rebuild a healthy microbiome (although the authors are careful not to make this speculation).</p><p>This paper provides an in-depth discussion of the results, and limitations are clearly addressed throughout the manuscript. Some of the potential limitations relate to the use of large publicly available datasets, where sample processing and the definition of healthy status varies between studies. The authors have recognised these issues and their results were robust to analyses performed on a per-cohort basis. These potential limitations, therefore, are unlikely to have affected the conclusions of this study.</p><p>Overall, this manuscript is a magnificent contribution to the field, likely to inspire many other studies to come.</p></disp-quote><p>We thank the reviewer for their endorsement of our study and their precision regarding the evaluation of its strengths. We also appreciate their high expectations for its impact in the field.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer 3:</bold></p><p>The major strength of this manuscript is the &quot;anvi-estimate-metabolism' tool, which is already accessible online, extensively documented, and potentially broadly useful to microbial ecologists.</p></disp-quote><p>We thank the reviewer for their recognition of the computational advances in this study. We also thank the reviewer for their suggestions that we have addressed below, which allowed us to strengthen our manuscript.</p><disp-quote content-type="editor-comment"><p>However, the context for this tool and its validation is lacking in the current version of the manuscript. It is unclear whether similar tools exist; if so, it would help to benchmark this new tool against prior methods.</p></disp-quote><p>The reviewer brings up a very good point about the lack of context for the `anvi-estimate-metabolism` program. While our efforts that led to the emergence of this software included detailed benchmarking efforts, a formal assessment of its performance and accuracy was indeed lacking. We are thankful for our reviewer to point this out, which motivated us to perform additional analyses to address such concerns. Our revision contains a new, 34-page long supplementary information file (Supplementary File 2) that includes a section titled “Comparison of anvi-estimate-metabolism to existing tools for metabolism reconstruction”. The text therein describes the landscape of currently available software for metabolism reconstruction and describes the features that make `anvi-estimate-metabolism` unique – namely, (1) its implementation of metrics that make it suitable for metagenome-level analyses (i.e., pathway copy number and stepwise interpretation of pathway definitions) and (2) its ability to process user-defined metabolic pathways rather than exclusively relying on KEGG. As described in that section, there is currently no other tool that can compute copy numbers of metabolic pathways from metagenomic data. Hence, it is not quite possible to benchmark the copy number methodology used in our study against prior methods; however, our benchmarking of this functionality with synthetic genomes and metagenomes (described later in this document) does provide necessary quantitative insights into its accuracy and efficiency.</p><p>While comparison of the copy number calculations to other tools was not possible due to the unique nature of this functionality, it was possible to benchmark our gene function annotation methodology against existing tools that also annotate genes with KEGG KOfams, which is a step commonly used by various tools that aim to estimate metabolic potential in genomes and metagenomes. In the anvi’o software ecosystem the annotation of genes for metabolic reconstruction is implemented in `anvi-run-kegg-kofams`, and represents a step that is required by `anvi-estimate-metabolism`. As our comparisons were quite extensive and involved additional researchers, we described them in another study which we titled “Adaptive adjustment of significance thresholds produces large gains in microbial gene annotations and metabolic insights” (doi:10.1101/2024.07.03.601779) that is now cited from within our revision in the appropriate context. Briefly, our comparison of anvi’o, Kofamscan, and MicrobeAnnotator using 396 publicly-available bacterial genomes from 11 families demonstrated that `anvi-run-kegg-kofams` is able to identify an average of 12.8% more KO annotations per genome than the other tools, especially in families commonly found in the gut environment (Figure 1). Furthermore, anvi’o recovered the highest proportion of annotations that were independently validated using eggNOG-mapper. Our comparisons also showed that annotations from anvi’o yield at least 11.6% more complete metabolic modules than Kofamscan or MicrobeAnnotator, including the identification of butyrate biosynthesis in <italic>Lachnospiraceae</italic> genomes at rates similar to manual identification of this pathway in this clade (Figure 2a). Overall, our findings that are now described extensively in DOI:10.1101/2024.07.03.601779 show that our method captures high-quality annotations for accurate downstream metabolism estimates.</p><p>We hope these new data help increase the reviewer’s confidence in our results.</p><disp-quote content-type="editor-comment"><p>Simulated datasets could be used to validate the approach and test its robustness to different levels of bacterial richness, genome sizes, and annotation level.</p></disp-quote><p>We thank the reviewer for this suggestion. It was an extremely useful exercise that not only helped us elucidate the nuances of our approach, but also enabled us to further highlight its strengths in our manuscript. We created simulated datasets including a total of 409 synthetic metagenomes that we used to test the robustness of our approach to different genome sizes, community sizes, and levels of diversity. Overall, our tests with these synthetic metagenomes demonstrated that our approach of computing PPCN values to summarize the metabolic capacity within a metagenomic community is accurate and robust to differences in all three critical variables. Most of these variables were weakly correlated between PPCN or PPCN accuracy, and the few correlations that were stronger in fact further supported our original hypothesis that we generated from our comparisons of healthy and IBD gut metagenomes. The methods and results of our validation efforts are explained in detail in our new Supplementary File 2 (see the section titled “Validation of per-population copy number (PPCN) approach on simulated metagenomic data”), but we copy here the subsection that summarizes our findings for the reviewer’s convenience:</p><p>Overall impact on the comparison between healthy and IBD gut metagenomes</p><p>“In summary, our validation strategy revealed good accuracy at estimating metagenome-level metabolic capacity relative to our genome-level knowledge in the simulated data. While it often underestimated average genomic completeness by ignoring partial copies of metabolic pathways and often overestimated average genomic copy number due to the effect of pathway complementarity between different community members, the magnitude of error was overall limited in range and the error distributions were centered at or near 0. Furthermore, we observed these broad error trends in all cases we tested, and therefore we expect that they would also apply to both sample groups in our comparative analysis. Thus, we next considered how the PPCN approach might have influenced our analyses that considered metagenomes from healthy individuals and from those who have IBD – two groups that differed from one another with respect to some of the variables considered in our tests.</p><p>Most of the correlations between PPCN or PPCN accuracy and sample parameters were weak, yet significant (Table 1). They showed that community size and diversity level have limited influence on the PPCN calculation, while genome size does not influence its accuracy. The only exception was the moderate correlation between PPCN and genome size, particularly for the subset of IBD-enriched pathways. It was a negative correlation with the proportion of small genomes in a metagenome, indicating that PPCN values for these pathways are larger when there are more large genomes in the community and suggesting that these pathways tend to occur frequently in larger genomes. This is in line with our observation that IBD communities contain more large genomes and therefore confirms our interpretation that the populations surviving in the IBD gut microbiome are those with the genomic space to encode more metabolic capacities.</p><p>If we consider even the weak correlations, two of those relationships indicate that our approach would be more accurate for IBD metagenomes than for healthy metagenomes. For instance, PPCN accuracy was slightly higher for smaller communities (as in IBD samples), with a weakly positive correlation between PPCN error and community size. It was also slightly more accurate for less diverse communities (as in IBD samples), with a weakly positive correlation between PPCN error and number of phyla. The only opposing trend was the weakly positive correlation between PPCN error and proportion of smaller genomes, which favors higher accuracy in communities with smaller genomes (as in healthy samples). Given that our analysis focuses on the pathways enriched in IBD samples, an overall higher accuracy in IBD samples would increase the confidence in our enrichment results.</p><p>We also examined the accuracy of our method to predict the number of populations within a metagenome based on the distribution and frequency of single-copy core genes (i.e., the denominator in the calculation of PPCN). Our benchmarks show that the estimates are overall accurate, where most errors reflect a negligible amount of underestimations of the actual number of populations. Errors occurred more frequently for the realistic synthetic assemblies generated from simulated short read data than for the ideal synthetic assemblies generated from the combination of genomic contigs. The correlations between estimation accuracy and sample parameters indicated that the population estimates are more accurate for smaller communities and communities with more large genomes, as in IBD samples (Table 2). Thus, this method is more likely to underestimate the community size in healthy samples, and these errors could lead to overestimation of PPCN in healthy samples relative to IBD samples. Thus, the enrichment of a given pathway in the IBD samples would have to overcome its relative overestimation in the healthy sample group, making it more likely that we identified pathways that were truly enriched in the IBD communities.</p><p>Overall, the consideration of our simulations in the context of healthy vs IBD metagenomes suggest that slight biases in our estimates as a function of unequal diversity with sample groups should have driven PPCN calculations towards a conclusion that is opposite of our observations under neutral conditions. Thus, clear differences between healthy vs IBD metagenomes that overcome these biases suggest that biology, and not potential bioinformatics artifacts, is the primary driver of our observations.”</p><p>Accordingly, we have added the following sentence summarizing the validation results to our paper:</p><p>“Our validation of this method on simulated metagenomic data demonstrated that it is accurate in capturing metagenome-level metabolic capacity relative to genome-level metabolic capacity estimated from the same data (Supplementary File 2, Supplementary Table 6).”</p><p>Early in this process of validation, we identified and fixed two minor bugs in our codebase. The bugs did not affect the results of our paper and therefore did not warrant a re-analysis of our data. The first bug, which is detailed in the Github issue <ext-link ext-link-type="uri" xlink:href="https://github.com/merenlab/anvio/issues/2231">https://github.com/merenlab/anvio/issues/2231</ext-link> and fixed in the pull request <ext-link ext-link-type="uri" xlink:href="https://github.com/merenlab/anvio/pull/2235">https://github.com/merenlab/anvio/pull/2235</ext-link>, led to the overestimation of the number of microbial populations in a metagenome when the metagenome contains both Bacteria and Archaea. None of the gut metagenomes analyzed in our paper contained archaeal populations, so this bug did not affect our community size estimates.</p><p>The second bug, which is detailed in the Github issue <ext-link ext-link-type="uri" xlink:href="https://github.com/merenlab/anvio/issues/2217">https://github.com/merenlab/anvio/issues/2217</ext-link> and fixed in the pull request <ext-link ext-link-type="uri" xlink:href="https://github.com/merenlab/anvio/pull/2218">https://github.com/merenlab/anvio/pull/2218</ext-link>, caused inflation of stepwise copy numbers for a specific type of metabolic pathway in which the definition contained an inner parenthetical clause. This bug affected only 3 pathways in the KEGG MODULE database we used for our analysis, M00083, M00144, and M00149. It is worth noting that one of those pathways, M00083, was identified as an IBD-enriched module in our analysis. However, the copy number inflation resulting from this bug would have occurred equivalently in both the healthy and IBD sample groups and thus should not have impacted our comparative analysis.</p><p>Regardless, we are grateful for the suggestion to validate our approach since it enabled us to identify and eliminate these minor issues.</p><disp-quote content-type="editor-comment"><p>The concept of metabolic independence was intriguing, although it also raises some concerns about the overinterpretation of metagenomic data. As mentioned by the authors, IBD is associated with taxonomic shifts that could confound the copy number estimates that are the primary focus of this analysis. It is unclear if the current results can be explained by IBD-associated shifts in taxonomic composition and/or average genome size. The level of prior knowledge varies a lot between taxa; especially for the IBD-associated gamma-Proteobacteria.</p></disp-quote><p>The reviewer brings up an important point, and we are thankful for the opportunity to clarify the impact of taxonomy on our analysis. Though IBD has been associated with taxonomic shifts in the gut microbiome, a major problem with such associations is that the taxonomic signal is extremely variable, leading to inconsistency in the observed shifts across different studies (doi:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/pathogens8030126">https://doi.org/10.3390/pathogens8030126</ext-link>). Indeed, one of the most comprehensive prior studies into this topic demonstrated that inter-individual variation is the largest contributor to all multi-omic measurements aiming to differentiate between the gut microbiome of individuals with IBD from that of healthy individuals, including taxonomy (doi:10.1038/s41586-019-1237-9). We therefore took a different approach to study this question that is independent of taxonomy, by focusing on metabolic potential estimated directly from metagenomes to elucidate an ecological explanation behind the reduced diversity of the IBD gut microbiome, which studies of taxonomic composition alone are not able to provide. Furthermore, the variability inherent to taxonomic profiles of the gut microbiome makes it unlikely that taxonomic shifts could confound our analysis, especially given our large sample set encompassing a variety of individuals with different origins, ages, and genders.</p><p>We agree with the reviewer that our level of prior knowledge varies substantially across taxa. Regardless, the only prior knowledge with any bearing on our ability to estimate metabolic capacity in a taxonomy-independent manner is the extent of sequence diversity captured by our annotation models for the enzymes used in metabolic pathways. During our analysis, we had observed that metagenomes in the healthy group had fewer gene annotations than those in the IBD group and we therefore shared the reviewer’s concern about potential annotation bias, whereby less-studied genomes are not always incorporated into the Hidden Markov Models for annotating KEGG Orthologs, perhaps making it more likely for us to miss annotations in these genomes (and leading to lower completeness scores for metabolic pathways in the healthy samples). Our annotation method partially addresses this limitation by taking a second look at any unannotated genes and mindfully relaxing the bit score similarity thresholds to capture annotations for any genes that are slightly too different from reference sequences for annotation with default thresholds. As mentioned previously, our recent preprint demonstrates the efficacy of this strategy (doi:10.1101/2024.07.03.601779). To further address this concern, we also investigated the extent of distant homology in these metagenomes using AGNOSTOS (doi:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7554/eLife.67667">https://doi.org/10.7554/eLife.67667</ext-link>), which showed a higher proportion of unknown genes in the healthy metagenomes and suggested that a substantial portion of the unannotated genes are not distant homologs of known enzymes that we failed to annotate due to lack of prior knowledge about them, but rather are completely novel functions. To describe these results, we added the following paragraph and two accompanying figures (Supplementary Figure 4g-h) to the section “Differential annotation efficiency between IBD and Healthy samples” in Supplementary File 1:</p><p>“To understand the potential origins of the reduced annotation rate in healthy metagenomes, we ran AGNOSTOS (Vanni et al. 2022) to classify known and unknown genes within the healthy and IBD sample groups. AGNOSTOS clusters genes to contextualize them within an extensive reference dataset and then categorizes each gene as ‘known’ (has homology to genes annotated with Pfam domains of known function), ‘genomic unknown’ (has homology to genes in genomic reference databases that do not have known functional domains), or ‘environmental unknown’ (has homology to genes from metagenomes or MAGs that do not have known functional domains). The resulting classifications confirm that healthy metagenomes contain fewer ‘known’ genes than metagenomes in the IBD sample group – the proportion of ‘known’ genes classified by AGNOSTOS is about 3.0% less in the healthy metagenomes than in the IBD sample group, which is similar to the ~3.5% decrease in the proportion of ‘unannotated’ genes observed by simply counting the number of genes with at least one functional annotation (Supplementary Figure 4g-h, Supplementary Table 1e). Furthermore, the majority of the unannotated genes in either sample group were categorized by AGNOSTOS as ‘genomic unknown’ (Supplementary Figure 4g), suggesting that the unannotated sequences are genes without biochemically-characterized functions currently associated with them and are thus legitimately lacking a functional annotation in our analysis, rather than representing distant homologs of known protein families that we failed to annotate. Based upon the classifications, a systematic technical bias is unlikely driving the annotation discrepancy between the sample groups.”</p><p>Furthermore, we have already discussed this limitation and its implications in our manuscript (see section “Key biosynthetic pathways are enriched in microbial populations from IBD samples”). To further clarify that our approach is independent of taxonomy, we have now also amended the following statement in our introduction:</p><p>“Here we implemented a high-throughput, taxonomy-independent strategy to estimate metabolic capabilities of microbial communities directly from metagenomes and investigate whether the enrichment of populations with high metabolic independence predicts IBD in the human gut.”</p><p>Finally, the reviewer is also correct that genome size is a part of the equation, as genome size and level of metabolic capacity are inextricable. In fact, we observed this in our analysis, as already stated in our paper:</p><p>“HMI genomes were on average substantially larger (3.8 Mbp) than non-HMI genomes (2.9 Mbp) and encoded more genes (3,634 vs. 2,683 genes, respectively)”</p><p>Since larger genomes have the space to encode more functional capacity, it follows that having higher metabolic independence would require a microbe to have a larger genome. The validation of our method on simulated metagenomic data supported this idea by demonstrating that the IBD-enriched metabolic pathways are commonly identified in large genomes. The validation also proved that genome size does not influence the accuracy of our approach (Supplementary File 2).</p><disp-quote content-type="editor-comment"><p>It can be difficult to distinguish genes for biosynthesis and catabolism just from the KEGG module names and the new normalization tool proposed herein markedly affects the results relative to more traditional analyses.</p></disp-quote><p>We agree with the reviewer that KEGG module names do not clearly indicate the presence of biosynthetic genes of interest. That said, KEGG is a commonly-used and extensively-curated resource, and many biologists (including ourselves) trust their categorization of genes into pathways. We hope that readers who are interested in specific genes within our results would make use of our publicly-available datasets (which include gene annotations) to conduct a targeted analysis based on their expertise and research question.</p><p>However, we would like to respectfully note that the ability to distinguish the genes within each KEGG module may not be very useful to most readers, and is unlikely to have a meaningful impact in our findings. As the reviewer most likely appreciates, the presence of individual genes in isolation can be insufficient to indicate biosynthetic capacity, considering that (1) most biosynthetic pathways involve several biochemical conversions requiring a series of enzymes, (2) enzymes are often multi-functional rather than exclusive to one pathway, and (3) different organisms in a community may utilize enzymes encoded by different genes to perform the same or similar biochemical reaction in a pathway. We therefore made the choice to analyze metabolic capacity at the pathway level, because this would better reflect the biosynthetic abilities encoded by the multiple microbial populations within each metagenome.</p><p>The reviewer also suggests that our novel normalization method affects our results, yet we believe that this normalization strategy is one of the strengths of our study in comparison to ‘more traditional analyses’ as it enables an appropriate comparison between metagenomes describing microbial communities of dramatically different degrees of richness. Indeed, we suspect that the lack of normalization in more traditional analyses may be one reason why prior analyses have so far failed to uncover any mechanistic explanation for the loss of diversity in the IBD gut microbiome. We hope that our validation efforts were sufficiently convincing in demonstrating the suitability of our approach, and copy here a particularly illuminating section of the validation results that we have added to Supplementary Information File 2:</p><p>“As expected, we observed a significant positive correlation between metagenomic copy number (the numerator of PPCN) and community size in each group, likely driven by the increase in the copy number of core metabolic pathways in larger communities (Supplementary Figure 18). Interestingly, this correlation was much stronger for the subset of IBD-enriched pathways (0.49 &lt;= R &lt;= 0.67) than for all modules (0.12 &lt;= R &lt;= 0.13).</p><p>“However, the correlation was much weaker and often nonsignificant for the normalized PPCN data in both groups of modules (all modules: 0.01 &lt; R &lt; 0.04, enriched modules: 0.04 &lt; R &lt; 0.09, Supplementary Table 6b, Supplementary Figure 19), which demonstrates the suitability of our normalization method to remove the effect of community size in comparisons of metagenome-level metabolic capacity.”</p><disp-quote content-type="editor-comment"><p>As such, it seems safer to view the current analysis as hypothesis-generating, requiring additional data to assess the degree to which metabolic dependencies are linked to IBD.</p></disp-quote><p>We certainly agree with the reviewer that our study, similar to the vast majority of studies published every year, is a hypothesis-generating work. Any idea proposed in any scientific study in life sciences will certainly benefit from additional data analyses, and therefore we respectfully do not accept this as a valid criticism of our work. The inception of this study is linked to an earlier work that hypothesized high metabolic independence as a determinant of microbial fitness in stressed gut communities (doi:10.1186/s13059-023-02924-x), which lacked validation on larger sets of data. Our study tests this original hypothesis using a large number of metagenomes, and lends further support for it with approaches that are now better validated. Furthermore, there are other studies that agree with our interpretation of the data (doi:10.1101/2023.02.17.528570, doi:10.1038/s41540-021-00178-6), and we look forward to more computational and/or experimental work in the future to generate more evidence to evaluate these insights further.</p><disp-quote content-type="editor-comment"><p><bold>Response to Recommendations for the Authors</bold></p><p><bold>Reviewer 1:</bold></p><p>My main comments include:</p><p>- From the results reported in lines 178-185, it seems that metabolic pathways in general were enriched in IBD microbiomes, not specifically biosynthetic pathways. Can we really say then that the signal is specific for biosynthesis capabilities?</p></disp-quote><p>We apologize for the confusion here. When we read the text again, we ourselves were confused with our phrasing.</p><p>The reviewer is correct that a similar proportion of both biosynthetic and non-biosynthetic pathways had elevated per-population copy number (PPCN) values in the IBD samples. However, the low microbial diversity associated with IBD and the on average larger genome size of individual populations contributes to this relative enrichment of the majority of metabolic modules. To remove this bias and identify specific modules whose enrichment was highly conserved across microbial populations associated with IBD, we implemented two criteria: (1) we selected modules that passed a high statistical significance threshold in our enrichment test (Wilcoxon Rank Sum Test, FDR-adjusted p-value &lt; 2e-10), and (2) we accounted for effect size by ranking these modules according to the difference between their median PPCN in IBD samples and their median PPCN in healthy samples, and keeping only those in the top 50% (which translated to an effect size threshold of &gt; 0.12).</p><p>This analysis revealed a set of metabolic modules that were consistently and highly significantly enriched in microbial communities associated with IBD. The majority of these metabolic modules encode biosynthesis pathways. Our use of the terms “elevated”, “enriched”, and “significantly enriched” in the previous version of the text was confusing to the reader. We thank the reviewer for pointing this out, and we hope that our revision of the text clarifies the analysis strategy and observations:</p><p>“To gain insight into potential metabolic determinants of microbial survival in the IBD gut environment, we assessed the distribution of metabolic modules within samples from each group (IBD and healthy) with and without using PPCN normalization. Without normalizing, module copy numbers were overall higher in healthy samples (Figure 2a) and modules exhibited weak differential occurrence between cohorts (Figure 2b, 2c, Supplementary Figure 3). The application of PPCN reversed this trend, and most metabolic modules were elevated in IBD (Supplementary Figure 5). This observation is influenced by two independent aspects of the healthy and IBD microbiota. The first one is the increased representation of microbial organisms with smaller genomes in healthy individuals (Watson et al. 2023), which increases the likelihood that the overall copy number of a given metabolic module is below the actual number of populations. In contrast, one of the hallmarks of the IBD microbiota is the generally increased representation of organisms with larger genomes (Watson et al. 2023). The second aspect is that the generally higher diversity of microbes in healthy individuals increases the denominator of the PPCN. This results in a greater reduction in the PPCN of metabolic modules that are not shared across all members of the diverse gut microbial populations in health.</p><p>To go beyond this general trend and identify modules that were highly conserved in the IBD group, we first selected those that passed a relatively high statistical significance threshold in our enrichment test (Wilcoxon Rank Sum Test, FDR-adjusted p-value &lt; 2e-10). We then accounted for effect size by ranking these modules according to the difference between their median PPCN in IBD samples and their median PPCN in healthy samples, and keeping only those in the top 50% (which translated to an effect size threshold of &gt; 0.12). This stringent filtering revealed a set of 33 metabolic modules that were significantly enriched in metagenomes obtained from individuals diagnosed with IBD (Figure 2d, 2e), 17 of which matched the modules that were associated with high metabolic independence previously (Watson et al. 2023) (Figure 2f). This result suggests that the PPCN normalization is an important step in comparative analyses of metabolisms between samples with different levels of microbial diversity.”</p><p>Lines 178-185 from our original submission have been removed to avoid further confusion. These results can be found in Supplementary File 1 (section “Module enrichment without consideration of effect size leads to nonspecific results”).</p><disp-quote content-type="editor-comment"><p>It is not entirely clear to me what is meant by PPCN normalization. Normalize the number of copy numbers to the overall number of genes?</p></disp-quote><p>The idea behind using per-population copy number (PPCN) is to normalize the prevalence of each metabolic module found in an environment with the number of microbial populations within the same sample. PPCN achieves this by dividing the pathway copy numbers by the number of microbial populations in a given metagenome, which we estimate from the frequency of bacterial single-copy core genes. We have updated the description of the per-population copy number (PPCN) calculation to clarify its use:</p><p>“Briefly, the PPCN estimates the proportion of microbes in a community with a particular metabolic capacity (Figure 1, Supplementary Figure 2) by normalizing observed metabolic module copy numbers with the ‘number of microbial populations in a given metagenome’, which we estimate using the single-copy core genes (SCGs) without relying on the reconstruction of individual genomes.”</p><p>We also note that the equation for PPCN is shown in Figure 1.</p><disp-quote content-type="editor-comment"><p>It is also not clear to me how the classifier predicts stress on microbiomes rather than dysbiosis.</p></disp-quote><p>The reviewer asks an interesting question since it is true that we could also use the term “dysbiosis” rather than “stress”. Yet we refrained from the use of dysbiosis as it is considered a poorly-defined term to describe an altered microbiome often associated with a specific disease (doi:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/microorganisms10030578">https://doi.org/10.3390/microorganisms10030578</ext-link>), such as IBD, relative to another poorly-defined state, “healthy microbiome” (doi:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/phar.2731">https://doi.org/10.1002/phar.2731</ext-link>). We do consider that stress is not necessarily a term that is less vague than dysbiosis, yet it has the advantage of being more common in studies of ecology compared to dysbiosis. Our relatively neutral stance towards which term to use has shifted dramatically due to one critical observation in our study: the identical patterns of enrichment of HMI microbes in individuals diagnosed with IBD as well as in healthy individuals treated with antibiotics. We appreciate that the observed changes in the antibiotics case can also fulfill the definition of “dysbiosis”, but the term “stress response” more accurately describes what the classifier identifies in our opinion.</p><disp-quote content-type="editor-comment"><p>What is the advantage of using the estimate-metabolism pipeline presented in this article over workflows such as those using genome-scale models, which are repeatedly cited and discussed?</p></disp-quote><p>Genome-scale models are often appropriate for a big-picture view of metabolism, and especially when the capability to perform quantitative simulations like flux-balance analysis is needed. For our investigation, we wanted a more specific and descriptive summary of metabolic capacity, so we focused on individual KEGG modules, which qualitatively describe subsets of the vast metabolic network with pathway names that all readers can understand, rather than working with an abstract model of the entire network. Furthermore, genome-scale models would have prevented us from assessing the redundancy (copy number) of metabolic pathways, as these networks usually focus on the presence-absence of gene annotations for enzymes in the network rather than the copy number of these annotations. The copy number metric has been critical for our analyses, considering that we are focusing on metabolic capacity at the community level and require the ability to normalize this metabolic capacity by the size of the community described by each metagenome. Finally, assessing a discrete set of metabolic pathways yielded a corresponding set of features that we used to create the machine learning classifier, whereas data from genome-scale models would not be as easily transferable into classifier features.</p><disp-quote content-type="editor-comment"><p>Minor comments:</p><p>Figure 2d and e are mentioned in the text before Figure 2a.</p></disp-quote><p>We thank the reviewer for catching this. We have rewritten the section as follows to put the figure references in numerical order:</p><p>!To gain insight into potential metabolic determinants of microbial survival in the IBD gut environment, we assessed the distribution of metabolic modules within samples from each group (IBD and healthy) with and without using PPCN normalization. Without normalizing, module copy numbers were overall higher in healthy samples (Figure 2a) and modules exhibited weak differential occurrence between cohorts (Figure 2b, 2c, Supplementary Figure 3). After the application of PPCN, most metabolic modules were elevated in IBD (Supplementary Figure 5). This observation is a product of two independent aspects of the healthy and IBD microbiota. The first one is the increased representation of microbial organisms with smaller genomes in healthy individuals (Watson et al. 2023), which increases the likelihood that the overall copy number of a given metabolic module is below the actual number of populations. In contrast, one of the hallmarks of the IBD microbiota is the generally increased representation of organisms with larger genomes (Watson et al. 2023). The second aspect is that the generally higher diversity of microbes in healthy individuals increases the denominator of the PPCN due to the higher number of populations detected in these samples. This results in a greater reduction in the PPCN of metabolic modules that are not shared across all members of the diverse gut microbial populations in health. To go beyond this general trend and identify modules that were highly conserved in the IBD group, we first selected those that passed a relatively high statistical significance threshold in our enrichment test (Wilcoxon Rank Sum Test, FDR-adjusted p-value &lt; 2e-10). We then accounted for effect size by ranking these modules according to the difference between their median PPCN in IBD samples and their median PPCN in healthy samples, and keeping only those in the top 50% (which translated to an effect size threshold of &gt; 0.12). This stringent filtering revealed a set of 33 metabolic modules that were significantly enriched in metagenomes obtained from individuals diagnosed with IBD (Figure 2d, 2e), 17 of which matched the modules that were associated with high metabolic independence previously (Watson et al. 2023) (Figure 2f). This result suggests that the PPCN normalization is an important step in comparative analyses of metabolisms between samples with different levels of microbial diversity.!</p><disp-quote content-type="editor-comment"><p>How much preparation is needed for users that want to apply the estimate-metabolism pipeline to their own datasets? From the documentation at anvi'o, it still seems like a significant effort.</p></disp-quote><p>We thank the reviewer for this important question. The use of anvi-estimate-metabolism is simple, but the concept it makes available and the means it offers its users to interact with their data are not basic, thus its use requires <italic>some</italic> effort. Anvi’o provides users with the ability to directly interact with their data at each step of the analysis to have full control over the analysis and to make informed decisions on the way. In comparison to pre-defined analysis pipelines that often require no additional input from the user, this approach requires some level of involvement of the user throughout the process – namely, they must run a few programs in series rather than running just one pipeline command that quietly handles everything on their behalf. The most basic workflow for using `anvi-estimate-metabolism` is quite straightforward and requires four simple steps following the installation of anvi’o: 1. Run the program `anvi-setup-kegg-data` to download the KEGG data. 2. Convert the assembly FASTA file into an anvi’o-compatible database format with gene calls by running `anvi-gen-contigs-database`. 3. Annotate genes with KOs with the program `anvi-run-kegg-kofams`. 4. Get module completeness scores and copy numbers by running `anvi-estimate-metabolism`. In addition, we provide simple tutorials (such as the one at <ext-link ext-link-type="uri" xlink:href="https://anvio.org/tutorials/fmt-mag-metabolism/">https://anvio.org/tutorials/fmt-mag-metabolism/</ext-link>) and reproducible bioinformatics workflows online (including for this study at <ext-link ext-link-type="uri" xlink:href="https://merenlab.org/data/ibd-gut-metabolism/">https://merenlab.org/data/ibd-gut-metabolism/</ext-link>) which helps early career researchers to apply similar strategies to their own datasets. We are happy to report that we have been using this tool in our undergraduate education, and observed that students with no background in computation were able to apply it to their questions without any trouble.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer 2:</bold></p><p>Congratulations on this great work, the manuscript is a pleasure to read. Minor questions that the authors might want to clarify:</p><p>L 275: Why use reference genomes from the GTDB (for only 3 phyla) instead of using MAGs reconstructed from the data? I understand that assemblies based on individual samples would probably not yield enough complete MAGs, but I would expect that co-binning the assemblies for the entire dataset would.</p></disp-quote><p>We thank the reviewer for their kind words. We certainly agree that metagenome assembled genomes (MAGs) reconstructed directly from the assemblies would by nature represent the populations in these communities better than reference genomes. However, one of our aims in this study was to avoid the often error-prone and time-consuming step of reconstructing MAGs. Most automatic binning algorithms inevitably make mistakes, and especially for metabolism estimation, low quality MAGs can introduce a bias in the analysis. At the same time the manual curation of each bin to remove any contamination would require a substantial effort and make the workflow less accessible for others to use. As an example, in our previous work (doi:10.1186/s13059-023-02924-x), careful refinement of MAGs from just two co-assemblies took two months. Here, we developed the PPCN workflow as a more scalable, assembly-level analysis to avoid the need for binning in the first place.</p><p>To supplement and confirm the metagenome-level results, we decided to run a genome-level analysis. We used the GTDB since it represents the most comprehensive, dereplicated collection of reference genomes across the tree of life. We chose those 3 phyla in particular because of their ecological relevance in the human gut environment. Bacteroidetes and</p><p>Firmicutes together represent the majority (up to ~90%) of the populations in healthy individuals (doi:10.1038/nature07540), and Proteobacteria represent the next most abundant phylum on average (2% ± 10%) (doi:10.1371/journal.pone.0206484).</p><disp-quote content-type="editor-comment"><p>L 403: Should the Franzosa and Papa papers be referenced as numbers?</p></disp-quote><p>Thanks for pointing this out. The rogue numerical citation was actually an artifact of the submission and was corrected to a long-format citation in the online version of the manuscript on the eLife website.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer 3:</bold></p><p>The lack of any experimental validation contributes to the tentative nature of the conclusions that can be drawn at this time. Numerous studies have looked at the metabolism of gut bacterial species during in vitro growth, which could be mined to test if the in silico predictions of metabolism can be supported. Alternatively, the authors could isolate key strains of interest and study them in culture or in mouse models of IBD.</p></disp-quote><p>We appreciate these suggestions and agree with the reviewer that experimental validation is important. However, we do not agree that either the use of mouse models or the isolation of individual microbial strains would be an appropriate experimental test in this case. The use of humanized gnotobiotic mice has critical limitations (see doi:10.1016/j.cell.2019.12.025 and references within the section on “human microbiota-associated murine models”). As it is not possible to establish a mouse model whose gut microbiota fully reflect the human gut microbiome, such an approach would neither be appropriate to validate our findings, nor would it have been possible to produce the insights we have gained based on environmental data. We are not sure how exactly a mouse model, even when ignoring the well established limitations, could improve or validate a comprehensive analysis of a large “environmental” datasets that resulted in highly significant signals.</p><p>We are also not sure that we understand how the reviewer believes that the isolation of individual strains would aid in validating our findings. While we appreciate that not all relevant genes are captured by the available annotation routines and that some genes may be misannotated, the large dataset used here renders these concerns negligible. Isolating a small subset of bacterial populations would hardly lead to a representative sample and testing their metabolic capacities in vitro would not improve the reliability of our analysis.</p><p>Boilerplate suggestions as vague as “isolate key strains of interest” or “experiment in mouse models of IBD” do not add or retract anything from our findings. Our findings and hypotheses are well supported by our data and extensive analyses.</p><disp-quote content-type="editor-comment"><p>Line 9 - not sure this approach is hypothesis testing in the traditional sense, you might reword.</p></disp-quote><p>Hypothesis testing occurs when one makes an observation, develops a hypothesis that explains the observation, and then gathers and analyzes data to investigate whether additional data support or disprove the hypothesis. We are not convinced a reword is necessary.</p><disp-quote content-type="editor-comment"><p>Line 40 - the lack of consistent differences in IBD and healthy individuals does not mean that the microbiome doesn't impact disease. It's important to consider all the mechanistic studies in animal models and other systems.</p></disp-quote><p>Our study does not claim that microbiome has no impact on the course of disease.</p><disp-quote content-type="editor-comment"><p>Line 50 - this seemed out of place and undercuts the current findings. Upon checking Ref. 31, the analysis seems distinct enough to not mention in the introduction.</p></disp-quote><p>We disagree. Ref 31 uses genome-scale metabolic models to identify the loss of cross-feeding interactions in the gut microbiome of individuals with IBD, which is another way of saying that the microbes in IBD no longer rely on their community for metabolic exchange – in other words, they are metabolically independent. This is an independent observation that is parallel to our results and confirms our analysis; hence, it is important to keep in our introduction.</p><disp-quote content-type="editor-comment"><p>Line 55 - Ref. 32 looked at FMT, which should be explicitly stated here.</p></disp-quote><p>The reviewer’s suggestion is not helpful. Ref 32 has a significant focus on IBD as it compares a total of 300 MAGs generated from individuals with IBD to 264 MAGs from healthy individuals and shows differences in metabolic enrichment between healthy and IBD samples independent of taxonomy, thus setting the stage for our current work. What model has been used to generate the initial insights that led to the IBD-related conclusion in Ref 32 has no significance in this context.</p><disp-quote content-type="editor-comment"><p>Lines 92-107 - this text is out of place in the Results section and reads more like a review article. Please trim it down and move it to the introduction.</p></disp-quote><p>We would like to draw the reviewer’s attention to the fact that this is a “Result and Discussion” section. In this specific case it is important for readers to appreciate the context for our new tool, as the reviewer commented in the public review. We kindly disagree with the reviewer’s suggestion to remove this text as that would diminish the context.</p><disp-quote content-type="editor-comment"><p>Line 107 - is &quot;selection&quot; the word you meant to use?</p></disp-quote><p>If the frequency of a given metabolic module remains the same or increases despite the decreasing diversity of the microbial community, it is conceivable to assume that its enrichment indicates the presence of a selective process to which the module responds. It is indeed the word we meant to use.</p><p>Line 110 - this is the first mention of this new method, need to add it to the abstract and introduction.</p><p>The reviewer must have overlooked the text passages in which we mention the strategy we developed within the abstract:</p><p>“Here, we tested this hypothesis on a large scale, by developing a software framework to quantify the enrichment of microbial metabolisms in complex metagenomes as a function of microbial diversity.”</p><p>And in the last paragraph of the introduction:</p><p>“Here we implemented a high-throughput, taxonomy-independent strategy to estimate metabolic capabilities of microbial communities directly from metagenomes…”</p><disp-quote content-type="editor-comment"><p>Figure 1 - a nice summary, but no data is shown to support the validity of this model. Consider shrinking the cartoon and adding validation with simulated datasets.</p></disp-quote><p>We hope we have addressed this recommendation with the extensive validation efforts summarized above.</p><disp-quote content-type="editor-comment"><p>Line 134 - need to state the FDR and effect size cutoffs used.</p></disp-quote><p>We have reworded this sentence as follows to clarify which thresholds were used:</p><p>“We identified significantly enriched modules using an FDR-adjusted p-value threshold of p &lt; 2e-10 and an effect size threshold of &gt; 0.12 from a Wilcoxon Rank Sum Test comparing IBD and healthy samples.”</p><disp-quote content-type="editor-comment"><p>I'm also concerned about the simple comparison of IBD to healthy without adjusting for confounders like study, geographical location, age, sex, drug use, diet, etc. More text is needed to explain the nature of these data, how much metadata is available, and which other variables distinguish IBD from healthy.</p></disp-quote><p>The reviewer is correct that there is a large amount of interindividual variation between samples due to host and environmental factors. However, the lack of adjusting for confounders was intentional, and in fact one of the critical strengths of our study. We observe a clear signal between healthy individuals and individuals diagnosed with IBD, <italic>despite</italic> the amount of interindividual variation in our diverse set of samples from 13 different studies (details of which are summarized in Supplementary Table 1). The clear increase in predicted metabolic capacity that we consistently observe in IBD patients using both metagenomes and genomes across diverse cohorts points to metabolic independence as a high-level trend that is predictive of microbial prevalence in stressed gut environments irrespective of host factors.</p><disp-quote content-type="editor-comment"><p>Line 145 - calling PPCN normalization an &quot;essential step&quot; is a huge claim and requires a lot more data to back it up. Might be best to qualify this statement.</p></disp-quote><p>We hope we have addressed this recommendation with our validation efforts. Supplementary Figures 18 and 19 in particular show evidence for the necessity of the normalization step. It is indeed an essential step <italic>if</italic> the purpose is to compare metabolic enrichment between cohorts of highly different microbial diversity.</p><disp-quote content-type="editor-comment"><p>Figure 2a - the use of a 1:1 trend line seems potentially misleading. I would replace it with a best-fit line.</p></disp-quote><p>Our purpose here was not to show the best fit. Instead, the 1:1 trend line separates the modules based on their relative abundance distribution between healthy individuals and individuals diagnosed with IBD. If the module is to the left of the line, it has a higher median copy number in healthy individuals and if the module is to the right, it has a higher median copy number in individuals with IBD. The line also helps to demonstrate the shift that occurs between the unnormalized data in Figure 2a. Without the normalization, more modules occur to the left of the</p><p>1/1 line as a result of the higher raw copy numbers in healthy metagenomes which simply contain more microbial populations. With the normalization (Figure 2d), more modules fall on the right side of the 1/1 line due to higher PPCN values. A best-fit line would not serve well for these purposes.</p><disp-quote content-type="editor-comment"><p>The text should be revised to state that this analysis actually did find many significant differences and to discuss whether they were the same modules identified in Figure 2d.</p></disp-quote><p>We apologize for the confusion and thank the reviewer for bringing this issue to our attention. As mentioned above, the disparate levels of microbial diversity between healthy individuals and individuals with IBD resulted in much larger copy numbers of metabolic modules in healthy samples reflecting the often much larger communities. Hence, we ran statistical tests only on normalized (PPCN) data. The p-values associated with each module in Figure 2a, as well as the colors of each point, are based on the PPCN data in Figure 2d. We aimed to improve the clarity of the visual comparison between normalized and unnormalized results by identifying the same set of IBD-enriched modules in plots a-c and plots d-f.</p><p>That being said, the reviewer’s comment made us realize the potential for confusion when using the normalized data’s statistical results in Figure 2a that otherwise shows results from unnormalized data. We have now run the same statistical test on the unnormalized (raw copy number) data and re-generated Figure 2a with the new FDR-adjusted p-values and points colored based on the statistical tests using unnormalized data. We’ve also removed the arrow connecting to Figure 2b (since we no longer show the same set of IBD-enriched modules in Figures 2a and 2b), and added a dashed line to indicate the effect size threshold (similar to the one in Figure 2d). We have updated the legend for Figure 2a-d to reflect these changes:</p><p>When we used the same p-value threshold (p &lt; 2e-10) as before and also filtered for an effect size larger than the mean (the same strategy used to set our effect size threshold for the normalized data), there are 10 modules that are significantly enriched based on the unnormalized data. Of course, it is difficult to gauge the relevance of these 10 modules to microbial fitness in the IBD gut environment since their raw copy numbers do not tell us anything about the relative proportion of community members that harbor these modules. Therefore, we are reluctant to add these modules to the results text. For the record, only 3 of those modules were also significantly enriched based on the normalized PPCN values: M00010 (Citrate cycle, first carbon oxidation), M00053 (Pyrimidine deoxyribonucleotide biosynthesis), and M00121 (Heme biosynthesis).</p><disp-quote content-type="editor-comment"><p>Figure 2c,f - these panels raise a lot of concerns given that the choice of method inverts the trend. Without additional data/validation, it's hard to know which method is right.</p></disp-quote><p>We hope we have addressed this recommendation with the extensive validation efforts summarized above. Inversion of the trend is an expected outcome, because the raw copy numbers of most metabolic modules are much lower in the IBD sample group due to lower community sizes.</p><disp-quote content-type="editor-comment"><p>Line 167 - Need to take the KEGG names with a grain of salt, just because it says &quot;biosynthesis&quot; doesn't mean that the pathway goes in that direction in your bacterium of interest.</p></disp-quote><p>We believe the reviewer is under a misapprehension regarding the general reversibility of KEGG metabolic modules, or indeed of metabolic pathways. Most metabolic pathways have one or several (practically) irreversible reactions. To demonstrate this for the 33 IBD-enriched modules, we evaluated their reversibility based upon their corresponding KEGG Pathway Maps, which indicate reaction reversibility via double-sided arrows. Aside from the signature modules M00705 and M00627, in 26 out of 31 pathway modules one or more irreversible reactions render these pathways one-directional. Indeed, on average the majority (54%) of the reactions in a given module are irreversible. When focusing on the 23 “biosynthesis” modules, 22 out of 23 (96%) modules have at least one irreversible reaction, and on average 64% of a given module’s reactions are irreversible. These data (which can be accessed at doi:10.6084/m9.figshare.27203226 for the reviewer’s convenience) challenge the reviewer’s notion that pathway directionality is free to change arbitrarily, since the presence of even one irreversible reaction effectively blocks the flux in the opposing direction. Thus, “biosynthesis” is indeed a meaningful term in KEGG module names.</p><p>That said, KEGG Pathway Maps, though highly curated, are likely not the final word on whether a given reaction in a metabolic pathway can be considered reversible or irreversible in each microbial population and under all conditions. And our analysis, like many others that rely on metagenomic data, does not consider the environmental conditions in the gut such as temperature or metabolite concentrations that might influence the Gibbs free energy and thus the directionality of these reactions in vivo. However, even assuming general reversibility of metabolic pathways, this would not invalidate the fact that these microbes have the metabolic capacity to synthesize the respective molecules. In other words, the potential reversibility of pathways is irrelevant to our analysis since we are describing metabolic <italic>potential</italic>. The <italic>lac</italic> operon in <italic>E. coli</italic> might only be expressed in the absence of glucose, but <italic>E. coli</italic> always has the capability to degrade lactose regardless of whether that pathway is active. Thus, our overall conclusion that gut microbes associated with IBD are metabolically self-sufficient (encoding the enzymatic capability to synthesize certain key metabolites) remains valid irrespective of fixed or flexible pathway directionality.</p><disp-quote content-type="editor-comment"><p>It's also important to be careful not to conflate KEGG modules (small subsets of a pathway) with the actual metabolic pathway. It's possible to have a module change in abundance while not altering the full pathway. Inspection of the individual genes could help in this respect - are they rate-limiting steps for biosynthesis or catabolism?</p></disp-quote><p>The reviewer is absolutely correct that KEGG modules do not necessarily represent full pathways. We have updated the language in our manuscript to explicitly refer to “modules” rather than “pathways” whenever appropriate, to restrict the scope of the analysis to metabolic modules rather than full pathways.</p><p>That said, we do not see how “inspection of individual genes” would improve our analysis. The strength of looking at complete modules rather than individual genes is that we can gain conclusive insights into a certain metabolic capacity. Of course, no pathway or module stands alone. However, the enrichment of metabolic modules does conclusively indicate that these modules are beneficial under the given conditions, such as stress caused by inflammation or antibiotic use. Whether a certain step in a module or pathway is rate limiting is completely irrelevant for this analysis.</p><disp-quote content-type="editor-comment"><p>Line 177 - I'm not a big fan of the HMI acronym. Is there a LMI group? It seems simplistic to lump all of metabolism into dependent or independent, which in reality will differ depending on the specific substrate, the growth condition, and the strain.</p></disp-quote><p>While we are sorry that our study failed to provide the reviewer with a term they could be a fan of, their input did not change our view that HMI, an acronym we have adapted from a previously peer-reviewed study (doi:10.1186/s13059-023-02924-x), is a powerfully simplistic means to describe a phenomenon we observe and demonstrate in multiple different ways with our extensive analyses. The argument that HMI or LMI status will differ given the growth condition, substrate availability, or strain differences is not helping this case either: our analyses cut across a large number of humans and naturally occurring microbial systems in their guts that are exposed to largely variable ‘growth conditions’ and ‘substrates’ and composed of many strain variants of similar populations. Yet, we observe a clear role for HMI despite all these differences. Perhaps it is because HMI simply describes a higher metabolic capacity based on a defined subset of largely biosynthetic pathways that we observe to be consistently enriched in a large dataset covering a large variety of host, environmental and diet factors and indicates that a population has a higher metabolic capacity to not rely on ecosystem services. We show in our analysis that in the inflamed gut these capacities are indeed required, which is why HMI populations are enriched in IBD samples. HMI has no relation to any of the constraints mentioned by the reviewer, which is one of the major strengths of this metric.</p><disp-quote content-type="editor-comment"><p>Line 198 - It seems like a big assumption to state that efflux and drug resistance are unrelated to biosynthesis, as they could be genetically or even phenotypically linked.</p></disp-quote><p>We agree with the reviewer and are thankful for their input. We have weakened the assertion in this statement.</p><p>“These capacities may provide an advantage since antibiotics are a common treatment for IBDs (Nitzan et al. 2016), but are not necessarily related to the systematic enrichment of biosynthesis modules that likely provide resilience to general environmental stress rather than to a specific stressor such as antibiotics.”</p><disp-quote content-type="editor-comment"><p>Lines 202-218 - I'd suggest removing this paragraph. The &quot;non-IBD&quot; data introduces even more complications to the meta-analysis and seems irrelevant to the current study.</p></disp-quote><p>We thank the reviewer for this suggestion. Non-IBD data is important, but its relevance to the primary aims of the study is indeed negligible. We now have moved this paragraph to Supplementary File 1 (under the section “‘Non-IBD’ samples are intermediate to IBD and healthy samples”).</p><disp-quote content-type="editor-comment"><p>The health gradient is particularly problematic, putting cancer closer to healthy than IBD.</p></disp-quote><p>We took the reviewer’s advice and have swapped the order of the studies in Supplementary Figure 6 to place the cancer samples from Feng et al. closer to the IBD samples, on the other side of the non-IBD samples from the IBD studies.</p><disp-quote content-type="editor-comment"><p>Lines 235-257 - should trim this down and move to the discussion.</p></disp-quote><p>As mentioned above, we have opted for a “Results and Discussion format” for our manuscript, so we believe this discussion is in the correct place. We find it important to clearly highlight the limitations and potential biases of our work and trimming this text would take away from that goal.</p><disp-quote content-type="editor-comment"><p>Figure 3 - panels are out of order. Need to put the current panel D below current panel C. Also, relabel panel letters to go top to bottom (the bottom panel should be D). Could change current panel 3D to a violin plot to match current 3C.</p></disp-quote><p>We have updated Figure 3 by converting panel A into a new supplementary figure (Supplementary Figure 8), moving panels C and D below panel B, and relabeling the panels accordingly.</p><disp-quote content-type="editor-comment"><p>Figure 3B - this panel was incredibly useful and quite surprising to me in many respects. I would have assumed that the Bacteroides would be in the &quot;HMI&quot; bin. Is this a function of the specific strains included here? Was B. theta or B. fragilis included?</p></disp-quote><p>The reviewer makes an excellent observation that has been keeping us awake at night, yet somehow was not appropriately discussed in the text until their input. We are very thankful for their attention to detail here.</p><p>It is indeed true that <italic>Bacteroides</italic> genomes are often detected with increased abundance in individuals with IBD and likely have a survival advantage in the IBD gut environment, <italic>Bacteroides fragilis</italic> and <italic>Bacteroides thetaiotaomicron</italic> being some of the most dominant residents of the IBD gut. Their non-HMI status is not a function of which strains were included, since all taxa here are represented by the representative genomes available in the publicly available Genome Taxonomy Database. Their non-HMI status comes from the fact that they have HMI scores of around 24 to 26, which fall slightly below the threshold score of 26.4 that we used to classify genomes as HMI. This threshold is back-calculated from the metabolic completion requirement of at least 80% average completion of all 33 metabolic modules that are significantly enriched in IBD. So these genomes are right there at the edge, but not quite over it.</p><p>Thanks to this comment by our reviewer, we started wondering whether we should follow a more ‘literature-driven’ approach to set the threshold for HMI, rather than the 80% cutoff, and in fact attempted to lower the HMI score threshold to see if we could include more of the IBD-associated <italic>Bacteroides</italic> in the HMI bin. Author response table 1 below shows the relevant subset of our new Supplementary Table 3h, which describes the data from our tests on different thresholds.</p><table-wrap id="sa4table1" position="float"><label>Author response table 1.</label><caption><title>Number and proportion of <italic>Bacteroides</italic> genomes classified as HMI at each HMI score threshold.</title><p>There were 20 total <italic>Bacteroides</italic> genomes in the set of 338 gut microbes identified from the GTDB. The HMI score is computed by adding the percent completeness of all 33 IBD-enriched KEGG modules. The full table can be viewed in Supplementary Table 3h.</p></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Average percent<break/>completeness of</th><th valign="bottom">Corresponding HMI<break/>score threshold</th><th valign="bottom">Number of</th><th valign="bottom"/></tr></thead><tbody><tr><td align="left" valign="bottom">IBD-enriched<break/>modules</td><td align="left" valign="bottom">24.75</td><td align="left" valign="bottom">Bacteroides genomes<break/>classified as HMI</td><td align="left" valign="bottom">Hercent of Bacteroides<break/>genomes classified as</td></tr><tr><td align="left" valign="bottom">75</td><td align="left" valign="bottom">25.08</td><td align="left" valign="bottom">6</td><td align="left" valign="bottom">30%</td></tr><tr><td align="left" valign="bottom">76</td><td align="left" valign="bottom">25.41</td><td align="left" valign="bottom">4</td><td align="left" valign="bottom">20%</td></tr><tr><td align="left" valign="bottom">77</td><td align="left" valign="bottom">25.74</td><td align="left" valign="bottom">3</td><td align="left" valign="bottom">15%</td></tr><tr><td align="left" valign="bottom">78</td><td align="left" valign="bottom">26.07</td><td align="left" valign="bottom">0</td><td align="left" valign="bottom">0%</td></tr><tr><td align="left" valign="bottom">79</td><td align="left" valign="bottom">26.4</td><td align="left" valign="bottom">0</td><td align="left" valign="bottom">0%</td></tr><tr><td align="left" valign="bottom">80</td><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr></tbody></table></table-wrap><p>Lowering the threshold to 24.75, which corresponds to an average of 75% completeness in the 33 IBD-enriched modules, enabled the classification of 6 <italic>Bacteroides</italic> genomes as HMI, including <italic>B. fragilis</italic>, <italic>B. intestinalis</italic>, <italic>B. theta,</italic> and <italic>B. faecis</italic>. However, it also identified several microbes that are not IBD-associated as HMI, including 75 genomes from the Lachnospiraceae family and 18 genomes from the Ruminococcaceae family. In the latter family, several <italic>Faecalibacterium</italic> genomes, including 10 representatives of <italic>Faecalibacterium prausnitzii</italic>, were considered HMI using this threshold. These microbes are empirically known to decrease in abundance during inflammatory gastrointestinal conditions (doi:10.3390/microorganisms8040573, doi:10.1093/femsre/fuad039), and therefore these genomes should not be considered HMI – at least not under the working definition of HMI used in our study. To avoid including such a large number of obvious false positives in the HMI bin, we decided to maintain a higher threshold despite the exclusion of <italic>Bacteroides</italic> genomes.</p><p>This outcome demonstrates that our reductionist approach does not successfully capture every microbial population that is associated with IBD. Nevertheless, and in our opinion very surprisingly, the metric does capture a very large proportion of genomes with increased detection and abundance in IBD samples, as demonstrated by the peaks of detection/abundance that match to HMI status Author response image 1.</p><fig id="sa4fig1" position="float"><label>Author response image 1.</label><caption><title>Screenshots of Figure 3 that demonstrate the overlapping signal between HMI status and genome detection/abundance in IBD.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89862-sa4-fig1-v1.tif"/></fig><p>Furthermore, the violin plots in Figure 3B (formerly Figure 3C) clearly reflect the increased representation of HMI populations in IBD metagenomes. Although our classification method is imperfect, it still demonstrates the predictive power of metabolic competencies in identifying which microbes will survive in stressful gut environments. To ensure that readers recognize the crude nature of this classification strategy and the possibility that high metabolic independence can be achieved in different ways, we have added the following sentences to the relevant section of our manuscript:</p><p>“Given the number of ways a genome can pass or fail this threshold, this arbitrary cut-off has significant shortcomings, which was demonstrated by the fact that several species in the Bacteroides group were not classified as HMI despite their frequent dominance of the gut microbiome of individuals with IBD (Saitoh et al. 2002; Wexler 2007; Vineis et al. 2016) (Supplementary File 1). That said, the genomes that were classified as HMI by this approach were consistently higher in their detection and abundance in IBD samples (Figure 3a). It is likely that there are multiple ways to have high metabolic independence which are not fully captured by the 33 IBD-enriched metabolic modules identified in this study.”</p><p>We have also included a discussion of these findings in Supplementary Information File 1 (see section “Examining the impact of different HMI score thresholds on genome-level results”).</p><disp-quote content-type="editor-comment"><p>This panel also makes it clear that many of these modules are widespread in all genomes and thus unlikely to meaningfully differ in the microbiome. It would be interesting to use this type of analysis to identify a subset of KEGG modules with high variability between strains.</p></disp-quote><p>The figure makes it ‘look like’ many of these modules are widespread in all genomes and thus unlikely to meaningfully differ in the microbiome, but our quantitative analyses clearly demonstrate that these modules indeed differ meaningfully between microbiomes of healthy individuals and those diagnosed with IBD. For instance, the classifier that we built relying exclusively upon these modules’ PPCN values was able to reliably distinguish between the healthy and IBD sample groups in our dataset. The fact that the differentiating signal does not rely on rare metabolic or signature modules is what makes the classifier powerful enough to differentiate between “healthy” and “stressed” microbiomes in 86% of cases. Modules that are by nature less common could not serve this purpose. That said, we do agree with the reviewer that it might be interesting to study variability of KEGG modules as a function of variability between strains. This does not fall into the scope of this work, but we hope to assist others with the technical aspects of such work.</p><p>Considering the entirety of the exchange in this section, perhaps there is a broader discussion to be had around this topic. In retrospect, not being able to perfectly split microbes into two groups that completely recapitulate their enrichment in healthy or IBD samples by a crude metric and an arbitrary threshold is not surprising at all. What is surprising is that such a crude metric in fact works for the vast majority of microbes and predicts their increased presence in the IBD gut by only considering their genetic make up. In some respects, we believe that the inability of this cutoff to propose a perfect classifier is similar to the limited power of metabolic independence concept and the classes of HMI or LMI to capture and fully explain microbial fitness in health and disease. What is again surprising here is that these almost offensively simple classes do capture more than what one would expect. We can envision a few ways to implement a more sophisticated HMI/LMI classifier, and it is certainly an important task that is achievable. However, we are hopeful that this technical work can also be done better by others in our field, and that step forward, along with further scrutinizing the relevance of HMI/LMI classes to understand metabolic factors that contribute to the biodiversity of stressful environments, will have to remain as future work.</p><p>We thank the reviewer again for their comment here and pushing us to think more carefully and address the oddity regarding the poor representation of <italic>Bacteroides</italic> as HMI by our cutoff.</p><disp-quote content-type="editor-comment"><p>Given that a lot of the gaps are in the Firmicutes, this panel also makes me more concerned about annotation bias. How many of these gaps are real?</p></disp-quote><p>Analyses relying on gene annotations all suffer equally from the potential for missannotation or missing annotations, which primarily result from limitations in our reference databases for functional data. For instance, the Hidden Markov models for microbial genes in the KEGG Ortholog database are generated from a curated set of gene sequences primarily originating from cultivable microorganisms and particularly from commonly-used model organisms; hence, they do not capture the full extent of sequence diversity observed in populations that are less well-represented in reference databases – a category which includes several Firmicutes, as the reviewer points out. For KEGG KOfams in particular, the precomputed bit score thresholds for distinguishing between ‘good’ and ‘bad’ matches to a given model are often too stringent to enable annotation of genes that are just slightly too divergent from the set of known sequences, thus resulting in missing annotations. Based on our experience with these sorts of issues, we implemented a heuristic that reduces the number of missing annotations for KOs and captures significantly more homologs than other state-of-the-art approaches, as described in doi:10.1101/2024.07.03.601779. We refer the reviewer to our response to the related public comment about annotation bias above, which includes additional details about our investigations of annotation bias in our data. In comparison to the current standard, the heuristic we implemented improves functional annotation results. However, neither our nor any other bioinformatic study that relies on functional gene annotation can exclude the potential for annotation bias.</p><disp-quote content-type="editor-comment"><p>Figure 3B plotting issues - need to use the full names of the modules; for example, M00844 is &quot;arginine biosynthesis, ornithine =&gt; arginine&quot;, which changes the interpretation. Need a key for the heatmap on the figure. The tree is difficult to see, needs a darker font.</p></disp-quote><p>We have darkened the lines of the tree and dendrogram, and added a legend for the heatmap gradient (see new version of Figure 3 above). Unfortunately, we could not fit the full names of the modules into the figure due to space constraints. However, the full module name and other relevant information can be found in Supplementary Table 2a, and the matrix of pathway completeness scores in these genomes (e.g., the values plotted in the heatmap) can be found in Supplementary Table 3b. We are not sure what the reviewer refers to when stating that “for example, M00844 is &quot;arginine biosynthesis, ornithine =&gt; arginine&quot;, which changes the interpretation”. There is no ambiguity regarding the identity of KEGG module M00844, which is arginine biosynthesis from ornithine.</p><disp-quote content-type="editor-comment"><p>Line 321 - more justification for the 80% cutoff is needed along with a sensitivity analysis to see if this choice matters for the key results.</p></disp-quote><p>Inspired by this comment, and the one above regarding the classification of <italic>Bacteroides</italic> genomes, we tested several HMI score thresholds ranging from 75% to 85% average completeness of the 33 IBD-enriched modules. For each threshold, we computed all the key statistics reported in this section of our paper, including the statistical tests. We found that the choice of HMI score threshold does not influence the overall conclusions drawn in this section of our manuscript. Author response table 2 below shows the relevant subset of our new Supplementary Table 3h, which describes the results for each threshold:</p><table-wrap id="sa4table2" position="float"><label>Author response table 2.</label><caption><title>Key genome-level results at each HMI score threshold.</title><p>The HMI score is computed by adding the percent completeness of all 33 IBD-enriched KEGG modules. WRS – Wilcoxon Rank Sum test; KW – Kruskal-Wallis test. The full table can be viewed in Supplementary Table 3h</p></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Average percent<break/>completeness of<break/>IBD-enriched modules</th><th valign="bottom">HMI<break/>score<break/>threshold</th><th valign="bottom">Number of<break/>HMI<break/>genomes</th><th valign="bottom">WRS p-value for HMI vs non-HMI detection in IBD</th><th valign="bottom">WRS p-value for<break/>HMI vs non-HMI detection in healthy</th><th valign="bottom">KW p-value for fraction of HMI in IBD vs nonIBD samples</th></tr></thead><tbody><tr><td align="left" valign="bottom">75</td><td align="left" valign="bottom">24.75</td><td align="left" valign="bottom">129</td><td align="left" valign="bottom">2.18E-08</td><td align="left" valign="bottom">0.001776512</td><td align="left" valign="bottom">1.24E-16</td></tr><tr><td align="left" valign="bottom">76</td><td align="left" valign="bottom">25.08</td><td align="left" valign="bottom">115</td><td align="left" valign="bottom">2.70E-08</td><td align="left" valign="bottom">0.002832713</td><td align="left" valign="bottom">2.54E-19</td></tr><tr><td align="left" valign="bottom">77</td><td align="left" valign="bottom">25.41</td><td align="left" valign="bottom">98</td><td align="left" valign="bottom">2.10E-06</td><td align="left" valign="bottom">0.012257189</td><td align="left" valign="bottom">1.60E-18</td></tr><tr><td align="left" valign="bottom">78</td><td align="left" valign="bottom">25.74</td><td align="left" valign="bottom">78</td><td align="left" valign="bottom">4.16E-06</td><td align="left" valign="bottom">0.087182579</td><td align="left" valign="bottom">4.36E-21</td></tr><tr><td align="left" valign="bottom">79</td><td align="left" valign="bottom">26.07</td><td align="left" valign="bottom">69</td><td align="left" valign="bottom">5.46E-06</td><td align="left" valign="bottom">0.078132573</td><td align="left" valign="bottom">1.75E-21</td></tr><tr><td align="left" valign="bottom">80</td><td align="left" valign="bottom">26.4</td><td align="left" valign="bottom">59</td><td align="left" valign="bottom">4.82E-06</td><td align="left" valign="bottom">0.267265165</td><td align="left" valign="bottom">8.98E-25</td></tr><tr><td align="left" valign="bottom">81</td><td align="left" valign="bottom">26.73</td><td align="left" valign="bottom">48</td><td align="left" valign="bottom">0.000168971</td><td align="left" valign="bottom">0.700660484</td><td align="left" valign="bottom">5.03E-25</td></tr><tr><td align="left" valign="bottom">82</td><td align="left" valign="bottom">27.06</td><td align="left" valign="bottom">39</td><td align="left" valign="bottom">0.000217924</td><td align="left" valign="bottom">0.836279996</td><td align="left" valign="bottom">2.89E-29</td></tr><tr><td align="left" valign="bottom">83</td><td align="left" valign="bottom">27.39</td><td align="left" valign="bottom">35</td><td align="left" valign="bottom">0.001750966</td><td align="left" valign="bottom">0.962660229</td><td align="left" valign="bottom">3.77E-30</td></tr><tr><td align="left" valign="bottom">84</td><td align="left" valign="bottom">27.72</td><td align="left" valign="bottom">28</td><td align="left" valign="bottom">0.024916547</td><td align="left" valign="bottom">0.997986302</td><td align="left" valign="bottom">5.66E-30</td></tr><tr><td align="left" valign="bottom">85</td><td align="left" valign="bottom">28.05</td><td align="left" valign="bottom">24</td><td align="left" valign="bottom">0.043823626</td><td align="left" valign="bottom">0.999220691</td><td align="left" valign="bottom">5.21E-30</td></tr></tbody></table></table-wrap><p>We’ve summarized these findings in a new section of Supplementary File 1 entitled “Examining the impact of different HMI score thresholds on genome-level results”. We copy below the relevant text for the reviewer’s convenience:</p><p>“Determining the HMI status of a given genome required us to set a threshold for the HMI score above which a genome would be considered to have high metabolic independence. We tested several different thresholds by varying the average percent completeness of the 33 IBD-enriched metabolic modules that we expected from the</p><p>‘HMI’ genomes from ≥ 75% (corresponding to an HMI score of ≥ 24.75) to ≥ 85% (corresponding to an HMI score of ≥ 28.05). For each threshold, we computed the same statistics and ran the same statistical tests as those reported in our main manuscript to assess the impact of these thresholds on the results (Supplementary Table 3h). At the highest threshold we tested (HMI score ≥ 28.05), a small proportion of the reference genomes (7%, or n = 24) were classified as HMI, so we did not test higher thresholds.</p><p>We found that the results from comparing HMI genomes to non-HMI genomes are similar regardless of which HMI score threshold is used to classify genomes into either group. No matter which HMI score threshold was used, the mean genome size and mean number of genes were higher for HMI genomes than for non-HMI genomes. On average, the HMI genomes were about 1 Mb larger and had 1,032 more gene calls than non-HMI genomes. We ran two Wilcoxon Rank Sum statistical tests to assess the following null hypotheses: (1) HMI genomes do not have higher detection in IBD samples than non-HMI genomes, and (2) HMI genomes do not have higher detection in healthy samples than non-HMI genomes. For both tests, the p-values decreased (grew more significant) as the HMI score threshold decreased due to the inclusion of more genomes in the HMI bin. The first test for higher detection of HMI genomes than non-HMI genomes in IBD samples yielded p-values less than α = 0.05 at all HMI score thresholds. The second test for higher detection of HMI genomes than non-HMI genomes in healthy samples yielded p-values less than α = 0.05 for the three lowest HMI score thresholds (HMI score ≥ 24.75, ≥ 25.08, or ≥ 25.41). However, irrespective of significance threshold and HMI score threshold, there was always far stronger evidence to reject the first null hypothesis than the second, given that the p-value for the first test in IBD samples was 1 to 5 orders of magnitude lower (more significant) than the p-value for the second test in healthy samples.</p><p>IBD samples harbored a significantly higher fraction of genomes classified as HMI than healthy or non-IBD samples, regardless of HMI score threshold (p &lt; 1e-15, Kruskal-Wallis Rank Sum test). The p-values for this test increased (grew less significant) as the HMI score threshold decreased. This suggests that, at higher thresholds, relatively more genomes drop out of the HMI fraction in healthy/non-IBD samples than in IBD samples, thereby leading to larger differences and more significant p-values. Consequently, the HMI scores of genomes detected in IBD samples must be higher than the HMI scores of genomes detected in the other sample groups – indeed, the average HMI score of genomes detected within at least one IBD sample is 24.75, while the average score of genomes detected within at least one healthy sample is 22.78. Within a given sample, the mean HMI score of genomes detected within that sample is higher for the IBD group than in the healthy group: the average per-sample mean HMI score is 25.14 across IBD samples compared to the average of 23.00 across healthy samples.”</p><disp-quote content-type="editor-comment"><p>Lines 357 and 454 - I would remove the discussion of the &quot;gut environment&quot; which isn't really addressed here. The observed trends could just as easily relate to microbial interactions or the effects of diet and pharmaceuticals. Perhaps the issue is the vague nature of this term, which I read to imply changes in the mammalian host. Given the level of evidence, I'd opt to keep the options open and discuss what additional data would help resolve these questions.</p></disp-quote><p>We are in complete agreement with the reviewer that microbial interactions are likely an important driver of our observations. In healthy communities, microbial cross-feeding enables microbes with lower metabolic independence to establish and increase microbial diversity. Which is exactly why we are stating that “Community-level signal translates to individual microbial populations and provides insights into the microbial ecology of stressed gut environments”.</p><p>Diet or usage of prescription drugs on the other hand, as discussed previously, likely varies substantially over the various cohorts investigated, and is thus not a driver of the observed trends. Instead, HMI works as a high level indicator that is not influenced by these variable host habits.</p><disp-quote content-type="editor-comment"><p>Lines 354-394 - Could remove or dramatically trim down this text. Too much discussion for a results section.</p></disp-quote><p>We kindly remind the reviewer that our manuscript is written following a “Results and Discussion” format. This section provides necessary context and justification for our classifier implementation, so we have left it as-is.</p><disp-quote content-type="editor-comment"><p>Lines 395-441 - This section raised a lot of issues and could be qualified or even removed. The model was trained on modules that were IBD-associated in the same dataset, so it's not surprising that it worked. An independent test set would be required to see if this model has any broader utility.</p></disp-quote><p>The point that we selected the IBD-enriched modules as features should not raise any concerns, as these modules would have emerged as the most important (ie, most highly weighted) features in our model even if we had included all modules in our training data. This is because machine learning classifiers by design pick out the features that best distinguish between classes, and the 33 IBD-associated modules are a selective subset of these (if they were not, they would not have been significantly enriched in the IBD sample group). That said, a carefully conducted feature selection process prior to model training is a standard best-practice in machine learning; thus, if anything, this should be interpreted as a point of confidence rather than a concern. Furthermore, we evaluated our model using cross-validation, a standard practice in the machine learning field that assesses the stability of model performance by training and testing the model on different subsets of the data. This effort established that the model is robust across different inputs as demonstrated by the per-fold confusion matrix and the ROC curve. These are all standard approaches in machine learning to quantify the model tradeoff between bias and variance. As for the independent test set, we went far and beyond, and applied our model to the antibiotic time-series dataset described later in this section, which, in our opinion, and likely also in the opinion of many experts, serves as one of the most convincing ways to test the utility of any model. Classification results here show that our hypothesis concerning the relevance of metabolic independence to microbial survival in stressed gut environments applies beyond the IBD case and includes antibiotic use, which is indeed a stronger validation for this hypothesis than any test we could have done on other IBD-related datasets. Regardless, we agree that any ‘broader’ utility of our model, such as its applications in clinical settings for diagnostic purposes, is something we certainly can not make strong claims about without more data. We have therefore qualified this section by adding the following sentence:</p><p>“Determining whether such a model has broader utility as a diagnostic tool requires further research and validation; however, these results demonstrate the potential of HMI as an accessible diagnostic marker of IBD.”</p><disp-quote content-type="editor-comment"><p>The application to the antibiotic intervention data raises additional concerns, as the model will predict IBD (labeled &quot;stress&quot; in Figure 5) where none exists.</p></disp-quote><p>We apologize for this misunderstanding. The label “stress” actually means stress, not IBD. The figure the reviewer is referring to demonstrates that metabolic modules enriched in the gut microbiome of IBD patients are also temporarily enriched in the gut microbiome of healthy individuals treated with antibiotics for the duration of the treatment. While the classifier uses PPCN values for 33 metabolic modules enriched in microbiomes of IBD patients, it does not mean that this enrichment is exclusive to IBD. The classifier will distinguish between metagenomes in which the PPCN values for those 33 metabolic modules is higher and metagenomes in which the PPCN values are lower. Hence, our analysis demonstrates that during antibiotic usage in healthy individuals, the PPCN values of these 33 metabolic modules spike <italic>in a similar fashion</italic> to how they would in the gut community of a person with IBD. This points to a more general trend of high metabolic independence as a factor supporting microbial survival in conditions of stress; that is, the increase in metabolic independence is not specific to the IBD condition but rather a more generic ecological response to perturbations in the gut microbial community. We have clarified this point with the following addition to the paragraph summarizing these results:</p><p>“All pre-treatment samples were classified as ‘healthy’ followed by a decline in the proportion of ‘healthy’ samples to a minimum 8 days post-treatment, and a gradual increase until 180 days post treatment, when over 90% of samples were classified as ‘healthy’ (Figure 5, Supplementary Table 4b). In other words, the increase in the HMI metric serves as an indicator of stress in the gut microbiome, regardless of whether that stress arises from the IBD condition or the application of antibiotics. These observations support the role of HMI as an ecological driver of microbial resilience during gut stress caused by a variety of environmental perturbations and demonstrate its diagnostic power in reflecting gut microbiome state.”</p><p>We’ve also added the following sentence to the end of the legend for Figure 5:</p><p>“Samples classified as ‘healthy’ by the model were considered to have ‘no stress’ (blue), while samples classified as ‘IBD’ were considered to be under ‘stress’ (red).”</p><disp-quote content-type="editor-comment"><p>Figure S5A - should probably split this into 2 graphs since different data is analyzed.</p></disp-quote><p>It is true that different sets of modules are used in either half of the figure; however, there is a significant amount of overlap between the sets (17 modules), which is why there are lines connecting the points for the same module as described in the figure legend. We are using this figure to make the point that the median PPCN value of each module increases, in both sets of modules, from the healthy sample group to the IBD sample group. Therefore, we believe the current presentation is appropriate.</p><disp-quote content-type="editor-comment"><p>Figure S6A – this shows a substantial study effect and raises concerns about reproducibility.</p></disp-quote><p>We examined potential batch effects in Supplementary Information File 1 (see section “Considerations of Batch Effect”), and found that any study effect was minor and overcome by the signal between groups:</p><p>“The similar distribution of the median normalized copy number for each of the 33 IBD-enriched metabolic modules (summarized across all samples within a given study), across all studies within a given sample group (Supplementary Figure 6b), confirms that the sample group explains more of the trend than the study of origin.”</p><p>Furthermore, within Supplementary Figure 6a, there is a clear increase between the non-IBD controls from Franzosa et al. 2018 and the IBD samples from the same study, as well as between the non-IBD controls from Schirmir et al. 2018 and the IBD samples from that study. As there is no study effect influencing those two comparisons, this reinforces the evidence that there is a true increase in the normalized copy numbers of these modules when comparing samples from more healthy individuals to those from less healthy individuals.</p><disp-quote content-type="editor-comment"><p>Figure S7B - check numbers, which I think should sum to 33.</p></disp-quote><p>The numbers should not sum to 33. In this test to determine whether the two largest studies had excessive influence on the identity of the IBD-enriched modules, we repeated our strategy to obtain 33 IBD-enriched modules (those with the 33 smallest p-values from the statistical test) from each set of samples – either (1) samples from Le Chatelier et al. 2013 and Vineis et al. 2016, or (2) samples that are not from those two studies. The 2 sets, containing 33 modules each, gives us a total of 66 IBD-enriched modules. By comparing those two sets, we found that 20 modules were present in both sets – hence the value of 20 in the center of the Venn Diagram. In each set, 13 modules were unique – hence the value of 13 on either side. 13 + 13 + 2*20 = 66 total modules.</p><p>We again thank our reviewers for their time and interest, and invaluable input.</p></body></sub-article></article>