<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.2 20190208//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.2" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">71105</article-id><article-id pub-id-type="doi">10.7554/eLife.71105</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Microbiology and Infectious Disease</subject></subj-group><subj-group subj-group-type="heading"><subject>Physics of Living Systems</subject></subj-group></article-categories><title-group><article-title>Gut bacterial aggregates as living gels</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-243142"><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-2280-0132</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="other" rid="fund7"/><xref ref-type="other" rid="fund8"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-15465"><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-6006-4749</contrib-id><email>raghu@uoregon.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund7"/><xref ref-type="other" rid="fund8"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Department of Physics, Institute of Molecular Biology, and Materials Science Institute, University of Oregon</institution><addr-line><named-content content-type="city">Eugene</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Physics and Department of Molecular and Cell Biology, University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Wood</surname><given-names>Kevin B</given-names></name><role>Reviewing Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Garrett</surname><given-names>Wendy S</given-names></name><role>Senior Editor</role><aff><institution>Harvard T.H. Chan School of Public Health</institution><country>United States</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>07</day><month>09</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e71105</elocation-id><history><date date-type="received" iso-8601-date="2021-06-09"><day>09</day><month>06</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2021-09-06"><day>06</day><month>09</month><year>2021</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2021-06-08"><day>08</day><month>06</month><year>2021</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2021.06.08.447595"/></event></pub-history><permissions><copyright-statement>© 2021, Schlomann and Parthasarathy</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Schlomann and Parthasarathy</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-71105-v4.pdf"/><abstract><p>The spatial organization of gut microbiota influences both microbial abundances and host-microbe interactions, but the underlying rules relating bacterial dynamics to large-scale structure remain unclear. To this end, we studied experimentally and theoretically the formation of three-dimensional bacterial clusters, a key parameter controlling susceptibility to intestinal transport and access to the epithelium. Inspired by models of structure formation in soft materials, we sought to understand how the distribution of gut bacterial cluster sizes emerges from bacterial-scale kinetics. Analyzing imaging-derived data on cluster sizes for eight different bacterial strains in the larval zebrafish gut, we find a common family of size distributions that decay approximately as power laws with exponents close to −2, becoming shallower for large clusters in a strain-dependent manner. We show that this type of distribution arises naturally from a Yule-Simons-type process in which bacteria grow within clusters and can escape from them, coupled to an aggregation process that tends to condense the system toward a single massive cluster, reminiscent of gel formation. Together, these results point to the existence of general, biophysical principles governing the spatial organization of the gut microbiome that may be useful for inferring fast-timescale dynamics that are experimentally inaccessible.</p></abstract><abstract abstract-type="executive-summary"><title>eLife digest</title><p>The human gut is home to vast numbers of bacteria that grow, compete and cooperate in a dynamic, densely packed space. The spatial arrangement of organisms – for example, if they are clumped together or broadly dispersed – plays a major role in all ecosystems; but how bacteria are organized in the human gut remains mysterious and difficult to investigate.</p><p>Zebrafish larvae provide a powerful tool for studying microbes in the gut, as they are optically transparent and anatomically similar to other vertebrates, including humans. Furthermore, zebrafish can be easily manipulated so that one species of bacteria can be studied at a time.</p><p>To investigate whether individual bacterial species are arranged in similar ways, Scholmann and Parthasarathy exposed zebrafish with no gut bacteria to one of eight different strains. Each species was then monitored using three-dimensional microscopy to see how the population shaped itself into clusters (or colonies).</p><p>Schlomann and Parthasarathy used this data to build a mathematical model that can predict the size of the clusters formed by different gut bacteria. This revealed that the spatial arrangement of each species depended on the same biological processes: bacterial growth, aggregation and fragmentation of clusters, and expulsion from the gut.</p><p>These new details about how bacteria are organized in zebrafish may help scientists learn more about gut health in humans. Although it is not possible to peer into the human gut and watch how bacteria behave, scientists could use the same analysis method to study the size of bacterial colonies in fecal samples. This may provide further clues about how microbes are spatially arranged in the human gut and the biological processes underlying this formation.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>gut microbiota</kwd><kwd>zebrafish</kwd><kwd>aggregation</kwd><kwd>cluster size</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Zebrafish</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>P50GM09891</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>P01GM125576</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>F32AI112094</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32GM007759</award-id><principal-award-recipient><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>1427957</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000913</institution-id><institution>James S. McDonnell Foundation</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100001201</institution-id><institution>Kavli Foundation</institution></institution-wrap></funding-source><award-id>Kavli Microbiome Ideas Challenge</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><award-group id="fund8"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>P01HD22486</award-id><principal-award-recipient><name><surname>Schlomann</surname><given-names>Brandon H</given-names></name><name><surname>Parthasarathy</surname><given-names>Raghuveer</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>A theory of gut bacterial aggregation produces a cluster size distribution that matches that of several strains observed in zebrafish, suggesting principles generally applicable to the vertebrate gut.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The bacteria inhabiting the gastrointestinal tracts of humans and other animals make up some of the densest and most diverse microbial ecosystems on Earth (<xref ref-type="bibr" rid="bib19">Lloyd-Price et al., 2017</xref>; <xref ref-type="bibr" rid="bib33">Sender et al., 2016</xref>). In both macroecological contexts and non-gut microbial ecosystems, spatial organization is well known to impact both intra- and inter-species interactions (<xref ref-type="bibr" rid="bib23">McNally et al., 2017</xref>; <xref ref-type="bibr" rid="bib36">Tilman and Kareiva, 2018</xref>; <xref ref-type="bibr" rid="bib40">Weiner et al., 2019</xref>). This general principle is likely to apply in the intestine as well, and the spatial structure of the gut microbiome is increasingly proposed as an important factor influencing both microbial population dynamics and health-relevant host processes (<xref ref-type="bibr" rid="bib37">Tropini et al., 2017</xref>; <xref ref-type="bibr" rid="bib7">Donaldson et al., 2016</xref>). Moreover, recent work has uncovered strong and specific consequences of spatial organization in the gut, such as proximity of bacteria to the epithelial boundary determining the strength of host-microbe interactions (<xref ref-type="bibr" rid="bib38">Vaishnava et al., 2011</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), and antibiotic-induced changes in aggregation causing large declines in gut bacterial abundance (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). Despite its importance, the physical organization of bacteria within the intestine remains poorly understood, in terms of both in vivo data that characterize spatial structure and quantitative models that explain the mechanisms by which structure arises.</p><p>Recent advances in the ability to image gut microbial communities in model animals have begun to reveal features of bacterial spatial organization common to multiple host species. Bacteria in the gut exist predominantly in the form of three-dimensional, multicellular aggregates, likely encased in mucus, whose sizes can span several orders of magnitude. Such aggregates have been observed in mice (<xref ref-type="bibr" rid="bib24">Moor et al., 2017</xref>), fruit flies (<xref ref-type="bibr" rid="bib15">Koyama et al., 2020</xref>), and zebrafish (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>; <xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), as well as in human fecal samples (<xref ref-type="bibr" rid="bib39">van der Waaij et al., 1996</xref>). However, an understanding of the processes that generate these structures is lacking.</p><p>The statistical distribution of object sizes can provide powerful insights into underlying generative mechanisms, a perspective that has long been applied to datasets as diverse as galaxy cluster sizes (<xref ref-type="bibr" rid="bib11">Hansen et al., 2005</xref>), droplet sizes in emulsions (<xref ref-type="bibr" rid="bib18">Lifshitz and Slyozov, 1961</xref>), allele frequency distributions in population genetics (<xref ref-type="bibr" rid="bib25">Neher and Hallatschek, 2013</xref>), immune receptor repertoires (<xref ref-type="bibr" rid="bib26">Nourmohammad et al., 2019</xref>), species abundance distributions in ecology (<xref ref-type="bibr" rid="bib12">Hubbell, 1997</xref>), protein aggregates within cells (<xref ref-type="bibr" rid="bib10">Greenfield et al., 2009</xref>), and linear chains of bacteria generated by antibody binding (<xref ref-type="bibr" rid="bib2">Bansept et al., 2019</xref>). A classic example of the understanding provided by examining size distributions comes from the study of gels. In polymer solutions, random thermal motion opposes the adhesion of molecules, resulting in cluster size distributions dominated by monomers and small clusters. Gels form as adhesion strength increases, and monomers stick to one another strongly enough to overcome thermal motion and form a giant connected cluster that spans the size of the system. This large-scale connectivity gives gels their familiar stiffness as seen, for example, in the wobbling of a set custard. Theoretical tools from statistical mechanics and the study of phase transitions relate the cluster size distribution to the inter-monomer attraction strength and the temperature (<xref ref-type="bibr" rid="bib16">Krapivsky et al., 2010</xref>). In addition to providing an example of the utility of analyzing size distributions, gels in particular are a ubiquitous state of matter in living systems whose physical properties influence a wide range of activities such as protection at intestinal mucosal barriers (<xref ref-type="bibr" rid="bib6">Datta et al., 2016</xref>) and transport of molecules through amyloid plaques (<xref ref-type="bibr" rid="bib44">Woodard et al., 2014</xref>).</p><p>Motivated by these analogies, we sought to understand the distribution of three-dimensional bacterial cluster sizes in the living vertebrate gut, aiming especially to construct a quantitative theory that connects bacterial-scale dynamics to global size distributions. Such a model could be used to infer dynamical information in systems that are not amenable to direct observation, such as the human gut. Identifying key processes that are conserved across animal hosts would further our ability to translate findings in model organisms to human health-related problems. At a finer level, validated mathematical models could be used to infer model parameters of specific bacterial species of interest, for example pathogenic invaders or deliberately introduced probiotic species, by measuring their cluster size distribution.</p><p>We analyzed bacterial cluster sizes obtained from recent imaging-based studies of the larval zebrafish intestine (<xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>). As detailed below, we find a common family of cluster size distributions with bacterial species-specific features. We show that these distributions arise naturally in a minimal model of bacterial dynamics that is supported by direct observation. The core mechanism of this model involves growth together with a fragmentation process in which single cells leave larger aggregates. Strikingly, this process can be mapped exactly onto population genetics models of mutation, with cluster size analogous to allele frequency and single-cell fragmentation analogous to mutation. The combination of growth and fragmentation generates size distributions with power law tails, consistent with the data. This process also maps onto classic network models of preferential attachment (<xref ref-type="bibr" rid="bib3">Barabasi and Albert, 1999</xref>). Further, we show that cluster aggregation can generate an overabundance of large clusters through a process analogous to the sol-gel transition in polymer and colloidal systems, leading to plateaus in the size distribution that are observed in the data. These features of the size distribution are robust to the inclusion of a finite carrying capacity that limits growth and cluster loss due to expulsion from the intestine. In summary, we find that gut microbiota can be described mathematically as 'living gels', combining the statistical features of evolutionary dynamics with those of soft materials. Based on the generality of our model and our observations across several different bacterial species, we predict that this family of size distributions is universal across animal hosts, and we provide suggestions for testing this prediction in various systems.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Different bacterial species share a common family of broad cluster size distributions in the larval zebrafish intestine</title><p>We combined and analyzed previously generated datasets of gut bacterial cluster sizes in larval zebrafish (<xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>). In these experiments, zebrafish were reared devoid of any microbes, that is ‘germ-free’, and then mono-associated with a single, fluorescently labeled bacterial strain (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). After a 24 hr colonization period the complete intestines of live hosts were imaged with light sheet fluorescence microscopy (<xref ref-type="bibr" rid="bib14">Keller et al., 2008</xref>; <xref ref-type="bibr" rid="bib28">Parthasarathy, 2018</xref>; <xref ref-type="fig" rid="fig1">Figure 1B</xref>). Bacteria were identified in the images (<xref ref-type="fig" rid="fig1">Figure 1C</xref>) using a previously described image analysis pipeline (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>; <xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>). Single bacterial cells and multicellular aggregates were identified separately, and then the number of cells per multicellular aggregate was estimated by dividing the total fluorescence intensity of the aggregate by the mean intensity of single cells (Materials and methods).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Overview of experimental methods.</title><p>Larval zebrafish were derived germ-free and then monoassociated with single bacterial species (left). After 24 hr of colonization, images spanning the entire gut were acquired with light sheet fluorescence microscopy (middle). An example image of the anterior intestine is shown on the right, with instances of single cells and multicellular aggregates marked. The image is a maximum intensity projection of a 3D image stack. The approximate boundary of the gut is outlined in orange. Sizes of bacterial clusters were estimated with image analysis by separately identifying single cells and multicellular aggregates, and then normalizing the fluorescence intensity of aggregates by the mean single cell fluorescence.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig1-v4.tif"/></fig><p>In total, we characterized eight different bacterial strains, summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Six of the strains were isolated from healthy zebrafish (<xref ref-type="bibr" rid="bib35">Stephens et al., 2016</xref>) and then engineered to express fluorescent proteins (<xref ref-type="bibr" rid="bib42">Wiles et al., 2018</xref>), and two are genetically engineered knockout mutants of <italic>Vibrio</italic> ZWU0020, defective in motility (specifically, knockout of the two-gene operon encoding the polar flagellar motor, <italic>pomAB</italic>, referred to as ‘Δmot’) and chemotaxis (specifically, knockout of the histidine kinase <italic>cheA2</italic>, referred to as ‘Δche’), as described in reference (<xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>). The parent strain of these mutants, <italic>Vibrio</italic> ZWU0020, scarcely forms aggregates at all, existing primarily as single, highly motile cells (<xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), and so was excluded from this analysis. All strains are of the phylum Proteobacteria (<xref ref-type="bibr" rid="bib42">Wiles et al., 2018</xref>). A table of all cluster sizes by sample is included in <xref ref-type="supplementary-material" rid="fig2sdata1">Figure 2—source data 1</xref>.</p><p>We calculated for each bacterial strain the reverse cumulative distribution of cluster sizes, <inline-formula><mml:math id="inf1"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> , denoting the probability that an intestinal aggregate will contain more than <inline-formula><mml:math id="inf2"><mml:mi>n</mml:mi></mml:math></inline-formula> bacterial cells. We computed <inline-formula><mml:math id="inf3"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> separately for each animal (<xref ref-type="fig" rid="fig2">Figure 2</xref>, small circles) and also pooled the sizes from different animals colonized by the same bacterial strain (<xref ref-type="fig" rid="fig2">Figure 2</xref>, large circles). There is substantial variation across fish, but the pooled distributions exhibit a well-defined average of the individual distributions. We also computed binned probability densities (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>), which show similar patterns, but focus our discussion on the cumulative distribution to circumvent technical issues related to bin sizes.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Different bacterial species exhibit similar cluster size distributions.</title><p>Reverse cumulative distributions, the probability that the cluster size is greater than <inline-formula><mml:math id="inf4"><mml:mi>n</mml:mi></mml:math></inline-formula> as a function of <inline-formula><mml:math id="inf5"><mml:mi>n</mml:mi></mml:math></inline-formula>, for eight bacterial strains in larval zebrafish intestines. Small circles connected by lines represent the distributions constructed from individual fish. Large circles are from pooled data from all fish. The dashed line represents <inline-formula><mml:math id="inf6"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and is a guide to the eye. Bottom right panel shows the pooled distributions for each strain as solid lines.</p><p><supplementary-material id="fig2sdata1"><label>Figure 2—source data 1.</label><caption><title>Spreadsheet with all cluster sizes by strain.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-71105-fig2-data1-v4.xlsx"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig2-v4.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Cluster size distributions as probability densities.</title><p>Different bacterial species exhibit similar cluster size distributions. Probability densities for eight bacterial strains monoassociated in larval zebrafish intestines. Small circles connected by lines represent the distributions constructed for each fish. Large circles are the result of pooling together sizes from all fish. Dashed line represents <inline-formula><mml:math id="inf7"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and is a guide to the eye. Bottom right panel shows the pooled distributions for each strain as solid lines. Summary of data is given in <xref ref-type="table" rid="table1">Table 1</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig2-figsupp1-v4.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Images of individual <inline-formula><mml:math id="inf8"><mml:mi>z</mml:mi></mml:math></inline-formula>-slices showing mild heterogeneity of fluorescence intensity within aggregates.</title><p>Fluorescence intensity is mostly homogeneous within clusters, although small dark regions do occur. Three individual z slices of a fish colonized with <italic>Enterobacter</italic> are shown. The approximate gut boundary is outlined in orange. Dark regions within the cluster are noted with white arrows. The approximate location of the field of view within the animal is noted with a dashed black box on the fish cartoon.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig2-figsupp2-v4.tif"/></fig></fig-group><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Summary of cluster data by bacterial strain.</title><p>Each row corresponds to one of the bacterial strains included in this study. Entries include strain name, total number of fish colonized with that strain, total number of clusters identified across all fish, and the original publication that the data was pulled from.</p></caption><table frame="hsides" rules="groups"><thead><tr><th>Bacterial strain</th><th>Number of fish</th><th>Number of clusters</th><th>Source publication</th></tr></thead><tbody><tr><td><italic>Aeromonas</italic> ZOR0001</td><td>6</td><td>445</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref></td></tr><tr><td><italic>Aeromonas</italic> ZOR0002</td><td>6</td><td>1901</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref></td></tr><tr><td><italic>Enterobacter</italic> ZOR0014</td><td>18</td><td>3597</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref></td></tr><tr><td><italic>Plesiomonas</italic> ZOR0011</td><td>3</td><td>223</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref></td></tr><tr><td><italic>Pseudomonas</italic> ZWU0006</td><td>6</td><td>133</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref></td></tr><tr><td><italic>Vibrio</italic> ZOR0036</td><td>6</td><td>2430</td><td><xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref></td></tr><tr><td><italic>Vibrio</italic> ZWU0020 Δmot</td><td>11</td><td>5888</td><td><xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref></td></tr><tr><td><italic>Vibrio</italic> ZWU0020 Δche</td><td>11</td><td>3551</td><td><xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref></td></tr></tbody></table></table-wrap><p>We find broad distributions of <inline-formula><mml:math id="inf9"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> across all strains (<xref ref-type="fig" rid="fig2">Figure 2</xref>, bottom right panel). For comparison, for each strain we overlay a dashed line representing the power law distribution <inline-formula><mml:math id="inf10"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. This <inline-formula><mml:math id="inf11"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is equivalent to a probability density of <inline-formula><mml:math id="inf12"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> since the latter is proportional to the derivative of the former. Each strain's cumulative distribution follows a similar power-law-like decay at low <inline-formula><mml:math id="inf13"><mml:mi>n</mml:mi></mml:math></inline-formula>, with an apparent exponent in the vicinity of -1, and then becomes shallower in a strain-dependent manner. For example, <italic>Aeromonas</italic> ZOR0002 has a quite straight distribution on a log-log plot (<xref ref-type="fig" rid="fig2">Figure 2</xref>, top row, middle column), while the distribution of <italic>Enterobacter</italic> ZOR0014 exhibits a plateau-like feature at large sizes (<xref ref-type="fig" rid="fig2">Figure 2</xref>, top row, right column). The mutant strains <italic>Vibrio</italic> ZWU0020 Δche and Δmot follow qualitatively similar distributions to the native strains (<xref ref-type="fig" rid="fig2">Figure 2</xref>, bottom row, left and middle columns).</p><p>We performed a sensitivity analysis and found that these two key features of the measured distributions—an initial power law-like decay with cumulative distribution exponent close to -1 and a strain-dependent plateau at large sizes—are robust to measurement error in enumeration of cluster sizes. For the initial decay of the distribution, the largest source of error is the misidentification of auto-fluorescent background as single cells. To assess the impact of our single-cell count uncertainty on the distribution, we fit a power law model to clusters sizes up to 100 cells two times: once including single cells and once considering only cells of size in the range 2–100 (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>, Materials and methods). In both fits we find cumulative distribution exponents consistent with −1 for most strains. The average exponent tended to decrease mildly when single cells were excluded from the fit (the distribution decayed more slowly), consistent with an over-estimation of the number of single cells, but the shifts were all within uncertainties. Estimates of distribution exponents from small sizes can easily be biased (<xref ref-type="bibr" rid="bib5">Clauset et al., 2009</xref>), so we performed our sensitivity analysis with two different methods: a linear fit to <inline-formula><mml:math id="inf14"><mml:mrow><mml:mi>log</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> vs. <inline-formula><mml:math id="inf15"><mml:mrow><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:math></inline-formula>, and maximum likelihood estimation (Materials and methods). The maximum likelihood estimate gave higher values than line-fitting, but the shifts upon removing single cells were within uncertainties for both methods.</p><p>For the large-size plateau, the existence of dim cells in the center of the aggregate, perhaps due to a state of low metabolic activity, would lead to an underestimate of total cluster size. Underestimating the size of large clusters would then result in a less extreme plateau; the plateaus we observe are therefore a lower bound. In cross-sections of large aggregates, we observe mostly homogeneous fluorescence, suggesting that this effect is mild, although small dark regions do occur (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). Whether these dark regions correspond to dead or inactive bacteria, mucus, or empty space, is not clear, although we note that small clumps of dead bacteria have been observed in expelled clusters via live/dead staining (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). Regardless of their origin, we conclude that these mild heterogeneities are unlikely to significantly alter the behavior of the size distributions, which span 4 orders of magnitude.</p><p>In summary, we find that different bacterial strains, which exhibit a variety of swimming and sticking behaviors (<xref ref-type="bibr" rid="bib42">Wiles et al., 2018</xref>; <xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>), abundances (<xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), and population dynamics (<xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), share a common family of cluster size distributions. This observation suggests that generic processes, rather than strain-specific ones, determine gut bacterial cluster sizes. Notably, these distributions are extremely broad, inconsistent with the exponential-tailed distributions found for linear chains of bacteria (<xref ref-type="bibr" rid="bib2">Bansept et al., 2019</xref>). We next sought to understand the kinetics that give rise to our measured cluster size distributions.</p><sec id="s2-1-1"><title>A growth-fragmentation process generates power-law distributions</title><p>Previous time-lapse imaging of bacteria in the zebrafish intestine revealed four core processes that can alter bacterial cluster sizes: (1) clusters can increase in size due to cell division, a process we refer to as ‘growth’; (2) clusters can decrease in size as single bacteria escape from them, a process we refer to as ‘fragmentation’ and believe to be linked to cell division at the surface; (3) clusters can increase in size by joining with another cluster during intestinal mixing, a process we refer to as ‘aggregation’; and (4) clusters can be removed from the system by transiting along and out of the intestine, a process we refer to as ‘expulsion’. The breakup of large clusters into medium ones appears to be rare in our system, so we ignore this process. The single cell fragmentation process we describe conserves cell number and is analogous to the ‘chipping’ kernel that has been used to describe the breaking off of monomers from the ends of linear polymers (<xref ref-type="bibr" rid="bib17">Krapivsky and Redner, 1996</xref>).</p><p>To understand how each of these process affect the distribution of cluster sizes, we used mathematical modeling. We attempted to construct a simple model that encoded these processes and retained salient biological and physical features. In our model, the relevant variable is a list of all cluster sizes, or equivalently, a list of the number of clusters of each size. Clusters can change size according to four reactions that correspond to each of the four processes listed above. There is no explicit spatial dependence in this model, but aspects of spatial structure, such as the fact that some cells in a cluster are confined to the center while others are on the surface, can be modeled by choosing how the rates of reactions depend on cluster size, as discussed below. We assume, however, that growth rates are the same for all cells within a cluster. Growth rates have been measured for seven strains to date and fall in the range of 0.3 to 0.8 hr<sup>-1</sup> (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>; <xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>); we use an intermediate value of 0.5 hr<sup>-1</sup> in all simulations below. In large systems, it is often valid to ignore fluctuations, in which case the model can be summarized by a single, deterministic equation for the likelihood of clusters of each size, for which analytic results are possible in some cases. In contrast, for small systems, which includes our experiments, random fluctuations will likely be relevant, and so we turn to computer simulations that capture stochastic dynamics.</p><p>We previously showed that a version of this model with all parameters measured (i.e., no remaining free parameters) generates a size distribution consistent with that of <italic>Enterobacter</italic> ZOR0014 (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). However, it was not clear which processes generated which features of the distribution, or how generalizable the model was. Therefore, we studied this model in more detail, starting from a simplified version and iteratively adding complexity.</p><p>The observation that all distributions appeared to be organized around <inline-formula><mml:math id="inf16"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> inspired us to consider connections to a classic populations genetics model that has this form for the distribution of allele frequencies, known as the Yule-Simons process (<xref ref-type="bibr" rid="bib45">Yule, 1925</xref>; <xref ref-type="bibr" rid="bib34">Simon, 1955</xref>; <xref ref-type="bibr" rid="bib1">Altan-Bonnet et al., 2020</xref>; <xref ref-type="bibr" rid="bib25">Neher and Hallatschek, 2013</xref>). An exponentially growing population subject to random neutral mutations that occur with probability <inline-formula><mml:math id="inf17"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> will amass an allele frequency distribution that follows <inline-formula><mml:math id="inf18"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>frequency</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>ϵ</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> for large sizes, with the limit to <inline-formula><mml:math id="inf19"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>frequency</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> for rare mutation. This heavy-tailed distribution reflects ‘jackpot’ events in which mutants that appear early rise to large frequencies through exponential growth. As long as mutation is rare compared to replication, this process robustly generates distributions <inline-formula><mml:math id="inf20"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, without the need for fine tuning of the microscopic details. We therefore saw it as an attractive hypothesis for generating similar size distributions across diverse bacterial species.</p><p>Analogously, the size of mutant clones maps onto the size of a bacterial cluster, and the mutation process that generates new clones maps onto the fragmentation process that generates new clusters (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). In situations where all cells in a cluster have the same probability of fragmenting, this analogy is exact and the same distribution emerges (Appendix). However, gut bacterial clusters are three-dimensional and likely encased in mucus (<xref ref-type="bibr" rid="bib39">van der Waaij et al., 1996</xref>), so spatial structure likely influences fragmentation rates. We hypothesized that this spatial structure could be a mechanism for generating distributions shallower than <inline-formula><mml:math id="inf21"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> that we observe in the data for large sizes (<xref ref-type="fig" rid="fig2">Figure 2</xref>) but that cannot be produced by the standard Yule-Simons mechanism. Therefore, we modified the Yule-Simons process by decoupling the growth and fragmentation processes and invoking a fragmentation rate, <inline-formula><mml:math id="inf22"><mml:msub><mml:mi>F</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:math></inline-formula>, that scales as a power of the cluster size, <inline-formula><mml:math id="inf23"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>∼</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup></mml:mrow></mml:mrow></mml:math></inline-formula> (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). A value of <inline-formula><mml:math id="inf24"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> corresponds to the well-mixed limit of the Yule-Simons process. A value of <inline-formula><mml:math id="inf25"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> corresponds to only bacteria on the surface of clusters being able to fragment. An extreme value of <inline-formula><mml:math id="inf26"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> means that all clusters have the same rate of fragmenting, regardless of their size, and can be thought of as representing a chain of cells where only the cells at ends of the chain can break off.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>A minimal model inspired by evolutionary dynamics generates power law distributions.</title><p>(<bold>A</bold>) Fragmentation is analogous to mutation and we can construct a genealogy that mirrors the physical structure of the clusters. (<bold>B</bold>) Summary of a growth/fragmentation process that includes the effect of spatially confined clusters. (<bold>C</bold>) Examples of reverse cumulative size distributions obtained from stochastic simulations of the model for different values of the fragmentation exponent, <inline-formula><mml:math id="inf27"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula>. The tails of the distribution are approximately power laws, defined as <inline-formula><mml:math id="inf28"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. Parameters: <inline-formula><mml:math id="inf29"><mml:mi>r</mml:mi></mml:math></inline-formula> = 0.5 hr<sup>-1</sup>, <inline-formula><mml:math id="inf30"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>0.4</mml:mn><mml:mo>,</mml:mo><mml:mn>0.2</mml:mn><mml:mo>,</mml:mo><mml:mn>0.167</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, for <inline-formula><mml:math id="inf31"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi/></mml:mrow></mml:math></inline-formula> 0, 2/3, 1, respectively, time <inline-formula><mml:math id="inf32"><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>24</mml:mn></mml:mrow></mml:math></inline-formula> hr, and the system was initialized with 10 single cells. (<bold>D</bold>) Dependence of the resulting distribution exponent, μ, on ratio of fragmentation to aggregation rate (<inline-formula><mml:math id="inf33"><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula>) and fragmentation exponent (<inline-formula><mml:math id="inf34"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula>). Markers show mean and standard deviation across 100 simulations. Solid lines are approximate analytic results (<xref ref-type="table" rid="table2">Table 2</xref>). Parameters: same as (<bold>C</bold>) with β varying.</p><p><supplementary-material id="fig3sdata1"><label>Figure 3—source data 1.</label><caption><title>Results of power-law fits to simulated distributions.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-71105-fig3-data1-v4.xlsx"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig3-v4.tif"/></fig><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>Analytic results for the minimal growth-fragmentation process.</title><p>Distribution exponent, μ, as a function of fragmentation exponent, <inline-formula><mml:math id="inf35"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula>, fragmentation rate, β, and growth rate, r, as plotted in <xref ref-type="fig" rid="fig3">Figure 3D</xref>. Results are expected to be valid for long times (<inline-formula><mml:math id="inf36"><mml:mrow><mml:mi>t</mml:mi><mml:mo>→</mml:mo><mml:mi mathvariant="normal">∞</mml:mi></mml:mrow></mml:math></inline-formula>), large sizes (<inline-formula><mml:math id="inf37"><mml:mrow><mml:mi>n</mml:mi><mml:mo>→</mml:mo><mml:mi mathvariant="normal">∞</mml:mi></mml:mrow></mml:math></inline-formula>), and slow fragmentation (<inline-formula><mml:math id="inf38"><mml:mrow><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mo>&lt;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>). See Appendix for details.</p></caption><table frame="hsides" rules="groups"><thead><tr><th/><th><inline-formula><mml:math id="inf39"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula></th><th><inline-formula><mml:math id="inf40"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula></th><th><inline-formula><mml:math id="inf41"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula></th></tr></thead><tbody><tr><td>distribution exponent, μ</td><td><inline-formula><mml:math id="inf42"><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf43"><mml:mrow><mml:mfrac><mml:mn>5</mml:mn><mml:mn>3</mml:mn></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mi>β</mml:mi><mml:mi>r</mml:mi></mml:mfrac></mml:mrow></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf44"><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></inline-formula></td></tr></tbody></table></table-wrap><p>In stochastic simulations of this model (Materials and methods) we find broad, power-law-like distributions for each value of <inline-formula><mml:math id="inf45"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula> (<xref ref-type="fig" rid="fig3">Figure 3C</xref>), but no signature of a shallow plateau at larger sizes. We define μ as the exponent of the probability distribution, <inline-formula><mml:math id="inf46"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, such that the cumulative distribution function has the form <inline-formula><mml:math id="inf47"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> (the latter is proportional to the integral of the former). Following established methods, we fit a power law, <inline-formula><mml:math id="inf48"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> for <inline-formula><mml:math id="inf49"><mml:mrow><mml:mi>n</mml:mi><mml:mo>&gt;</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mtext>min</mml:mtext></mml:msub></mml:mrow></mml:math></inline-formula> to simulation outputs using maximum likelihood estimation (<xref ref-type="bibr" rid="bib5">Clauset et al., 2009</xref>) and examined the dependence on fragmentation rate. Faster fragmentation results in larger values of μ, reflecting steeper distributions, with the dependence being superlinear for <inline-formula><mml:math id="inf50"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, approximately linear for <inline-formula><mml:math id="inf51"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, and sublinear for <inline-formula><mml:math id="inf52"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> (<xref ref-type="fig" rid="fig3">Figure 3D</xref>, circles). Increasing values of <inline-formula><mml:math id="inf53"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula> also appeared to have increasing minimum values of μ, corresponding to rare fragmentation.</p><p>The minimum value of the distribution exponent can be computed by considering, for example, the total rate of fragmentation events. Denoting the total number of clusters by <inline-formula><mml:math id="inf54"><mml:mi>M</mml:mi></mml:math></inline-formula> and the number of clusters of size <inline-formula><mml:math id="inf55"><mml:mi>n</mml:mi></mml:math></inline-formula> by <italic>c</italic><sub><italic>n</italic></sub>, the rate of cluster production follows <inline-formula><mml:math id="inf56"><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>≈</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mi>n</mml:mi></mml:msub><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> (Appendix). Assuming a power-law solution <inline-formula><mml:math id="inf57"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and approximating the sum by an integral, we see that the rate of cluster production is finite only if<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>&gt;</mml:mo><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>consistent with simulations. Therefore, spatial structure—modeled by decreasing <inline-formula><mml:math id="inf58"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula>—is indeed a mechanism to generate distributions shallower than <inline-formula><mml:math id="inf59"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. A heuristic argument for the rate dependence of the exponents in the long time, large size limit is provided in the Appendix, with the results summarized in <xref ref-type="table" rid="table2">Table 2</xref> and plotted as solid lines in <xref ref-type="fig" rid="fig3">Figure 3D</xref>. The analytic results agree reasonably well with simulations, with deviations becoming prominent as <inline-formula><mml:math id="inf60"><mml:mrow><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>.</p><p>In summary, we identified a minimal growth-fragmentation process that generates power-law distributions with tuneable exponents in the experimentally observed range. However, this model does not include other features known to occur in the experimental system, including a finite carrying capacity that limits growth, cluster aggregation, and cluster expulsion, which may alter the asymptotic distributions. Moreover, this model fails to capture the large-size behavior of many of the experimental distributions, which exhibit a plateau (<xref ref-type="fig" rid="fig2">Figure 2</xref>). Therefore, we investigated extensions of the model.</p></sec><sec id="s2-1-2"><title>Size-dependent aggregation enhances the abundance of large clusters</title><p>We explored a number of potential mechanisms for generating plateaus in the size distribution at large cluster sizes. As shown below, several plausible models fail to produce this feature. It emerges, however, from the incorporation of size-dependent aggregation rates.</p><p>First we tested whether finite time effects could introduce plateaus to the distributions of the minimal growth-fragmentation model, since our power-law solutions are only valid asymptotically. Indeed, stochastic simulations with <inline-formula><mml:math id="inf61"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> and rare fragmentation (<inline-formula><mml:math id="inf62"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, <inline-formula><mml:math id="inf63"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>) showed that for systems initialized with 10 single cells (a reasonable comparison with initial colonization in the experiments <xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>), slight curvature appears in the distribution that weakens with time but is still detectable at 24 hr (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, circles). We confirmed that this effect was solely due to dynamics and not to any finite system effect by numerically integrating the master equation for this model, which describes the deterministic dynamics of an infinite system yet agrees with the stochastic simulation results (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, lines; Materials and methods). However, the curvature observed at finite times is substantially smaller than what occurs for some of the strains, such as <italic>Enterobacter</italic> ZOR0014 and <italic>Vibrio</italic> ZOR0036, so we believe it is not the dominant effect.</p><p>We next asked whether including additional processes to the model could produce the plateau effect, focusing on stationary distributions. As discussed above, populations in the larval zebrafish gut are known to reach carrying capacities that halt growth (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>). Since we believe fragmentation is tied to growth, we modeled this as the fragmentation rate being slowed as the total number of cells, <inline-formula><mml:math id="inf64"><mml:mi>N</mml:mi></mml:math></inline-formula>, approaches carrying capacity, K, in the same way as the growth rate: <inline-formula><mml:math id="inf65"><mml:mrow><mml:mi>r</mml:mi><mml:mo>→</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>/</mml:mo><mml:mi>K</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf66"><mml:mrow><mml:mi>β</mml:mi><mml:mo>→</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>/</mml:mo><mml:mi>K</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. Carrying capacities have been estimated to range from 10<sup>3</sup>-10<sup>6</sup> cells, depending on the bacterial strain (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>; <xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>).</p><p>With this addition to the model, fragmentation halts in the steady state. However, in the larval zebrafish gut it has been well-documented that large bacterial aggregates are quasi-stochastically expelled out the intestine, after which exponential growth by the remaining cells is restarted (<xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>). We modeled expulsion by having clusters removed from the system altogether at a size-dependent rate <inline-formula><mml:math id="inf67"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:msup></mml:mrow></mml:mrow></mml:math></inline-formula>. It is unclear what value of the exponent <inline-formula><mml:math id="inf68"><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:math></inline-formula> best describes the experimental system, but previous studies measured expulsion rates for the largest clusters, typically of order <inline-formula><mml:math id="inf69"><mml:mrow><mml:mi>K</mml:mi><mml:mo>∼</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mn>3</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> cells, in the range of 0.07 to 0.11 hr<sup>-1</sup> (<xref ref-type="bibr" rid="bib41">Wiles et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). Therefore, we co-varied <inline-formula><mml:math id="inf70"><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf71"><mml:mi>λ</mml:mi></mml:math></inline-formula> such that <inline-formula><mml:math id="inf72"><mml:mrow><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>K</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:msup></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>. Combining finite carrying capacity and expulsion leads to a non-trivial stationary distribution of the model that lacks a plateau for <inline-formula><mml:math id="inf73"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi/></mml:mrow></mml:math></inline-formula> 0, 1/3, or 2/3 (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>).</p><p>Finally, we considered the effect of cluster aggregation, which has been directly observed in live imaging experiments (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). We model aggregation with pairwise interactions where clusters come together and form a single cluster with size equal to the sum of the individual sizes. The aggregation rate is allowed to be size-dependent with the homogenous kernel <inline-formula><mml:math id="inf74"><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi>α</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:msup></mml:mrow></mml:mrow></mml:math></inline-formula>. As with expulsion, it is not clear which exponent value is the most realistic. Accurate measurements of aggregation rates are lacking, but we estimate bounds to be between 1 and 100 total aggregation events per hour for a typical population (Materials and methods), so we consider only pairs of <inline-formula><mml:math id="inf75"><mml:mi>α</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf76"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula> that match these bounds. Further, an important theoretical distinction is that in purely aggregating systems, models with <inline-formula><mml:math id="inf77"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>≥</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> exhibit a finite-time singularity corresponding to a gelation transition, at which point the distribution acquires a power-law tail, while distributions have exponential tails when <inline-formula><mml:math id="inf78"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>&lt;</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib16">Krapivsky et al., 2010</xref>). We considered both regimes.</p><p>We added aggregation to our growth-driven process and arrived at the general model described in <xref ref-type="fig" rid="fig4">Figure 4A</xref>. Parameters are also summarized in <xref ref-type="table" rid="table3">Table 3</xref>. Strikingly, we found that increasing aggregation rate produces the large-size plateau seen in our data, but only when aggregation rate scales sufficiently quickly with cluster size (<xref ref-type="fig" rid="fig4">Figure 4B</xref>, right, <inline-formula><mml:math id="inf79"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula> = 2/3) and not when aggregation is size-independent (<xref ref-type="fig" rid="fig4">Figure 4B</xref>, left, <inline-formula><mml:math id="inf80"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>). A mild effect is observed for <inline-formula><mml:math id="inf81"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> (<xref ref-type="fig" rid="fig4">Figure 4B</xref>, middle). The largest plateau (<xref ref-type="fig" rid="fig4">Figure 4B</xref>, <inline-formula><mml:math id="inf82"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, highest curve) corresponds to 15 ± 3 (mean ± std. dev) total aggregation events per hour. This value is consistent with our rough experimental bounds of 1–100 hr<sup>-1</sup>.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Size-dependent aggregation introduces a plateau in the size distribution.</title><p>(<bold>A</bold>) Schematic of the generalized model. Parameters summarized in <xref ref-type="table" rid="table3">Table 3</xref>. (<bold>B</bold>) Reverse cumulative distributions obtained from simulations for different values of <inline-formula><mml:math id="inf83"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula> (left, middle, right) and <inline-formula><mml:math id="inf84"><mml:mi>α</mml:mi></mml:math></inline-formula> (different colored lines within each panel). Increasing aggregation produces a plateau if the aggregation depends strongly enough on cluster size. (<bold>C</bold>) The plateau arises only in stochastic simulation of finite systems with size-dependent aggregation. Solid lines are stochastic simulations, dashed lines are the result of numerically integrating the master equation. Parameters: <inline-formula><mml:math id="inf85"><mml:mi>r</mml:mi></mml:math></inline-formula> = 0.5 hr<sup>-1</sup>, <inline-formula><mml:math id="inf86"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf87"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, <inline-formula><mml:math id="inf88"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf89"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, <inline-formula><mml:math id="inf90"><mml:mrow><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mn>3</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, and the number of simulation was replicates = 150 per parameter set. For each value of <inline-formula><mml:math id="inf91"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula>, we considered <inline-formula><mml:math id="inf92"><mml:mi>α</mml:mi></mml:math></inline-formula> values of 0 (no aggregation) and then varied α logarithmically, with the following (min, max) values for <inline-formula><mml:math id="inf93"><mml:mrow><mml:msub><mml:mi>log</mml:mi><mml:mn>10</mml:mn></mml:msub><mml:mo>⁡</mml:mo><mml:mi>α</mml:mi></mml:mrow></mml:math></inline-formula>: (−4,–2) for <inline-formula><mml:math id="inf94"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>, (−4.5,–2.5) for <inline-formula><mml:math id="inf95"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, and (−5,–3) for <inline-formula><mml:math id="inf96"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig4-v4.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Distributions of growth/fragmentation process at short times.</title><p>Mild curvature appears in the minimal growth/fragmentation process distribution at finite time, but the result is inconsistent with experimental data. Circles are the result of stochastic simulation, solid lines are the result of numerical integration of the master equation. Color denotes simulation time. Dashed black line indicates <inline-formula><mml:math id="inf97"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and is a guide to the eye.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig4-figsupp1-v4.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Distributions for a process with density-dependent growth, fragmentation, and expulsion.</title><p>A modified process with carrying capacity and expulsion does not produce a plateau in the stationary size distribution. Reverse cumulative distributions computed from stochastic simulations with different values of the expulsion exponent, <inline-formula><mml:math id="inf98"><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:math></inline-formula>. For each value of <inline-formula><mml:math id="inf99"><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:math></inline-formula>, the expulsion rate, <inline-formula><mml:math id="inf100"><mml:mi>λ</mml:mi></mml:math></inline-formula>, was chosen such that for clusters of size <inline-formula><mml:math id="inf101"><mml:mrow><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:mn>1000</mml:mn></mml:mrow></mml:math></inline-formula> cells, <inline-formula><mml:math id="inf102"><mml:mrow><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>K</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:msup></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.1</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, consistent with experimental data. Three different simulation times are shown in each panel in differently colored solid lines. Long simulation times are required to approach steady state when <inline-formula><mml:math id="inf103"><mml:mi>λ</mml:mi></mml:math></inline-formula> becomes small, and the steady state is not quite reached in the right panel even after 720 hr. Dashed line indicates <inline-formula><mml:math id="inf104"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and is a guide to the eye. Other parameters: <inline-formula><mml:math id="inf105"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, <inline-formula><mml:math id="inf106"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf107"><mml:mrow><mml:mi>β</mml:mi><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, and the number of simulations = 100.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig4-figsupp2-v4.tif"/></fig><fig id="fig4s3" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 3.</label><caption><title>Distributions for a model with only aggregation and <inline-formula><mml:math id="inf108"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>.</title><p>Plateaus arise at the gelation transition of purely aggregating systems. Reverse cumulative distributions computed from stochastic simulations are shown. Different curves represent different simulation times, ranging linearly from 0.1 to 0.175 hr (magenta to cyan). Dashed line represents an approximate analytic prediction of <inline-formula><mml:math id="inf109"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>3</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> in the sol phase at the transition point and is a guide to the eye. Parameters: <inline-formula><mml:math id="inf110"><mml:mrow><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:mrow></mml:math></inline-formula> hr<sup>-1</sup>, number of cells = 10<sup>3</sup>, number of simulations = 100.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-71105-fig4-figsupp3-v4.tif"/></fig></fig-group><table-wrap id="table3" position="float"><label>Table 3.</label><caption><title>Summary of model variables and parameters.</title></caption><table frame="hsides" rules="groups"><thead><tr><th>Variable/parameter</th><th>Description</th></tr></thead><tbody><tr><td><inline-formula><mml:math id="inf111"><mml:mi>n</mml:mi></mml:math></inline-formula></td><td>Cluster size (number of cells)</td></tr><tr><td><inline-formula><mml:math id="inf112"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula></td><td>Probability of cluster size, <inline-formula><mml:math id="inf113"><mml:mi>n</mml:mi></mml:math></inline-formula></td></tr><tr><td><inline-formula><mml:math id="inf114"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula></td><td>Cumulative probability; probability of size being larger than <inline-formula><mml:math id="inf115"><mml:mi>n</mml:mi></mml:math></inline-formula></td></tr><tr><td>μ</td><td>Exponent of power law; <inline-formula><mml:math id="inf116"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf117"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula></td></tr><tr><td><inline-formula><mml:math id="inf118"><mml:mi>r</mml:mi></mml:math></inline-formula></td><td>Cell division rate</td></tr><tr><td><inline-formula><mml:math id="inf119"><mml:mi>K</mml:mi></mml:math></inline-formula></td><td>Carrying capacity; maximum number of cells</td></tr><tr><td><inline-formula><mml:math id="inf120"><mml:mi>β</mml:mi></mml:math></inline-formula></td><td>Fragmentation rate</td></tr><tr><td><inline-formula><mml:math id="inf121"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula></td><td>Fragmentation exponent; clusters of size <inline-formula><mml:math id="inf122"><mml:mi>n</mml:mi></mml:math></inline-formula> fragment with rate <inline-formula><mml:math id="inf123"><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup></mml:mrow></mml:math></inline-formula></td></tr><tr><td><inline-formula><mml:math id="inf124"><mml:mi>α</mml:mi></mml:math></inline-formula></td><td>Aggregation rate</td></tr><tr><td><inline-formula><mml:math id="inf125"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula></td><td>Aggregation exponent; clusters of sizes <inline-formula><mml:math id="inf126"><mml:mi>n</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf127"><mml:mi>m</mml:mi></mml:math></inline-formula> aggregate with rate <inline-formula><mml:math id="inf128"><mml:mrow><mml:mi>α</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:msup></mml:mrow></mml:math></inline-formula></td></tr><tr><td><inline-formula><mml:math id="inf129"><mml:mi>λ</mml:mi></mml:math></inline-formula></td><td>Expulsion rate</td></tr><tr><td><inline-formula><mml:math id="inf130"><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:math></inline-formula></td><td>Expulsion exponent; clusters of size <inline-formula><mml:math id="inf131"><mml:mi>n</mml:mi></mml:math></inline-formula> are expelled with rate <inline-formula><mml:math id="inf132"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:msup></mml:mrow></mml:math></inline-formula></td></tr></tbody></table></table-wrap><p>We further found that this plateau effect is intrinsic to finite systems (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). For the most aggregated cases in <xref ref-type="fig" rid="fig4">Figure 4B</xref>, we numerically solved the corresponding master equation, representing the deterministic dynamics of an infinite system, and found that the plateau did not occur. Master equation and stochastic simulation solutions agree for <inline-formula><mml:math id="inf133"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>, but for larger values of <inline-formula><mml:math id="inf134"><mml:msub><mml:mi>ν</mml:mi><mml:mi>A</mml:mi></mml:msub></mml:math></inline-formula>, the two solutions only agree in the small size regime. At large sizes, stochastic simulations produce an overabundance of large clusters compared to the master equation solution. This result indicates that in a finite system, strong aggregation can deplete small clusters, condensing them into a small number of large clusters on the order of the system-size.</p><p>This overall process is reminiscent of the gelation transition in soft materials. Stochastic dynamics of finite systems of purely aggregating particles at the gelation transition also produces distributions with plateaus, but with an initial decay given approximately by a power law with <inline-formula><mml:math id="inf135"><mml:mrow><mml:mi>μ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>5</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>, see also <xref ref-type="bibr" rid="bib21">Matsoukas, 2015</xref>). Combined with a growth/fragmentation/expulsion process, we found that size-dependent aggregation produces a distribution that initially decays in a power-law-like manner with tunable exponents and then exhibits a tuneable plateau, as we observe in the experimental data.</p></sec></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>We analyzed image-derived measurements of bacterial cluster sizes from larval zebrafish intestines and discovered a common family of size distributions shared across bacterial species. These distributions are extremely broad, exhibiting a power-law-like decay at small sizes that becomes shallower at large sizes in a strain-specific manner. We then demonstrated how these distributions emerge naturally from realistic kinetics: growth and single-cell fragmentation together generate power-law distributions, analogous to the distribution of neutral alleles in expanding populations, while size-dependent aggregation leads to a plateau representing the depletion of mid-sized clusters in favor for a single large one. In summary, we found that gut bacterial clusters are well-described by a model that combines the features of evolutionary dynamics in growing populations with those of inanimate systems of aggregating particles; intestinal bacteria form a ‘living gel’.</p><p>Gels are characterized by the emergence of a massive connected cluster that is on the order of the system size. In the larval zebrafish intestine, we often find for some bacterial species that the majority of the cells in the gut are contained within a single cluster, similar to a gel-like state. While growth by cell division generates large clusters, it is the aggregation process that leads to system-sized clusters being over-represented. This enhancement of massive clusters manifests as a plateau in the size distribution and is reminiscent of a true gelation phase transition. In our model, the prominence of this plateau appears to follow the same trend as in non-living, purely aggregating systems: the plateau depends strongly on the aggregation exponent that dictates the size-dependence of aggregation, with exponents larger than <inline-formula><mml:math id="inf136"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:math></inline-formula> leading to strong plateaus and exponents less than <inline-formula><mml:math id="inf137"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:math></inline-formula> leading to weak or no plateaus.</p><p>How this strong size-dependence in aggregation emerges in the intestine is unclear, although we hypothesize that active mixing by intestinal contractions, which can in fact merge multiple clusters at once (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>), is an important driver. We envision that the exponents for aggregation and also for fragmentation are likely generic, set by physical aspects of the intestine and the geometry of clusters, while the rates of these processes are bacterial-species dependent. In our system, we predict that differences in aggregation and/or fragmentation rates between strains underly the differences in measured size distributions. Further, it is possible that individual bacteria can tune these rates by altering their behavior, for example, modulating swimming motility (<xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>), in response to environmental cues. Quantitatively understanding how the combination of intestinal fluid mechanics and bacterial behaviors determine aggregation and fragmentation rates would be a fruitful avenue of future research. More abstractly, active growth combined with different aggregation processes, for example the fractal structures of diffusion-limited aggregation, may lead to different families of size distributions that would be interesting to explore.</p><p>On the experimental front, direct measurements of aggregation and fragmentation rates from time-lapse imaging would be an extremely useful next step. However, these measurements are technically challenging. Even by eye, unambiguously identifying that a single bacterium fragmented out of a larger aggregate, and did not simply float into the field of view, requires faster imaging speeds than we can currently obtain. Sparse, two-color labeling may improve reliability of detection, but would decrease the frequency of observing an event. Automatic identification of fragmentation events in time-lapse movies is a daunting task, but recent computational advances, for example using convolutional neural networks to automatically identify cell division events in mouse embryos (<xref ref-type="bibr" rid="bib22">McDole et al., 2018</xref>), may provide a good starting point. Aggregation is easier to observe by eye, but its automatic identification presents similar challenges in analysis.</p><p>Given the general and minimal nature of the model's assumptions, we predict that the form of the cluster size distributions we described here is common to the intestines of animals, including humans. This prediction of generality could be tested in a variety of systems using existing methods. In fruit flies, live imaging protocols have been developed that have revealed the presence of three-dimensional gut bacterial clusters highly reminiscent of what we observe in zebrafish, particularly in the midgut (<xref ref-type="bibr" rid="bib15">Koyama et al., 2020</xref>). Quantifying the sizes of these clusters would allow further tests of our model.</p><p>In mice, substantial progress has been made in imaging histological slices of the intestine with the luminal contents preserved (<xref ref-type="bibr" rid="bib37">Tropini et al., 2017</xref>). Intestinal contents are very dense in the distal mouse colon, however, and it is not clear how one should define cluster size. Other intestinal regions are likely more amenable to cluster analysis. Moreover, with species-specific labeling, it is possible to measure the distribution of clonal regions in these dense areas (<xref ref-type="bibr" rid="bib20">Mark Welch et al., 2017</xref>). One could imagine then comparing these data to a spatially-explicit, multispecies extension of the model we studied here.</p><p>Our model could also be tested indirectly for humans and other animals incompatible with direct imaging by way of fecal samples. Two decades ago, bacterial clusters spanning three orders of magnitude in volume were observed in gently dissociated fecal samples stained for mucus, but precise quantification of size statistics was not reported (<xref ref-type="bibr" rid="bib39">van der Waaij et al., 1996</xref>). Repeating these measurements with quantification, from for example imaging or flow cytometry, would also provide a test of our model, albeit on the microbiome as a whole rather than a single species at a time. The interpretation therefore would be of an effective species with kinetic rates representing average rates of different species.</p><p>To close, we emphasize that the degree of bacterial clustering in the gut is an important parameter for both microbial population dynamics and host-bacteria interactions. More aggregation leads to larger fluctuations in abundance due to the expulsion of big clusters, and also thereby increase the likelihood of extinction (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>; <xref ref-type="bibr" rid="bib32">Schlomann and moments, 2018</xref>). Further, aggregation within the intestinal lumen can reduce access to the epithelium and reduce pro-inflammatory signaling (<xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>). Therefore, measurements of cluster sizes may be an important biomarker for microbiota-related health issues, and inference of dynamics from size statistics using models like this one could aid the development of therapeutics.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th>Reagent type (species) <break/>or resource</th><th>Designation</th><th>Source or reference</th><th>Identifiers</th><th>Additional <break/>information</th></tr></thead><tbody><tr><td>Software, algorithm</td><td>Analysis code</td><td>This study</td><td/><td>see Materials and methods, Simulations</td></tr><tr><td>Other</td><td>Cluster size data</td><td><xref ref-type="bibr" rid="bib32">Schlomann and moments, 2018</xref></td><td/><td/></tr><tr><td>Other</td><td>Cluster size data</td><td><xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref></td><td/><td/></tr><tr><td>Other</td><td>Cluster size data</td><td><xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref></td><td/><td/></tr></tbody></table></table-wrap><sec id="s4-1"><title>Data</title><p>We assembled data on gut bacterial cluster sizes from three different studies on larval zebrafish (<xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>; <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref>). Size data from <xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref> and <xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref> were taken directly from the supplementary data files associated with those publications. The raw size data from <xref ref-type="bibr" rid="bib43">Wiles et al., 2020</xref> was not included in its associated supplementary data file, but summary statistics such as planktonic fraction were. All sizes were rounded up to the nearest integer.</p><p>Details of experimental procedures can be found in the original papers. In brief, as described in <xref ref-type="fig" rid="fig1">Figure 1</xref>, animals were reared germ-free, mono-associated with a single bacterial strain, each carrying a chromosomal GFP tag, and then imaged 24 hr later using a custom-built light sheet fluorescence microscope (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>). The gut is imaged in four tiled sub-regions that are registered via cross-correlation and manual adjustment. Imaging a full gut volume (≈1200 μm × 300 μm × 150 μm) with 1 μm slices takes approximately 45 s. Laser power (5 mW) and exposure time (30 ms) were identical for all experiments.</p><p>The image analysis pipeline used to enumerate bacterial cluster sizes is also described in detail in the original publications and in reference (<xref ref-type="bibr" rid="bib13">Jemielita et al., 2014</xref>). In brief, single cells (small objects) and multicellular aggregates (large objects) are identified separately. The number of cells per aggregate is then estimated as the total fluorescence intensity of the aggregate divided by the mean fluorescence intensity of a single cell. Small objects are identified in three dimensions with a combination of difference-of-gaussians and wavelet filters (<xref ref-type="bibr" rid="bib27">Olivo-Marin, 2002</xref>) and then culled using a support vector machine classifier and manual curation. Large objects are segmented in maximum intensity projections using a graph-cut algorithm (<xref ref-type="bibr" rid="bib4">Boykov and Kolmogorov, 2004</xref>) seeded by either an intensity- or gradient-thresholded mask. The total intensity of an aggregate is computed by extending the two-dimensional mask in the <inline-formula><mml:math id="inf138"><mml:mi>z</mml:mi></mml:math></inline-formula>-direction and summing fluorescence intensities above a threshold calculated from the boundary of the mask, with pixels detected as part of single cells removed. The boundary of the gut is manually outlined prior to image analysis and used to exclude extra-intestinal fluorescence.</p></sec><sec id="s4-2"><title>Size distribution</title><p>For the experimental data, reverse cumulative distributions were computed as<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">z</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo>&gt;</mml:mo><mml:mi mathvariant="normal">n</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">z</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mo>&gt;</mml:mo><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>In combining data from different samples colonized with the same strain, we pooled together all sizes and computed the distribution in the same way. For simulations with large numbers clusters, we computed this distribution iteratively, looping through each simulation replicate and independently updating (number clusters with size <inline-formula><mml:math id="inf139"><mml:mrow><mml:mi/><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:math></inline-formula>) and (total number of clusters), and normalizing at the end.</p><p>For the binned probability densities in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>, data were similarly pooled across samples and then sorted into logarithmically spaced bins of <italic>log</italic><sub>10</sub> width = 0.4.</p></sec><sec id="s4-3"><title>Estimates on bounds of agg rates</title><p>We estimated approximate bounds on the rate of total aggregation events as follows. For the maximum rate, we note that a typical population contains approximately 200 clusters (mean ± std. dev of 244 ± 182). In the absence of other processes, condensing this system into one cluster would require 100 aggregation events. Populations consisting of almost entirely one large cluster are rare but have been documented (<xref ref-type="bibr" rid="bib29">Schlomann et al., 2018</xref>). Therefore, we estimate that this complete condensation can occur no more than once an hour, leading to an upper bound on the total rate of aggregation events of 100 per hour.</p><p>For the minimum rate, we start with the observation that aggregation has been directly observed between small clusters and also between small clusters and a single large cluster during a large expulsion event (<xref ref-type="bibr" rid="bib30">Schlomann et al., 2019</xref>). Considering just the latter process, we know that large expulsion events happen roughly once every 10 hr. If approximately 10 small clusters are grouped into the large cluster during transit out of the gut, that would correspond 10 total aggregation events in 10 hr, or, 1 per hour, which we take as a lower bound.</p></sec><sec id="s4-4"><title>Simulations</title><p>We used three different numerical approaches for studying the models discussed here. The minimal growth-fragmentation process in <xref ref-type="fig" rid="fig3">Figure 3</xref> was simulated with a Poisson tau-leaping algorithm <xref ref-type="bibr" rid="bib9">Gillespie, 2001</xref> with a simple fixed tau value of <inline-formula><mml:math id="inf140"><mml:mrow><mml:mi>τ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.1</mml:mn></mml:mrow></mml:math></inline-formula> hr. At each time step, the number of growth and fragmentation events was drawn from a Poisson distribution with the rates given in <xref ref-type="fig" rid="fig3">Figure 3B</xref> along with the constraint that clusters must be of size two or larger to fragment.</p><p>For the full model including aggregation and expulsion, we used Gillespie's algorithm <xref ref-type="bibr" rid="bib8">Gillespie, 1977</xref> for fragmentation, aggregation, and expulsion events, while growth was updated deterministically according to a continuous logistic growth law approximated by an Euler step with <inline-formula><mml:math id="inf141"><mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext>min</mml:mtext><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>τ</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mn>0.1</mml:mn><mml:mo>⁢</mml:mo> <mml:mtext>hr</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf142"><mml:mi>τ</mml:mi></mml:math></inline-formula> here refers to the time to next reaction. For the Gillespie steps, if the time to next reaction exceeded the doubling time, <inline-formula><mml:math id="inf143"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula>, the growth steps were performed and then the propensity functions were re-calculated.</p><p>Finally, we compared these stochastic simulations to a model in the thermodynamic limit where individual clusters are replaced with cluster densities that evolve deterministically, which is referred to as a master equation (<xref ref-type="bibr" rid="bib16">Krapivsky et al., 2010</xref>). The master equation for the general model reads<disp-formula id="equ3"><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mtable columnalign="left left" columnspacing="1em" rowspacing="4pt"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mover><mml:mi>c</mml:mi><mml:mo>˙</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo></mml:mstyle></mml:mtd><mml:mtd><mml:mtext> </mml:mtext><mml:mfrac><mml:mi>α</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mo stretchy="false">[</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mo stretchy="false">]</mml:mo><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi>α</mml:mi><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mi>m</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>+</mml:mo><mml:mi>r</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>K</mml:mi></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi>n</mml:mi><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>λ</mml:mi><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mo>+</mml:mo><mml:mi>β</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>K</mml:mi></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>δ</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mi>m</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>This set of equations was solved numerically on a bounded size grid using an Euler method with step size <inline-formula><mml:math id="inf144"><mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.0001</mml:mn></mml:mrow></mml:math></inline-formula> hr. Models that include a carrying capacity, <inline-formula><mml:math id="inf145"><mml:mi>K</mml:mi></mml:math></inline-formula>, are already defined on a finite domain of integers ranging from one to <inline-formula><mml:math id="inf146"><mml:mi>K</mml:mi></mml:math></inline-formula> and the master equation is naturally represented by a set of <inline-formula><mml:math id="inf147"><mml:mi>K</mml:mi></mml:math></inline-formula> ordinary differential equations. For models without a carrying capacity, we introduced a maximum size given by the average population size at the last time point, <inline-formula><mml:math id="inf148"><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mtext>max</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>exp</mml:mtext><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mtext>max</mml:mtext></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> (rounded up to the nearest integer), and used reflecting boundary conditions at <inline-formula><mml:math id="inf149"><mml:msub><mml:mi>n</mml:mi><mml:mtext>max</mml:mtext></mml:msub></mml:math></inline-formula>.</p><p>A distribution was deemed stationary if it was visibly unchanged after an additional 50% of simulation time.</p><p>MATLAB code for simulating these models and plotting data can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/rplab/cluster_kinetics">https://github.com/rplab/cluster_kinetics</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:391e41adbd1fc14fa4034db6776cb14cd728342c;origin=https://github.com/rplab/cluster_kinetics;visit=swh:1:snp:fa394090d0ed426ab9a4af4ff5e6d510cac176ac;anchor=swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae">swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae</ext-link>, <xref ref-type="bibr" rid="bib31">Schlomann, 2021</xref>).</p></sec><sec id="s4-5"><title>Estimating distribution exponents</title><p>For the simulated distributions in <xref ref-type="fig" rid="fig3">Figure 3</xref> we estimated a power law exponent using the maximum likelihood-based method described in <xref ref-type="bibr" rid="bib5">Clauset et al., 2009</xref> and the plfit.m code supplied therein. This model includes a minimum size as a free parameter that dictates when the power-law tail begins. The minimum size is chosen to minimize the Kolmogorov-Smirnov distance between the data and model distributions for sizes greater than the minimum size. Best fit values of the exponent and minimum size are included in <xref ref-type="supplementary-material" rid="fig3sdata1">Figure 3—source data 1</xref>.</p><p>For the experimentally measured distributions, we used both maximum likelihood estimation and linear fitting to the log-transformed cumulative distribution to calculate exponents.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We thank Jayson Paulose for helpful discussions. Research was supported by the National Institutes of Health (<ext-link ext-link-type="uri" xlink:href="http://www.nih.gov/">http://www.nih.gov/</ext-link>), under Awards P50GM09891, P01GM125576, F32AI112094, and T32GM007759. Work was also supported by the National Science Foundation under Award 1427957, and an award from the Kavli Microbiome Ideas Challenge, a project led by the American Society for Microbiology in partnership with the American Chemical Society and the American Physical Society and supported by The Kavli Foundation. The University of Oregon Zebrafish Facility is supported by a grant from the National Institute of Child Health and Human Development (P01HD22486). BHS was supported by the James S McDonnell Foundation postdoctoral fellowship. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Resources, Software, Formal analysis, Supervision, Funding acquisition, Investigation, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Resources, Software, Formal analysis, Supervision, Funding acquisition, Investigation, Methodology, Writing - original draft, Writing - review and editing</p></fn></fn-group><fn-group content-type="ethics-information"><title>Ethics</title><fn fn-type="other"><p>Animal experimentation: The studies that generated the data analyzed in this paper (see cited references) were done in strict accordance with protocols approved by the University of Oregon Institutional Animal Care and Use Committee and following standard protocols.</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Cumulative distribution exponents of cluster size distributions by strain, with analysis of sensitivity to single-cell detection.</title><p>The small size regime of the cluster size distributions were fit to a power-law model for sizes up to 100 cells using two methods: a linear fit to <inline-formula><mml:math id="inf150"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and maximum likelihood estimation (Materials and methods). The fits were done for each animal and the resulting mean ± std. dev of the exponents (corresponding to <inline-formula><mml:math id="inf151"><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, as defined in the text) are given for each strain. For each method, the fits were done twice, once including single cells, and once considering only cells of size two or greater. As discussed in the main text, the largest uncertainty in cluster size enumeration from the images occurs at small sizes. Ignoring single cells in the fit only mildly changes the average exponent, and all changes are within uncertainties. Most exponents are consistent with <inline-formula><mml:math id="inf152"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-71105-supp1-v4.xlsx"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-71105-transrepform-v4.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>A table of all bacterial cluster sizes analysed in this study is included in the Source Data Files. MATLAB code for simulating the models described in the study is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/rplab/cluster_kinetics">https://github.com/rplab/cluster_kinetics</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae">https://archive.softwareheritage.org/swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae</ext-link>).</p><p>The following datasets were generated:</p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Altan-Bonnet</surname> <given-names>G</given-names></name><name><surname>Mora</surname> <given-names>T</given-names></name><name><surname>Walczak</surname> <given-names>AM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Quantitative immunology for physicists</article-title><source>Physics Reports</source><volume>849</volume><fpage>1</fpage><lpage>83</lpage><pub-id pub-id-type="doi">10.1016/j.physrep.2020.01.001</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bansept</surname> <given-names>F</given-names></name><name><surname>Schumann-Moor</surname> <given-names>K</given-names></name><name><surname>Diard</surname> <given-names>M</given-names></name><name><surname>Hardt</surname> <given-names>WD</given-names></name><name><surname>Slack</surname> <given-names>E</given-names></name><name><surname>Loverdo</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Enchained growth and cluster dislocation: a possible mechanism for Microbiota homeostasis</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1006986</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006986</pub-id><pub-id pub-id-type="pmid">31050663</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barabasi</surname> <given-names>AL</given-names></name><name><surname>Albert</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Emergence of scaling in random networks</article-title><source>Science</source><volume>286</volume><fpage>509</fpage><lpage>512</lpage><pub-id pub-id-type="doi">10.1126/science.286.5439.509</pub-id><pub-id pub-id-type="pmid">10521342</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boykov</surname> <given-names>Y</given-names></name><name><surname>Kolmogorov</surname> <given-names>V</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>An experimental comparison of min-cut/max-flow algorithms for energy minimization in vision</article-title><source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source><volume>26</volume><fpage>1124</fpage><lpage>1137</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2004.60</pub-id><pub-id pub-id-type="pmid">15742889</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Clauset</surname> <given-names>A</given-names></name><name><surname>Shalizi</surname> <given-names>CR</given-names></name><name><surname>Newman</surname> <given-names>MEJ</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Power-Law distributions in empirical data</article-title><source>SIAM Review</source><volume>51</volume><fpage>661</fpage><lpage>703</lpage><pub-id pub-id-type="doi">10.1137/070710111</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Datta</surname> <given-names>SS</given-names></name><name><surname>Preska Steinberg</surname> <given-names>A</given-names></name><name><surname>Ismagilov</surname> <given-names>RF</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Polymers in the gut compress the colonic mucus hydrogel</article-title><source>PNAS</source><volume>113</volume><fpage>7041</fpage><lpage>7046</lpage><pub-id pub-id-type="doi">10.1073/pnas.1602789113</pub-id><pub-id pub-id-type="pmid">27303035</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Donaldson</surname> <given-names>GP</given-names></name><name><surname>Lee</surname> <given-names>SM</given-names></name><name><surname>Mazmanian</surname> <given-names>SK</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Gut biogeography of the bacterial Microbiota</article-title><source>Nature Reviews Microbiology</source><volume>14</volume><fpage>20</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1038/nrmicro3552</pub-id><pub-id pub-id-type="pmid">26499895</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gillespie</surname> <given-names>DT</given-names></name></person-group><year iso-8601-date="1977">1977</year><article-title>Exact stochastic simulation of coupled chemical reactions</article-title><source>The Journal of Physical Chemistry</source><volume>81</volume><fpage>2340</fpage><lpage>2361</lpage><pub-id pub-id-type="doi">10.1021/j100540a008</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gillespie</surname> <given-names>DT</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Approximate accelerated stochastic simulation of chemically reacting systems</article-title><source>The Journal of Chemical Physics</source><volume>115</volume><fpage>1716</fpage><lpage>1733</lpage><pub-id pub-id-type="doi">10.1063/1.1378322</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greenfield</surname> <given-names>D</given-names></name><name><surname>McEvoy</surname> <given-names>AL</given-names></name><name><surname>Shroff</surname> <given-names>H</given-names></name><name><surname>Crooks</surname> <given-names>GE</given-names></name><name><surname>Wingreen</surname> <given-names>NS</given-names></name><name><surname>Betzig</surname> <given-names>E</given-names></name><name><surname>Liphardt</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Self-organization of the <italic>Escherichia coli</italic> chemotaxis network imaged with super-resolution light microscopy</article-title><source>PLOS Biology</source><volume>7</volume><elocation-id>e1000137</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1000137</pub-id><pub-id pub-id-type="pmid">19547746</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hansen</surname> <given-names>SM</given-names></name><name><surname>McKay</surname> <given-names>TA</given-names></name><name><surname>Wechsler</surname> <given-names>RH</given-names></name><name><surname>Annis</surname> <given-names>J</given-names></name><name><surname>Sheldon</surname> <given-names>ES</given-names></name><name><surname>Kimball</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Measurement of Galaxy Cluster Sizes, Radial Profiles, and Luminosity Functions from SDSS Photometric Data</article-title><source>The Astrophysical Journal</source><volume>633</volume><fpage>122</fpage><lpage>137</lpage><pub-id pub-id-type="doi">10.1086/444554</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hubbell</surname> <given-names>SP</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>A unified theory of biogeography and relative species abundance and its application to tropical rain forests and coral reefs</article-title><source>Coral Reefs</source><volume>16</volume><fpage>S9</fpage><lpage>S21</lpage><pub-id pub-id-type="doi">10.1007/s003380050237</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jemielita</surname> <given-names>M</given-names></name><name><surname>Taormina</surname> <given-names>MJ</given-names></name><name><surname>Burns</surname> <given-names>AR</given-names></name><name><surname>Hampton</surname> <given-names>JS</given-names></name><name><surname>Rolig</surname> <given-names>AS</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Spatial and temporal features of the growth of a bacterial species colonizing the zebrafish gut</article-title><source>mBio</source><volume>5</volume><elocation-id>14</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.01751-14</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Keller</surname> <given-names>PJ</given-names></name><name><surname>Schmidt</surname> <given-names>AD</given-names></name><name><surname>Wittbrodt</surname> <given-names>J</given-names></name><name><surname>Stelzer</surname> <given-names>EH</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Reconstruction of zebrafish early embryonic development by scanned light sheet microscopy</article-title><source>Science</source><volume>322</volume><fpage>1065</fpage><lpage>1069</lpage><pub-id pub-id-type="doi">10.1126/science.1162493</pub-id><pub-id pub-id-type="pmid">18845710</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Koyama</surname> <given-names>LAJ</given-names></name><name><surname>Aranda-Díaz</surname> <given-names>A</given-names></name><name><surname>Su</surname> <given-names>YH</given-names></name><name><surname>Balachandra</surname> <given-names>S</given-names></name><name><surname>Martin</surname> <given-names>JL</given-names></name><name><surname>Ludington</surname> <given-names>WB</given-names></name><name><surname>Huang</surname> <given-names>KC</given-names></name><name><surname>O'Brien</surname> <given-names>LE</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Bellymount enables longitudinal, intravital imaging of abdominal organs and the gut Microbiota in adult <italic>Drosophila</italic></article-title><source>PLOS Biology</source><volume>18</volume><elocation-id>e3000567</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.3000567</pub-id><pub-id pub-id-type="pmid">31986129</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Krapivsky</surname> <given-names>PL</given-names></name><name><surname>Redner</surname> <given-names>S</given-names></name><name><surname>Ben-Naim</surname> <given-names>EA</given-names></name></person-group><year iso-8601-date="2010">2010</year><source>Kinetic View of Statistical Physics</source><publisher-name>Cambridge University Press</publisher-name></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krapivsky</surname> <given-names>PL</given-names></name><name><surname>Redner</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>Transitional aggregation kinetics in dry and damp environments</article-title><source>Physical Review E</source><volume>54</volume><fpage>3553</fpage><lpage>3561</lpage><pub-id pub-id-type="doi">10.1103/PhysRevE.54.3553</pub-id><pub-id pub-id-type="pmid">9965501</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lifshitz</surname> <given-names>IM</given-names></name><name><surname>Slyozov</surname> <given-names>VV</given-names></name></person-group><year iso-8601-date="1961">1961</year><article-title>The kinetics of precipitation from supersaturated solid solutions</article-title><source>Journal of Physics and Chemistry of Solids</source><volume>19</volume><fpage>35</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1016/0022-3697(61)90054-3</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lloyd-Price</surname> <given-names>J</given-names></name><name><surname>Mahurkar</surname> <given-names>A</given-names></name><name><surname>Rahnavard</surname> <given-names>G</given-names></name><name><surname>Crabtree</surname> <given-names>J</given-names></name><name><surname>Orvis</surname> <given-names>J</given-names></name><name><surname>Hall</surname> <given-names>AB</given-names></name><name><surname>Brady</surname> <given-names>A</given-names></name><name><surname>Creasy</surname> <given-names>HH</given-names></name><name><surname>McCracken</surname> <given-names>C</given-names></name><name><surname>Giglio</surname> <given-names>MG</given-names></name><name><surname>McDonald</surname> <given-names>D</given-names></name><name><surname>Franzosa</surname> <given-names>EA</given-names></name><name><surname>Knight</surname> <given-names>R</given-names></name><name><surname>White</surname> <given-names>O</given-names></name><name><surname>Huttenhower</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Strains, functions and dynamics in the expanded human microbiome project</article-title><source>Nature</source><volume>550</volume><fpage>61</fpage><lpage>66</lpage><pub-id pub-id-type="doi">10.1038/nature23889</pub-id><pub-id pub-id-type="pmid">28953883</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mark Welch</surname> <given-names>JL</given-names></name><name><surname>Hasegawa</surname> <given-names>Y</given-names></name><name><surname>McNulty</surname> <given-names>NP</given-names></name><name><surname>Gordon</surname> <given-names>JI</given-names></name><name><surname>Borisy</surname> <given-names>GG</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Spatial organization of a model 15-member human gut Microbiota established in gnotobiotic mice</article-title><source>PNAS</source><volume>114</volume><fpage>E9105</fpage><lpage>E9114</lpage><pub-id pub-id-type="doi">10.1073/pnas.1711596114</pub-id><pub-id pub-id-type="pmid">29073107</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matsoukas</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Statistical thermodynamics of irreversible aggregation: the Sol-Gel transition</article-title><source>Scientific Reports</source><volume>5</volume><elocation-id>srep08855</elocation-id><pub-id pub-id-type="doi">10.1038/srep08855</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McDole</surname> <given-names>K</given-names></name><name><surname>Guignard</surname> <given-names>L</given-names></name><name><surname>Amat</surname> <given-names>F</given-names></name><name><surname>Berger</surname> <given-names>A</given-names></name><name><surname>Malandain</surname> <given-names>G</given-names></name><name><surname>Royer</surname> <given-names>LA</given-names></name><name><surname>Turaga</surname> <given-names>SC</given-names></name><name><surname>Branson</surname> <given-names>K</given-names></name><name><surname>Keller</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>In Toto Imaging and Reconstruction of Post-Implantation Mouse Development at the Single-Cell Level</article-title><source>Cell</source><volume>175</volume><fpage>859</fpage><lpage>876</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.09.031</pub-id><pub-id pub-id-type="pmid">30318151</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McNally</surname> <given-names>L</given-names></name><name><surname>Bernardy</surname> <given-names>E</given-names></name><name><surname>Thomas</surname> <given-names>J</given-names></name><name><surname>Kalziqi</surname> <given-names>A</given-names></name><name><surname>Pentz</surname> <given-names>J</given-names></name><name><surname>Brown</surname> <given-names>SP</given-names></name><name><surname>Hammer</surname> <given-names>BK</given-names></name><name><surname>Yunker</surname> <given-names>PJ</given-names></name><name><surname>Ratcliff</surname> <given-names>WC</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Killing by type VI secretion drives genetic phase separation and correlates with increased cooperation</article-title><source>Nature Communications</source><volume>8</volume><elocation-id>14371</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms14371</pub-id><pub-id pub-id-type="pmid">28165005</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moor</surname> <given-names>K</given-names></name><name><surname>Diard</surname> <given-names>M</given-names></name><name><surname>Sellin</surname> <given-names>ME</given-names></name><name><surname>Felmy</surname> <given-names>B</given-names></name><name><surname>Wotzka</surname> <given-names>SY</given-names></name><name><surname>Toska</surname> <given-names>A</given-names></name><name><surname>Bakkeren</surname> <given-names>E</given-names></name><name><surname>Arnoldini</surname> <given-names>M</given-names></name><name><surname>Bansept</surname> <given-names>F</given-names></name><name><surname>Co</surname> <given-names>AD</given-names></name><name><surname>Völler</surname> <given-names>T</given-names></name><name><surname>Minola</surname> <given-names>A</given-names></name><name><surname>Fernandez-Rodriguez</surname> <given-names>B</given-names></name><name><surname>Agatic</surname> <given-names>G</given-names></name><name><surname>Barbieri</surname> <given-names>S</given-names></name><name><surname>Piccoli</surname> <given-names>L</given-names></name><name><surname>Casiraghi</surname> <given-names>C</given-names></name><name><surname>Corti</surname> <given-names>D</given-names></name><name><surname>Lanzavecchia</surname> <given-names>A</given-names></name><name><surname>Regoes</surname> <given-names>RR</given-names></name><name><surname>Loverdo</surname> <given-names>C</given-names></name><name><surname>Stocker</surname> <given-names>R</given-names></name><name><surname>Brumley</surname> <given-names>DR</given-names></name><name><surname>Hardt</surname> <given-names>WD</given-names></name><name><surname>Slack</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>High-avidity IgA protects the intestine by enchaining growing Bacteria</article-title><source>Nature</source><volume>544</volume><fpage>498</fpage><lpage>502</lpage><pub-id pub-id-type="doi">10.1038/nature22058</pub-id><pub-id pub-id-type="pmid">28405025</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Neher</surname> <given-names>RA</given-names></name><name><surname>Hallatschek</surname> <given-names>O</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Genealogies of rapidly adapting populations</article-title><source>PNAS</source><volume>110</volume><fpage>437</fpage><lpage>442</lpage><pub-id pub-id-type="doi">10.1073/pnas.1213113110</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nourmohammad</surname> <given-names>A</given-names></name><name><surname>Otwinowski</surname> <given-names>J</given-names></name><name><surname>Łuksza</surname> <given-names>M</given-names></name><name><surname>Mora</surname> <given-names>T</given-names></name><name><surname>Walczak</surname> <given-names>AM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Fierce selection and interference in B-Cell repertoire response to chronic HIV-1</article-title><source>Molecular Biology and Evolution</source><volume>36</volume><fpage>2184</fpage><lpage>2194</lpage><pub-id pub-id-type="doi">10.1093/molbev/msz143</pub-id><pub-id pub-id-type="pmid">31209469</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Olivo-Marin</surname> <given-names>J-C</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Extraction of spots in biological images using multiscale products</article-title><source>Pattern Recognition</source><volume>35</volume><fpage>1989</fpage><lpage>1996</lpage><pub-id pub-id-type="doi">10.1016/S0031-3203(01)00127-3</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parthasarathy</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Monitoring microbial communities using light sheet fluorescence microscopy</article-title><source>Current Opinion in Microbiology</source><volume>43</volume><fpage>31</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1016/j.mib.2017.11.008</pub-id><pub-id pub-id-type="pmid">29175679</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>Wiles</surname> <given-names>TJ</given-names></name><name><surname>Wall</surname> <given-names>ES</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Bacterial cohesion predicts spatial distribution in the larval zebrafish intestine</article-title><source>Biophysical Journal</source><volume>115</volume><fpage>2271</fpage><lpage>2277</lpage><pub-id pub-id-type="doi">10.1016/j.bpj.2018.10.017</pub-id><pub-id pub-id-type="pmid">30448038</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>Wiles</surname> <given-names>TJ</given-names></name><name><surname>Wall</surname> <given-names>ES</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Sublethal antibiotics collapse gut bacterial populations by enhancing aggregation and expulsion</article-title><source>PNAS</source><volume>116</volume><fpage>21392</fpage><lpage>21400</lpage><pub-id pub-id-type="doi">10.1073/pnas.1907567116</pub-id><pub-id pub-id-type="pmid">31591228</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Schlomann</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>cluster_kinetics</data-title><source>Software Heritage</source><version designator="swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae">swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae</version><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:391e41adbd1fc14fa4034db6776cb14cd728342c;origin=https://github.com/rplab/cluster_kinetics;visit=swh:1:snp:fa394090d0ed426ab9a4af4ff5e6d510cac176ac;anchor=swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae">https://archive.softwareheritage.org/swh:1:dir:391e41adbd1fc14fa4034db6776cb14cd728342c;origin=https://github.com/rplab/cluster_kinetics;visit=swh:1:snp:fa394090d0ed426ab9a4af4ff5e6d510cac176ac;anchor=swh:1:rev:f55a54a9c88e4fb8376dfc91e25ac4383c4240ae</ext-link></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>moments</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Stationary moments, diffusion limits, and extinction times for logistic growth with random catastrophes</article-title><source>Journal of Theoretical Biology</source><volume>454</volume><fpage>154</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jtbi.2018.06.007</pub-id><pub-id pub-id-type="pmid">29885410</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sender</surname> <given-names>R</given-names></name><name><surname>Fuchs</surname> <given-names>S</given-names></name><name><surname>Milo</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Revised Estimates for the Number of Human and Bacteria Cells in the Body</article-title><source>PLOS Biology</source><volume>14</volume><elocation-id>e1002533</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1002533</pub-id><pub-id pub-id-type="pmid">27541692</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simon</surname> <given-names>HA</given-names></name></person-group><year iso-8601-date="1955">1955</year><article-title>On a class of skew distribution functions</article-title><source>Biometrika</source><volume>4</volume><fpage>425</fpage><lpage>440</lpage></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stephens</surname> <given-names>WZ</given-names></name><name><surname>Burns</surname> <given-names>AR</given-names></name><name><surname>Stagaman</surname> <given-names>K</given-names></name><name><surname>Wong</surname> <given-names>S</given-names></name><name><surname>Rawls</surname> <given-names>JF</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Bohannan</surname> <given-names>BJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The composition of the zebrafish intestinal microbial community varies across development</article-title><source>The ISME Journal</source><volume>10</volume><fpage>644</fpage><lpage>654</lpage><pub-id pub-id-type="doi">10.1038/ismej.2015.140</pub-id><pub-id pub-id-type="pmid">26339860</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Tilman</surname> <given-names>D</given-names></name><name><surname>Kareiva</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2018">2018</year><source>Spatial Ecology: The Role of Space in Population Dynamics and Interspecific Interactions (MPB-30</source><publisher-name>Princeton University Press</publisher-name></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tropini</surname> <given-names>C</given-names></name><name><surname>Earle</surname> <given-names>KA</given-names></name><name><surname>Huang</surname> <given-names>KC</given-names></name><name><surname>Sonnenburg</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The gut microbiome: connecting spatial organization to function</article-title><source>Cell Host &amp; Microbe</source><volume>21</volume><fpage>433</fpage><lpage>442</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2017.03.010</pub-id><pub-id pub-id-type="pmid">28407481</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vaishnava</surname> <given-names>S</given-names></name><name><surname>Yamamoto</surname> <given-names>M</given-names></name><name><surname>Severson</surname> <given-names>KM</given-names></name><name><surname>Ruhn</surname> <given-names>KA</given-names></name><name><surname>Yu</surname> <given-names>X</given-names></name><name><surname>Koren</surname> <given-names>O</given-names></name><name><surname>Ley</surname> <given-names>R</given-names></name><name><surname>Wakeland</surname> <given-names>EK</given-names></name><name><surname>Hooper</surname> <given-names>LV</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The antibacterial lectin RegIIIgamma promotes the spatial segregation of Microbiota and host in the intestine</article-title><source>Science</source><volume>334</volume><fpage>255</fpage><lpage>258</lpage><pub-id pub-id-type="doi">10.1126/science.1209791</pub-id><pub-id pub-id-type="pmid">21998396</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van der Waaij</surname> <given-names>LA</given-names></name><name><surname>Limburg</surname> <given-names>PC</given-names></name><name><surname>Mesander</surname> <given-names>G</given-names></name><name><surname>van der Waaij</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>In vivo IgA coating of anaerobic Bacteria in human faeces</article-title><source>Gut</source><volume>38</volume><fpage>348</fpage><lpage>354</lpage><pub-id pub-id-type="doi">10.1136/gut.38.3.348</pub-id><pub-id pub-id-type="pmid">8675085</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weiner</surname> <given-names>BG</given-names></name><name><surname>Posfai</surname> <given-names>A</given-names></name><name><surname>Wingreen</surname> <given-names>NS</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Spatial ecology of territorial populations</article-title><source>PNAS</source><volume>116</volume><fpage>17874</fpage><lpage>17879</lpage><pub-id pub-id-type="doi">10.1073/pnas.1911570116</pub-id><pub-id pub-id-type="pmid">31434790</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wiles</surname> <given-names>TJ</given-names></name><name><surname>Jemielita</surname> <given-names>M</given-names></name><name><surname>Baker</surname> <given-names>RP</given-names></name><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>Logan</surname> <given-names>SL</given-names></name><name><surname>Ganz</surname> <given-names>J</given-names></name><name><surname>Melancon</surname> <given-names>E</given-names></name><name><surname>Eisen</surname> <given-names>JS</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Host gut motility promotes competitive exclusion within a model intestinal Microbiota</article-title><source>PLOS Biology</source><volume>14</volume><elocation-id>e1002517</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1002517</pub-id><pub-id pub-id-type="pmid">27458727</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wiles</surname> <given-names>TJ</given-names></name><name><surname>Wall</surname> <given-names>ES</given-names></name><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>Hay</surname> <given-names>EA</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name><name><surname>Graf</surname> <given-names>J</given-names></name><name><surname>Salama</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Modernized tools for streamlined genetic manipulation and comparative study of wild and diverse proteobacterial lineages</article-title><source>mBio</source><volume>9</volume><elocation-id>18</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.01877-18</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wiles</surname> <given-names>TJ</given-names></name><name><surname>Schlomann</surname> <given-names>BH</given-names></name><name><surname>Wall</surname> <given-names>ES</given-names></name><name><surname>Betancourt</surname> <given-names>R</given-names></name><name><surname>Parthasarathy</surname> <given-names>R</given-names></name><name><surname>Guillemin</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Swimming motility of a gut bacterial symbiont promotes resistance to intestinal expulsion and enhances inflammation</article-title><source>PLOS Biology</source><volume>18</volume><elocation-id>e3000661</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.3000661</pub-id><pub-id pub-id-type="pmid">32196484</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Woodard</surname> <given-names>D</given-names></name><name><surname>Bell</surname> <given-names>D</given-names></name><name><surname>Tipton</surname> <given-names>D</given-names></name><name><surname>Durrance</surname> <given-names>S</given-names></name><name><surname>Burnett</surname> <given-names>LC</given-names></name><name><surname>Cole</surname> <given-names>L</given-names></name><name><surname>Li</surname> <given-names>B</given-names></name><name><surname>Xu</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Gel formation in protein amyloid aggregation: a physical mechanism for cytotoxicity</article-title><source>PLOS ONE</source><volume>9</volume><elocation-id>e94789</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0094789</pub-id><pub-id pub-id-type="pmid">24740416</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yule</surname> <given-names>GU</given-names></name></person-group><year iso-8601-date="1925">1925</year><article-title>II.—A mathematical theory of evolution, based on the conclusions of Dr. JC Willis, FRS</article-title><source>Philosophical Transactions of the Royal Society of London. Series B, Containing Papers of a Biological Character</source><volume>213</volume><fpage>21</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1098/rstb.1925.0002</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec id="s8" sec-type="appendix"><title>Analytic calculations for growth-fragmentation processes</title><boxed-text><p>We consider a model with only growth and fragmentation processes and make heuristic arguments for the form of the asymptotic size distribution. In particular, we are interested in how the exponent of the resulting power law tails depends on the growth and fragmentation rates. We derive here the results listed in <xref ref-type="table" rid="table2">Table 2</xref> of the main text.</p><sec id="s9"><title>Model summary</title><p>Clusters grow according to<disp-formula id="equ4"><label>(3)</label><mml:math id="m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/></mml:mrow><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:mstyle></mml:math></disp-formula>and they fragment according to<disp-formula id="equ5"><label>(4)</label><mml:math id="m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">→</mml:mo><mml:mi>n</mml:mi><mml:mrow><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/></mml:mrow><mml:mi>β</mml:mi><mml:msup><mml:mi mathvariant="normal">n</mml:mi><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mrow><mml:mi mathvariant="normal">F</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mrow><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mspace width="thinmathspace"/><mml:mspace width="thinmathspace"/></mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>1.</mml:mn></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>The cell lost during fragmentation becomes its own cluster of size one.</p><p>We now consider the deterministic dynamics of a large system. Putting these reactions together, we can write the master equation for the density of cells of size <inline-formula><mml:math id="inf153"><mml:mi>n</mml:mi></mml:math></inline-formula>, <italic>c</italic><sub><italic>n</italic></sub>:<disp-formula id="equ6"><label>(5)</label><mml:math id="m6"><mml:mrow><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>c</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mi>n</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>δ</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>⁢</mml:mo><mml:mrow><mml:munder><mml:mo largeop="true" movablelimits="false" symmetric="true">∑</mml:mo><mml:mi>m</mml:mi></mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>m</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>In what follows we will use the terms ‘‘density’’ and ‘‘total number’’ interchangeably, measuring volume in units of our system size (i.e., number of cells per gut). The first moment of this equation gives the total number of cells,<disp-formula id="equ7"><label>(6)</label><mml:math id="m7"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>N</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>The zeroth moment gives the total number of clusters,<disp-formula id="equ8"><label>(7)</label><mml:math id="m8"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:munderover><mml:mo largeop="true" movablelimits="false" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi mathvariant="normal">∞</mml:mi></mml:munderover><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Here, the <italic>c</italic><sub>1</sub> term reflects the fact that in this model, cells must have size 2 or greater to fragment.</p><p>Finally, in a continuum picture, the size of a particular cluster is described by<disp-formula id="equ9"><label>(8)</label><mml:math id="m9"><mml:mrow><mml:mover accent="true"><mml:mi>n</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>A well-known heuristic derivation of the stationary distribution of this type of process is based on the relationship between the number of clusters, <inline-formula><mml:math id="inf154"><mml:mrow><mml:mi>M</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mi>n</mml:mi></mml:msub><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>, and the total number of cells, <inline-formula><mml:math id="inf155"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mi>n</mml:mi></mml:msub><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. The key to this derivation is to recognize that <inline-formula><mml:math id="inf156"><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> acts as a proxy for the rank of the cluster that arises at time <inline-formula><mml:math id="inf157"><mml:mi>t</mml:mi></mml:math></inline-formula>: for the <inline-formula><mml:math id="inf158"><mml:msup><mml:mi>j</mml:mi><mml:mtext>𝑡ℎ</mml:mtext></mml:msup></mml:math></inline-formula> cluster to arise, there are <inline-formula><mml:math id="inf159"><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> clusters that have a larger size, since the relative ordering of cluster sizes is preserved during exponential growth. For large sizes, when cluster rank is expressed as a function of cluster size it becomes proportional to the reverse cumulative distribution function, from which we obtain the density.</p><p>It turns out that the differences in behaviors of exponents measured in simulations for different values of <inline-formula><mml:math id="inf160"><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:math></inline-formula> can be understood by considering the importance of two terms in particular: the <italic>c</italic><sub>1</sub> term in the equation for <inline-formula><mml:math id="inf161"><mml:mi>M</mml:mi></mml:math></inline-formula>, and the <inline-formula><mml:math id="inf162"><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup></mml:mrow></mml:math></inline-formula> term in the equation for <inline-formula><mml:math id="inf163"><mml:mi>n</mml:mi></mml:math></inline-formula>.</p><p>Case 1: <inline-formula><mml:math id="inf164"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula></p><p>The total number of cells follows simple exponential growth, <inline-formula><mml:math id="inf165"><mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:mrow><mml:mtext>exp</mml:mtext><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. For <inline-formula><mml:math id="inf166"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, the total number of clusters is governed by the equation<disp-formula id="equ10"><label>(9)</label><mml:math id="m10"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>where the <italic>c</italic><sub>1</sub> term arises in our model because clusters can only fragment if they have size <inline-formula><mml:math id="inf167"><mml:mrow><mml:mi/><mml:mo>≥</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:math></inline-formula>. At long times, however, we expect <inline-formula><mml:math id="inf168"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>≪</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:math></inline-formula> and we therefore ignore this term, leading to <inline-formula><mml:math id="inf169"><mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mtext>exp</mml:mtext><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. A cluster that arises at time <inline-formula><mml:math id="inf170"><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:math></inline-formula> will at a later time <inline-formula><mml:math id="inf171"><mml:mi>t</mml:mi></mml:math></inline-formula> have a size <disp-formula id="equ11"><label>(10)</label><mml:math id="m11"><mml:mrow><mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mtext>e</mml:mtext><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>-</mml:mo><mml:mi>β</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Ignoring overall <inline-formula><mml:math id="inf172"><mml:mi>t</mml:mi></mml:math></inline-formula> dependence, we can express this size as a function of the rank of this cluster,<disp-formula id="equ12"><label>(11)</label><mml:math id="m12"><mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>∼</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Inverting this relationship, and invoking the proportionality between <inline-formula><mml:math id="inf173"><mml:mi>M</mml:mi></mml:math></inline-formula> and the reverse cumulative distribution, <inline-formula><mml:math id="inf174"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, results in<disp-formula id="equ13"><label>(12)</label><mml:math id="m13"><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>size</mml:mtext><mml:mo>&gt;</mml:mo><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>∼</mml:mo><mml:mi>M</mml:mi><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p> and differentiating produces the expected result <disp-formula id="equ14"><label>(13)</label><mml:math id="m14"><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>where <italic>c</italic><sub><italic>n</italic></sub> is normalized by <inline-formula><mml:math id="inf175"><mml:mrow><mml:mrow><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:math></inline-formula>. This result matches the traditional Yule-Simons process, where each organism divides at rate <inline-formula><mml:math id="inf176"><mml:mi>r</mml:mi></mml:math></inline-formula> and then mutates with probability <inline-formula><mml:math id="inf177"><mml:mi>ϵ</mml:mi></mml:math></inline-formula>, with <inline-formula><mml:math id="inf178"><mml:mrow><mml:mi>ϵ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>.</p><p>Case 2: <inline-formula><mml:math id="inf179"><mml:mrow><mml:mn>0</mml:mn><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula></p><p>In this case, when considering the equation for the size of a particular cluster, <disp-formula id="equ15"><label>(14)</label><mml:math id="m15"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>n</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup></mml:mrow></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>for <inline-formula><mml:math id="inf180"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> we ignore the second term on the right hand side. This term represents loss due to fragmentation, can be ignored for large sizes. Specifically, we consider sizes greater than a critical size, <inline-formula><mml:math id="inf181"><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, below which clusters will shrink. Ignoring this term, the size of a particular clusters that arose at time <inline-formula><mml:math id="inf182"><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:math></inline-formula> is given by<disp-formula id="equ16"><label>(15)</label><mml:math id="m16"><mml:mrow><mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mtext>e</mml:mtext><mml:mrow><mml:mi>r</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>The total number of clusters follows<disp-formula id="equ17"><label>(16)</label><mml:math id="m17"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:munderover><mml:mo largeop="true" movablelimits="false" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi mathvariant="normal">∞</mml:mi></mml:munderover><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Like above with <inline-formula><mml:math id="inf183"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, we ignore the <italic>c</italic><sub>1</sub> term. Unlike for <inline-formula><mml:math id="inf184"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, we don’t have a closed equation for the fractional moment <inline-formula><mml:math id="inf185"><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi mathvariant="normal">∞</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>. Therefore, we take the approach of making a power law ansatz<disp-formula id="equ18"><label>(17)</label><mml:math id="m18"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>≡</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>Z</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:math></disp-formula>with normalization<disp-formula id="equ19"><label>(18)</label><mml:math id="m19"><mml:mrow><mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>Z</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:munder><mml:mo largeop="true" movablelimits="false" symmetric="true">∑</mml:mo><mml:mi>n</mml:mi></mml:munder><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p> and then solve for the exponent μ self-consitently First, we approximate the sums by integrals and arrive at an equation for <inline-formula><mml:math id="inf186"><mml:mi>M</mml:mi></mml:math></inline-formula><disp-formula id="equ20"><label>(19)</label><mml:math id="m20"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Then we follow the same logic as for the <inline-formula><mml:math id="inf187"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> case. Solving for <inline-formula><mml:math id="inf188"><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, we get<disp-formula id="equ21"><label>(20)</label><mml:math id="m21"><mml:mrow><mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>∼</mml:mo><mml:mrow><mml:mtext>exp</mml:mtext><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Combining terms into<disp-formula id="equ22"><label>(21)</label><mml:math id="m22"><mml:mrow><mml:mrow><mml:mrow><mml:mi>η</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>μ</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>≡</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p> we then relate the size of a cluster that arose at time <inline-formula><mml:math id="inf189"><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:math></inline-formula> to the rank of that cluster,<inline-formula><mml:math id="inf190"><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>,<disp-formula id="equ23"><label>(22)</label><mml:math id="m23"><mml:mrow><mml:mi>n</mml:mi><mml:mo>∼</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mi>r</mml:mi><mml:mi>η</mml:mi></mml:mfrac></mml:mrow></mml:msup></mml:mrow></mml:math></disp-formula>from which we compute the scaling behavior of <italic>c</italic><sub><italic>n</italic></sub>,<disp-formula id="equ24"><label>(23)</label><mml:math id="m24"><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mi>η</mml:mi><mml:mi>r</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Equating this expression with the original ansatz, we arrive at a self-consistency equation for μ<disp-formula id="equ25"><label>(24)</label><mml:math id="m25"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mfrac><mml:mi>β</mml:mi><mml:mi>r</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>which we solve to obtain an exponent linear in the rates,<disp-formula id="equ26"><label>(25)</label><mml:math id="m26"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mi>β</mml:mi><mml:mi>r</mml:mi></mml:mfrac></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>This result is plotted in <xref ref-type="fig" rid="fig3">Figure 3D</xref> of the main text with <inline-formula><mml:math id="inf191"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> and agrees reasonably well with simulations.</p><p>Case 3: <inline-formula><mml:math id="inf192"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula></p><p>For <inline-formula><mml:math id="inf193"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>, the equation for <inline-formula><mml:math id="inf194"><mml:mi>M</mml:mi></mml:math></inline-formula> simplifies to<disp-formula id="equ27"><label>(26)</label><mml:math id="m27"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>The simulation results in <xref ref-type="fig" rid="fig3">Figure 3D</xref> indicate that the relationship between μ and <inline-formula><mml:math id="inf195"><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> is no longer linear, which we expect to be due to the <italic>c</italic><sub>1</sub> term reducing the propensity for fragmentation. That this single-cell effect is relevant for <inline-formula><mml:math id="inf196"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> makes sense because we expect most clusters to be of order 1, which would make <italic>c</italic><sub>1</sub> of order <inline-formula><mml:math id="inf197"><mml:mi>M</mml:mi></mml:math></inline-formula> . To account for this term explicitly, we make the same power law ansatz as before, <disp-formula id="equ28"><label>(27)</label><mml:math id="m28"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>≡</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mi>μ</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:math></disp-formula></p><p> and then extrapolate down to <inline-formula><mml:math id="inf198"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> to estimate <italic>c</italic><sub>1</sub>,<disp-formula id="equ29"><label>(28)</label><mml:math id="m29"><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>This extrapolation is purely a convenient approximation, as the distribution is likely not a true power law down to sizes of <inline-formula><mml:math id="inf199"><mml:mrow><mml:mi class="ltx_font_mathcaligraphic">𝒪</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>. With this ansatz, the equation for <inline-formula><mml:math id="inf200"><mml:mi>M</mml:mi></mml:math></inline-formula> reads<disp-formula id="equ30"><label>(29)</label><mml:math id="m30"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Combining terms we can define<disp-formula id="equ31"><label>(30)</label><mml:math id="m31"><mml:mrow><mml:mi>η</mml:mi><mml:mo>≡</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula></p><p> and write<disp-formula id="equ32"><label>(31)</label><mml:math id="m32"><mml:mrow><mml:mrow><mml:mover accent="true"><mml:mi>M</mml:mi><mml:mo>˙</mml:mo></mml:mover><mml:mo>≡</mml:mo><mml:mrow><mml:mi>η</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>μ</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Then, following the same protocol as above, we can relate the frequency of a cluster to its rank, <disp-formula id="equ33"><label>(32)</label><mml:math id="m33"><mml:mrow><mml:mi>n</mml:mi><mml:mo>∼</mml:mo><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mi>r</mml:mi><mml:mi>η</mml:mi></mml:mfrac></mml:mrow></mml:msup></mml:mrow></mml:math></disp-formula>from which we compute the scaling behavior of <italic>c</italic><sub><italic>n</italic></sub>, <disp-formula id="equ34"><label>(33)</label><mml:math id="m34"><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>∼</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mi>η</mml:mi><mml:mi>r</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Equating this result with the original ansatz, we get the self-consistency equation<disp-formula id="equ35"><label>(34)</label><mml:math id="m35"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mfrac><mml:mi>β</mml:mi><mml:mi>r</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>μ</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Solving this equation we get<disp-formula id="equ36"><label>(35)</label><mml:math id="m36"><mml:mrow><mml:mrow><mml:mi>μ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>This result is plotted in <xref ref-type="fig" rid="fig3">Figure 3D</xref> of the main text and agrees reasonably well with simulations, with notable deviations occurring once <inline-formula><mml:math id="inf201"><mml:mrow><mml:mrow><mml:mi>β</mml:mi><mml:mo>/</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mo>≈</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>.</p></sec><sec id="s10"><title>Discussion</title><p>In this model growth and fragmentation are treated as separate processes. This choice is convenient in the context of the full model including aggregation because classic reversible aggregation models (<xref ref-type="bibr" rid="bib16">Krapivsky et al., 2010</xref>) are contained within this general framework when growth and expulsion rates are set to zero, and also because fragmentation conserves total cell number. As a consequence of this choice, single cells are forbidden from fragmenting and don’t contribute to the total rate of fragmentation events. This feature differs from common evolutionary variants of this model for asexual populations, where growth and mutation are linked. In those models, all organisms divide and in each division have a probability of mutating (analogous to fragmenting in our model), so the rate of mutant production scales with the total population, including singletons. This is mostly a minor difference, but, as we showed, it does lead to different behaviors of the resulting distribution exponents in the limit of fast fragmentation.</p><p>We showed that we can account for this effect when it is important (in the case <inline-formula><mml:math id="inf202"><mml:mrow><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>), but we cannot say <italic>when</italic> it will be important. That is because this effect depends on the number of single cells, which lies outside the regime of our large-size asymptotics that underly the continuum approximation. Ultimately the effect depends on how large the the number of single cells, <italic>c</italic><sub>1</sub>, is compared to the fractional moment <inline-formula><mml:math id="inf203"><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mi>n</mml:mi></mml:msub><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:msub><mml:mi>ν</mml:mi><mml:mi>F</mml:mi></mml:msub></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>. If the distribution has a significant shoulder, than <italic>c</italic><sub>1</sub> may be smaller than the extrapolation of the power-law form down to <inline-formula><mml:math id="inf204"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>. In that case, the single-cell effect may be less important than this extrapolation would predict it to be.</p></sec></boxed-text></sec></app></app-group></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.71105.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Wood</surname><given-names>Kevin B</given-names></name><role>Reviewing Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Ludington</surname><given-names>William B</given-names></name><role>Reviewer</role><aff><institution>Carnegie Institute</institution><country>United States</country></aff></contrib><contrib contrib-type="reviewer"><name><surname>Iyer-Biswas</surname><given-names>Srividya</given-names> </name><role>Reviewer</role><aff><institution>Purdue University</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2021.06.08.447595">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2021.06.08.447595v1">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>This study investigates bacterial aggregates in the larval zebrafish gut, revealing that the cluster distributions observed in vivo share mathematical similarities with those that occur during gel formation in soft condensed matter physics. The work links complex in vivo dynamics with a simple and potentially general biophysical phenomenon – making it an example of a &quot;living&quot; material in a natural biological context – and opens the door to new studies on the interplay between cell clustering and microbiome dynamics in a living host.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Gut bacterial aggregates as living gels&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, and the evaluation has been overseen by a Reviewing Editor and Wendy Garrett as the Senior Editor. The following individuals involved in review of your submission have agreed to reveal their identity: William B Ludington (Reviewer #1); Srividya Iyer-Biswas (Reviewer #3).</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>The reviewers and editors appreciate the authors' innovative study on the physical basis for spatial aggregation in gut microbial communities, a timely topic that bridges soft condensed matter physics and microbiology. The reviewers noted many strengths of the work, including the elegant combination of model system and a simple biophysical model, the mechanistic insight provided by the model, the compelling agreement between data and theoretical predictions, and the potential generality of the proposed mechanism to other systems. We also commend the authors on code that is well documented, compact, and easily accessible.</p><p>We also identified several places where additional discussion would improve clarity and help the message reach a broader audience.</p><p>1) We would like to see an expanded explanation of the theoretical model to make the paper more accessible to a broader audience. For example, readers with a more biological background may find a more detailed discussion of the intuitive and mechanistic basis of the model helpful (see, in particular, the suggestions from Reviewer 2, including suggestions about adding context related to aggregation in other biological systems). Similarly, readers with a more physics background might find it helpful to connect the model (and the phenomenon) more directly to previous work – for example, the authors might note that this is a variant of a preferential attachment (Yule-Simons) model, and the authors could clarify the motivation for the particular model (e.g. what is special about the slope -1?). This discussion should also include further explanation of the gel characterization, and perhaps a brief introduction to gels in the Intro section, particularly since gel is mentioned in the title and abstract.</p><p>2) While not required, a sensitivity analysis (or similar discussion) that addresses potential sources of error in the parameter estimates would add to the work. How robust are the results to (for example) experimental errors in the measurements? In the absence of a detailed analysis, the authors could explicitly discuss any limitations imposed on the results due to these uncertainties.</p><p>3) The reviewers point out several places where the text could be strengthened and clarified, and also places where a general reader might benefit from a brief mention of future extensions or future work.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>I really enjoyed reading this paper, and I learned from it. My comments below are meant to help improve the work.</p><p>1. Figure 1C: do the smaller clusters tend to be closer to the larger cluster? Why? Is that due to the interaction with fluid flows?</p><p>2. I think readers might find it helpful if the authors could note that this is a variant of a preferential attachment model, which I believe is better known than the Yule-Simons name.</p><p>3. Because single gut dynamic data is available, I would have appreciated seeing some time lapse imaging of clusters merging. Are fragmentations of large clusters into smaller clusters observed?</p><p>4. I found the use of the term fragmentation somewhat confusing. To me what is described with single cells falling off the cluster is more akin to dispersal of propagules. Is there actual fragmentation of large clusters into medium clusters? Or is the vast majority of fragmentation due to single cells? In that case, a Leslie matrix approach might be more appropriate. If the authors could clarify whether the fragmentation of a big cluster into medium clusters occurs, that would solve the point of confusion for me. If this type of fragmentation does not occur, fragmentation might not be the best term.</p><p>5. To make the paper align better with the title, I would have found it helpful if there had been some discussion of the importance of gels in the introduction. An argument could be that it is too mundane, but if such a simple and well understood model can be mechanistically applied to a complex system such as the gut microbiome, I would argue that is a major advance. The mechanistic cases presented in the supplement do a nice job of establishing the role of individual parameters in shaping the distribution.</p><p>6. I struggled a bit with the fact that expulsion was not considered until later in the manuscript. If the major factors examined (including expulsion) were mentioned in the last paragraph of the introduction, it would have helped me be patient in reading.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>1. Results section: a more detailed explanation in the main text of how these processes are modeled by the theory would provide added support to the generalizability of the model and enable others to clearly see how they might consider test predictions in their own systems. For example, the description of the biological interpretation and limits of the distribution exponent µ would be very valuable.</p><p>2. Results section: How robust are the size distributions and model to potential error in bacterial cell enumeration? The number of cells per aggregate was estimated by dividing total fluorescence intensity by mean intensity of single cells. This could be OK if all cells were phenotypically similar throughout the aggregate, however it's quite possible that growth rate varies spatially and that cells in the center of the aggregate may be slower growing or dead, which would lower fluorescence. Do the authors have any data contradicting this? Aggregates also auto fluoresce more strongly than single cells, which could skew cell counts by signal intensity. Did the authors perform controls that calibrated or confirmed their ability to use signal intensity as a cell count measurement?</p><p>3. Results section: it would be valuable if the authors could expand their analysis to describe the range of parameter values within which their model is realistic. For example, what rates of cell division are too fast or too slow to provide the observable distributions? What cell division rates are observed in the data? Similarly, can the authors quantify fragmentation from their image data and relate that to the rates of fragmentation in their model? Could this explain why aggregates do not form in their non-aggregating wild-type strain?</p><p>4. Results and Discussion sections: A more thorough explanation about what aggregate dynamics parameters (e.g., cell division and fragmentation) makes this process like a gel is required, particularly given the title of the paper. Were the likening of bacterial aggregates a key goal of this paper, it would be stronger to have the aggregate size model explicitly compared to a soft matter model (i.e., a model of gelation transition etc.). Alternatively, the model presented in the paper has value independent of the conclusion about gels, and this value may be highlighted perhaps by reducing the emphasis on the gel-like nature of this system in the title and abstract.</p><p>5. Discussion section: The paper would be strengthened with a discussion on how different bacterial phenotypes (e.g. surface adhesiveness, heterogeneity in growth) may affect aggregate formation. It may be that mechanistic differences between some phenotypes are functionally similar for the model, enabling the model to apply across different species. For example, marine phytoplankton species have different mechanisms of cell aggregation (adhesive cell surface properties vs. mucus-mediated coagulation) [Kiørboe and Hansen, 1993]. A discussion on whether such differences can or cannot be treated by this model would help readers understand when this model is most appropriately applied.</p><p>• Thomas Kiørboe, Jørgen L.S. Hansen, Phytoplankton aggregate formation: observations of patterns and mechanisms of cell sticking and the significance of exopolymeric material, Journal of Plankton Research, Volume 15, Issue 9, 1993, Pages 993-1018, https://doi.org/10.1093/plankt/15.9.993</p><p>6. Discussion section: The suggestion that this model could be generally applied across diverse guts would be strengthened with a discussion on how the proposed model incorporates or account for these environmental factors. Fluid flow and cell/aggregate morphology, for example, are known to impact microbial aggregate formation. Could the authors discuss or speculate on how such processes affect the rates of aggregate growth, fragmentation, fusion and loss by their model?</p><p>• Kiørboe, Thomas. A Mechanistic Approach to Plankton Ecology, Princeton: Princeton University Press, 2018. https://doi.org/10.1515/9780691190310</p><p>• Jonasz Słomka, Roman Stocker, On the collision of rods in a quiescent fluid, Proceedings of the National Academy of Sciences Feb 2020, 117 (7) 3372-3374; DOI: 10.1073/pnas.1917163117</p><p>• Falkovich, G., Fouxon, A. and Stepanov, M. Acceleration of rain initiation by cloud turbulence. Nature 419, 151-154 (2002). https://doi.org/10.1038/nature00983</p><p>7. Discussion section: The authors suggest analysis of fecal aggregates from other guts (mouse, human); however, these aggregates are different from those included in their distributions as fecal aggregates are those to be expelled rather than maintained in the gut. Could the authors please describe more specifically how size distributions of expelled cells can be used to determine the rates in their model?</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>To address the goal of characterizing the distributions of gut bacterial aggregate sizes, the authors have motivated, from the ground up, an excellent first-principles-based model, and have added in complexity in layers. The model is capable of describing the aggregate size dynamics for a wide variety of gut bacteria in zebrafish. Given the specifics of the first-principles based approach used, it is plausible that it's directly applicable to gut bacteria in other animals too. Sufficient complexity is systematically added, clearly distinguishing the individual effects of each added factor on the model. Overall, the the model sufficiently explains the important features of the gut bacterial aggregate size distribution, namely, the initial power law and the final plateau.</p><p>That said, a minor issue is that the initial motivation behind building the model in this way seems somewhat unnecessary. The authors motivated the basis for the model by claiming P(size &gt; n) ∼ n−1, using Fig. 2. But the model seems to work for any slope (depending on fragmentation rates etc). So why is the slope of -1 special?</p><p>Also, in Fig. 2, since the dashed line is separated from the actual data, it is tricky to visually compare them, and some experimental plots appear to have quite different slopes. It would be helpful if the best fit slope for the small n part is also reported.</p><p>Another minor issue: they claim that the decrease in size due to fragmentation is linked to cell division at the surface. However, after the cell divides, if only one daughter leaves the cluster then it shouldn't change the cluster's size (since size is measured in terms of numbers of cells rather than total volume). But if both daughters leave the surface, then what does it have to do with division?</p><p>These are minor issues which can be readily addressed through clear prose and presentation in the manuscript. They do not affect the model or the overall results.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.71105.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>The reviewers and editors appreciate the authors' innovative study on the physical basis for spatial aggregation in gut microbial communities, a timely topic that bridges soft condensed matter physics and microbiology. The reviewers noted many strengths of the work, including the elegant combination of model system and a simple biophysical model, the mechanistic insight provided by the model, the compelling agreement between data and theoretical predictions, and the potential generality of the proposed mechanism to other systems. We also commend the authors on code that is well documented, compact, and easily accessible.</p><p>We also identified several places where additional discussion would improve clarity and help the message reach a broader audience.</p><p>1) We would like to see an expanded explanation of the theoretical model to make the paper more accessible to a broader audience. For example, readers with a more biological background may find a more detailed discussion of the intuitive and mechanistic basis of the model helpful (see, in particular, the suggestions from Reviewer 2, including suggestions about adding context related to aggregation in other biological systems). Similarly, readers with a more physics background might find it helpful to connect the model (and the phenomenon) more directly to previous work – for example, the authors might note that this is a variant of a preferential attachment (Yule-Simons) model, and the authors could clarify the motivation for the particular model (e.g. what is special about the slope -1?). This discussion should also include further explanation of the gel characterization, and perhaps a brief introduction to gels in the Intro section, particularly since gel is mentioned in the title and abstract.</p></disp-quote><p>We have added several sentences about gels to the introduction, noting both how they exemplify the link between size distributions and underlying mechanisms, and the importance of gels in living systems. We have also added sentences about preferential attachment and expulsion to the introduction, and additional explanation of the slope -1.</p><p>We provide paragraph generally introducing the modeling framework, its philosophy, and the importance of stochasticity.</p><p>We have added a paragraph to discussion about gel characterization, describing both the nature of the bacterial gel and its consequences for the scaling plots.</p><disp-quote content-type="editor-comment"><p>2) While not required, a sensitivity analysis (or similar discussion) that addresses potential sources of error in the parameter estimates would add to the work. How robust are the results to (for example) experimental errors in the measurements? In the absence of a detailed analysis, the authors could explicitly discuss any limitations imposed on the results due to these uncertainties.</p></disp-quote><p>We agree that a sensitivity analysis / discussion of scaling uncertainties would be useful. We have performed a sensitivity analysis; we describe this in the revised text, and also provide a new supplemental table of the best-fit values for the scaling exponents. In brief, the largest source of uncertainty in the cluster data is the mis-identification of single bacterial cells; analyzing the cumulative distribution slopes with and without the single-cell datapoints gives slopes near -1 in both cases (Table S1). Also, it is notoriously challenging to fit power laws; we consider two different methods, a linear fit to log P (size &gt; n) vs. log n, and maximum likelihood estimation; they agree within uncertainties (see text). We also added a figure supplement addressing the potential issue of fluorescence heterogeneity within aggregates, which we find to be minor.</p><disp-quote content-type="editor-comment"><p>3) The reviewers point out several places where the text could be strengthened and clarified, and also places where a general reader might benefit from a brief mention of future extensions or future work.</p></disp-quote><p>We have made several changes to the text to clarify our meaning and expand our discussions. Some of these overlap the above suggestions / responses. Others are separate:</p><p>– We specify the nature of the <italic>Vibrio</italic> mutants;</p><p>– We clarified the relationship between the probability density and the cumulative distribution;</p><p>– We changed the Figure 3 caption;</p><p>– We added a paragraph to discussion about directly measuring aggregation and fragmentation rates as a useful future direction;</p><p>– We added a line about “chipping” fragmentation of individual bacteria, in contrast to breakup into large clusters; the latter is not observed;</p><p>– We added a line explicitly defining distribution exponent \mu in text;</p><p>– We added a line to methods specifying that each strain was tagged with GFP;</p><p>– We added a line about range of growth rates;</p><p>– We added a legend to data distributions emphasizing the “guide to the eye” slope;</p><p>– We discuss how different bacterial phenotypes can lead to different rates.</p><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>I really enjoyed reading this paper, and I learned from it. My comments below are meant to help improve the work.</p><p>1. Figure 1C: do the smaller clusters tend to be closer to the larger cluster? Why? Is that due to the interaction with fluid flows?</p></disp-quote><p>This is a good question, but we hesitate to give a quantitative answer. We have expanded the discussion to better note experimental challenges, the surmounting of which may give more direct insight into aggregation and fragmentation behaviors.</p><disp-quote content-type="editor-comment"><p>2. I think readers might find it helpful if the authors could note that this is a variant of a preferential attachment model, which I believe is better known than the Yule-Simons name.</p></disp-quote><p>Yes; we note this name also in the revised manuscript. We caution that the terminology is a bit confusing, since “attachment” in a network model is like growth in our model!</p><disp-quote content-type="editor-comment"><p>3. Because single gut dynamic data is available, I would have appreciated seeing some time lapse imaging of clusters merging. Are fragmentations of large clusters into smaller clusters observed?</p></disp-quote><p>As with #1, this is challenging to be precise about, and we hesitate to give rough data. We have expanded the discussion to better note experimental challenges, the surmounting of which may give more direct insight into aggregation and fragmentation behaviors.</p><disp-quote content-type="editor-comment"><p>4. I found the use of the term fragmentation somewhat confusing. To me what is described with single cells falling off the cluster is more akin to dispersal of propagules. Is there actual fragmentation of large clusters into medium clusters? Or is the vast majority of fragmentation due to single cells? In that case, a Leslie matrix approach might be more appropriate. If the authors could clarify whether the fragmentation of a big cluster into medium clusters occurs, that would solve the point of confusion for me. If this type of fragmentation does not occur, fragmentation might not be the best term.</p></disp-quote><p>This is a good point. Our terminology matches the literature from the perspective of the model, but perhaps isn’t ideal from the perspective of the organisms. (We’ve struggled with the terminology, and there probably isn’t an ideal solution.) We have added clarifying text distinguishing single cell fragmentation from “symmetric” or larger-scale break up.</p><disp-quote content-type="editor-comment"><p>5. To make the paper align better with the title, I would have found it helpful if there had been some discussion of the importance of gels in the introduction. An argument could be that it is too mundane, but if such a simple and well understood model can be mechanistically applied to a complex system such as the gut microbiome, I would argue that is a major advance. The mechanistic cases presented in the supplement do a nice job of establishing the role of individual parameters in shaping the distribution.</p></disp-quote><p>We agree. Please see the “essential” responses; we have added text to the introduction and discussion.</p><disp-quote content-type="editor-comment"><p>6. I struggled a bit with the fact that expulsion was not considered until later in the manuscript. If the major factors examined (including expulsion) were mentioned in the last paragraph of the introduction, it would have helped me be patient in reading.</p></disp-quote><p>We agree. Please see the “essential” responses; we have added a line about this to the introduction.</p><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>1. Results section: a more detailed explanation in the main text of how these processes are modeled by the theory would provide added support to the generalizability of the model and enable others to clearly see how they might consider test predictions in their own systems. For example, the description of the biological interpretation and limits of the distribution exponent µ would be very valuable.</p></disp-quote><p>Please see above, regarding our new and more expansive introduction of the model.</p><disp-quote content-type="editor-comment"><p>2. Results section: How robust are the size distributions and model to potential error in bacterial cell enumeration? The number of cells per aggregate was estimated by dividing total fluorescence intensity by mean intensity of single cells. This could be OK if all cells were phenotypically similar throughout the aggregate, however it's quite possible that growth rate varies spatially and that cells in the center of the aggregate may be slower growing or dead, which would lower fluorescence. Do the authors have any data contradicting this? Aggregates also auto fluoresce more strongly than single cells, which could skew cell counts by signal intensity. Did the authors perform controls that calibrated or confirmed their ability to use signal intensity as a cell count measurement?</p></disp-quote><p>Please see our “essential” reply above on the new uncertainty / sensitivity analysis we have included.</p><disp-quote content-type="editor-comment"><p>3. Results section: it would be valuable if the authors could expand their analysis to describe the range of parameter values within which their model is realistic. For example, what rates of cell division are too fast or too slow to provide the observable distributions? What cell division rates are observed in the data? Similarly, can the authors quantify fragmentation from their image data and relate that to the rates of fragmentation in their model? Could this explain why aggregates do not form in their non-aggregating wild-type strain?</p></disp-quote><p>We have added text about the range of growth rates we have measured. We now discuss the range of rates for all four processes.</p><disp-quote content-type="editor-comment"><p>4. Results and Discussion sections: A more thorough explanation about what aggregate dynamics parameters (e.g., cell division and fragmentation) makes this process like a gel is required, particularly given the title of the paper. Were the likening of bacterial aggregates a key goal of this paper, it would be stronger to have the aggregate size model explicitly compared to a soft matter model (i.e., a model of gelation transition etc.). Alternatively, the model presented in the paper has value independent of the conclusion about gels, and this value may be highlighted perhaps by reducing the emphasis on the gel-like nature of this system in the title and abstract.</p></disp-quote><p>Done – please see the revised Discussion, and our “essential” response above.</p><disp-quote content-type="editor-comment"><p>5. Discussion section: The paper would be strengthened with a discussion on how different bacterial phenotypes (e.g. surface adhesiveness, heterogeneity in growth) may affect aggregate formation. It may be that mechanistic differences between some phenotypes are functionally similar for the model, enabling the model to apply across different species. For example, marine phytoplankton species have different mechanisms of cell aggregation (adhesive cell surface properties vs. mucus-mediated coagulation) [Kiørboe and Hansen, 1993]. A discussion on whether such differences can or cannot be treated by this model would help readers understand when this model is most appropriately applied.</p><p>• Thomas Kiørboe, Jørgen L.S. Hansen, Phytoplankton aggregate formation: observations of patterns and mechanisms of cell sticking and the significance of exopolymeric material, Journal of Plankton Research, Volume 15, Issue 9, 1993, Pages 993-1018, https://doi.org/10.1093/plankt/15.9.993</p></disp-quote><p>These are good points. The manuscript now includes a (brief) discussion of both bacterial phenotypes (motility) and more abstract settings / morphologies such as diffusion limited aggregation and fractal forms.</p><disp-quote content-type="editor-comment"><p>6. Discussion section: The suggestion that this model could be generally applied across diverse guts would be strengthened with a discussion on how the proposed model incorporates or account for these environmental factors. Fluid flow and cell/aggregate morphology, for example, are known to impact microbial aggregate formation. Could the authors discuss or speculate on how such processes affect the rates of aggregate growth, fragmentation, fusion and loss by their model?</p><p>• Kiørboe, Thomas. A Mechanistic Approach to Plankton Ecology, Princeton: Princeton University Press, 2018. https://doi.org/10.1515/9780691190310</p><p>• Jonasz Słomka, Roman Stocker, On the collision of rods in a quiescent fluid, Proceedings of the National Academy of Sciences Feb 2020, 117 (7) 3372-3374; DOI: 10.1073/pnas.1917163117</p><p>• Falkovich, G., Fouxon, A. and Stepanov, M. Acceleration of rain initiation by cloud turbulence. Nature 419, 151-154 (2002). https://doi.org/10.1038/nature00983</p></disp-quote><p>We have noted that investigating how fluid flow governs aggregation scaling would be a good topic of future study.</p><disp-quote content-type="editor-comment"><p>7. Discussion section: The authors suggest analysis of fecal aggregates from other guts (mouse, human); however, these aggregates are different from those included in their distributions as fecal aggregates are those to be expelled rather than maintained in the gut. Could the authors please describe more specifically how size distributions of expelled cells can be used to determine the rates in their model?</p></disp-quote><p>We don’t guarantee that this approach will work! We think our text is clear that this is a suggestion for future study and validation.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>To address the goal of characterizing the distributions of gut bacterial aggregate sizes, the authors have motivated, from the ground up, an excellent first-principles-based model, and have added in complexity in layers. The model is capable of describing the aggregate size dynamics for a wide variety of gut bacteria in zebrafish. Given the specifics of the first-principles based approach used, it is plausible that it's directly applicable to gut bacteria in other animals too. Sufficient complexity is systematically added, clearly distinguishing the individual effects of each added factor on the model. Overall, the the model sufficiently explains the important features of the gut bacterial aggregate size distribution, namely, the initial power law and the final plateau.</p><p>That said, a minor issue is that the initial motivation behind building the model in this way seems somewhat unnecessary. The authors motivated the basis for the model by claiming P(size &gt; n) ∼ n−1, using Fig. 2. But the model seems to work for any slope (depending on fragmentation rates etc). So why is the slope of -1 special?</p></disp-quote><p>We have clarified this in the Results, explaining that the slope of -1 (only) robustly emerges from growth/fragmentation</p><disp-quote content-type="editor-comment"><p>Also, in Fig. 2, since the dashed line is separated from the actual data, it is tricky to visually compare them, and some experimental plots appear to have quite different slopes. It would be helpful if the best fit slope for the small n part is also reported.</p></disp-quote><p>We now include all the slope values in a table.</p><disp-quote content-type="editor-comment"><p>Another minor issue: they claim that the decrease in size due to fragmentation is linked to cell division at the surface. However, after the cell divides, if only one daughter leaves the cluster then it shouldn't change the cluster's size (since size is measured in terms of numbers of cells rather than total volume). But if both daughters leave the surface, then what does it have to do with division?</p></disp-quote><p>We have revised the text to clarify what is meant by fragmentation.</p></body></sub-article></article>