<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">82979</article-id><article-id pub-id-type="doi">10.7554/eLife.82979</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Comparative genomics reveals insight into the evolutionary origin of massively scrambled genomes</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-215484"><name><surname>Feng</surname><given-names>Yi</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-2393-1700</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-38090"><name><surname>Neme</surname><given-names>Rafik</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-8462-5291</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-22635"><name><surname>Beh</surname><given-names>Leslie Y</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf3"/></contrib><contrib contrib-type="author" id="author-290740"><name><surname>Chen</surname><given-names>Xiao</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-1432-268X</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf4"/></contrib><contrib contrib-type="author" id="author-290741"><name><surname>Braun</surname><given-names>Jasper</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-1250-4399</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="pa1">†</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-290742"><name><surname>Lu</surname><given-names>Michael W</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4926-8839</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes" id="author-18151"><name><surname>Landweber</surname><given-names>Laura F</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-7030-8540</contrib-id><email>Laura.Landweber@columbia.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf2"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hj8s172</institution-id><institution>Departments of Biochemistry and Molecular Biophysics and Biological Sciences, Columbia University</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/031e6xm45</institution-id><institution>Department of Chemistry and Biology, Universidad del Norte</institution></institution-wrap><addr-line><named-content content-type="city">Barranquilla</named-content></addr-line><country>Colombia</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00fcszb13</institution-id><institution>Pacific Biosciences</institution></institution-wrap><addr-line><named-content content-type="city">Menlo Park</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/032db5x82</institution-id><institution>Department of Mathematics and Statistics, University of South Florida</institution></institution-wrap><addr-line><named-content content-type="city">Tampa</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><author-notes><fn fn-type="present-address" id="pa1"><label>†</label><p>Department of Pathology, Beth Israel Deaconess Medical Center, Boston, United States</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>24</day><month>11</month><year>2022</year></pub-date><pub-date pub-type="collection"><year>2022</year></pub-date><volume>11</volume><elocation-id>e82979</elocation-id><history><date date-type="received" iso-8601-date="2022-08-25"><day>25</day><month>08</month><year>2022</year></date><date date-type="accepted" iso-8601-date="2022-11-03"><day>03</day><month>11</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at .</event-desc><date date-type="preprint" iso-8601-date="2022-05-10"><day>10</day><month>05</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.05.09.490778"/></event></pub-history><permissions><copyright-statement>© 2022, Feng et al</copyright-statement><copyright-year>2022</copyright-year><copyright-holder>Feng et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-82979-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-82979-figures-v2.pdf"/><abstract><p>Ciliates are microbial eukaryotes that undergo extensive programmed genome rearrangement, a natural genome editing process that converts long germline chromosomes into smaller gene-rich somatic chromosomes. Three well-studied ciliates include <italic>Oxytricha trifallax</italic>, <italic>Tetrahymena thermophila,</italic> and <italic>Paramecium tetraurelia</italic>, but only the <italic>Oxytricha</italic> lineage has a massively scrambled genome, whose assembly during development requires hundreds of thousands of precisely programmed DNA joining events, representing the most complex genome dynamics of any known organism. Here we study the emergence of such complex genomes by examining the origin and evolution of discontinuous and scrambled genes in the <italic>Oxytricha</italic> lineage. This study compares six genomes from three species, the germline and somatic genomes for <italic>Euplotes woodruffi</italic>, <italic>Tetmemena sp</italic>., and the model ciliate <italic>O. trifallax</italic>. We sequenced, assembled, and annotated the germline and somatic genomes of <italic>E. woodruffi,</italic> which provides an outgroup<italic>,</italic> and the germline genome of <italic>Tetmemena sp</italic>. We find that the germline genome of <italic>Tetmemena</italic> is as massively scrambled and interrupted as <italic>Oxytricha</italic>’s: 13.6% of its gene loci require programmed translocations and/or inversions, with some genes requiring hundreds of precise gene editing events during development. This study revealed that the earlier diverged spirotrich, <italic>E. woodruffi</italic>, also has a scrambled genome, but only roughly half as many loci (7.3%) are scrambled. Furthermore, its scrambled genes are less complex, together supporting the position of <italic>Euplotes</italic> as a possible evolutionary intermediate in this lineage, in the process of accumulating complex evolutionary genome rearrangements, all of which require extensive repair to assemble functional coding regions. Comparative analysis also reveals that scrambled loci are often associated with local duplications, supporting a gradual model for the origin of complex, scrambled genomes via many small events of DNA duplication and decay.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>genome rearrangement</kwd><kwd>transposable elements</kwd><kwd><italic>Oxytricha trifallax</italic></kwd><kwd>Euplotes</kwd><kwd>scrambled gene</kwd><kwd>comparative genomics</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Other</kwd><kwd><italic>Oxytricha trifallax</italic></kwd><kwd><italic>Tetmemena</italic></kwd><kwd><italic>Euplotes woodruffi</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM122555</award-id><principal-award-recipient><name><surname>Landweber</surname><given-names>Laura F</given-names></name><name><surname>Feng</surname><given-names>Yi</given-names></name><name><surname>Beh</surname><given-names>Leslie Y</given-names></name><name><surname>Lu</surname><given-names>Michael W</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>DMS1764366</award-id><principal-award-recipient><name><surname>Feng</surname><given-names>Yi</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution>Pew Latin American Fellows Program</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Neme</surname><given-names>Rafik</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100008982</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>DBI1062432</award-id><principal-award-recipient><name><surname>Feng</surname><given-names>Yi</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100008982</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>ABI1458641</award-id><principal-award-recipient><name><surname>Feng</surname><given-names>Yi</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100008982</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>ABI1759906</award-id><principal-award-recipient><name><surname>Feng</surname><given-names>Yi</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The comparison of three ciliate species that share complex pathways for natural genome editing allows capture of intermediate states in the acquisition of scrambled genes and elucidating a pathway for the origin and evolution of extremely rearranged chromosomes.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Organisms do not always contain a single, static genome. Programmed genome editing is a naturally occurring and essential part of development in many organisms, including ciliates (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>), nematodes (<xref ref-type="bibr" rid="bib75">Mitreva et al., 2005</xref>), lampreys (<xref ref-type="bibr" rid="bib97">Smith et al., 2012</xref>), and zebra finches (<xref ref-type="bibr" rid="bib8">Biederman et al., 2018</xref>). Most of these events involve precise removal and rejoining of large regions of DNA during postzygotic differentiation of a somatic genome from a germline genome. Ciliates are microbial eukaryotes with two types of nuclei: a somatic macronucleus (MAC) that differentiates from a germline micronucleus (MIC). In the model ciliate <italic>Oxytricha</italic>, the MAC is entirely active chromatin (<xref ref-type="bibr" rid="bib7">Beh et al., 2019</xref>) and the hub of transcription. The three species that we compare are all spirotrichs, which have gene-sized ‘nanochromosomes’ in the MAC, present at high copy number (<xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>; <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>; <xref ref-type="bibr" rid="bib108">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="bib23">Chen et al., 2019</xref>; <xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>; <xref ref-type="bibr" rid="bib105">Vinogradov et al., 2012</xref>). The diploid MIC participates in sexual reproduction, but its megabase-sized chromosomes are mostly transcriptionally silent.</p><p>Gene loci are often arranged discontinuously in the MIC, with short genic segments called macronuclear destined sequences (MDSs), interrupted by stretches of non-coding DNA called internally eliminated sequences (IESs) (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). During sexual development, a new MAC genome rearranges from a copy of the zygotic MIC genome. MDSs join in the correct order and orientation, whereas MIC-limited genomic regions undergo programmed deletion, including repetitive elements, intergenic regions, and IESs (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Though analogous to intron splicing, these events occur on DNA. The MDSs for some MAC chromosomes are <italic>scrambled</italic> if they require translocation or inversion during MAC development (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Pairs of short repeats, called <italic>pointers</italic>, are present at MDS-IES junctions in both scrambled and nonscrambled loci (<xref ref-type="bibr" rid="bib74">Mitcham et al., 1992</xref>; <xref ref-type="bibr" rid="bib83">Prescott, 1994</xref>). Pointer sequences are present twice in the MIC, at the end of MDS <italic>n</italic> and the beginning of MDS n+1. One copy of the repeat is retained at each MDS-MDS junction in a mature MAC chromosome (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). These microhomologous regions help guide MDS recombination, but most are non-unique, and the shortest pointers are just 2 bp. Thousands of long, noncoding template RNAs collectively program MDS joining (<xref ref-type="bibr" rid="bib79">Nowacki et al., 2008</xref>; <xref ref-type="bibr" rid="bib64">Lindblad et al., 2017</xref>; <xref ref-type="bibr" rid="bib111">Yerlici and Landweber, 2014</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Genome rearrangements in representative ciliate species.</title><p>(<bold>A</bold>) Diagram of genome rearrangement in <italic>Oxytricha</italic>. Each ciliate cell contains a somatic macronucleus (MAC) and a germline micronucleus (MIC). During development, the MAC genome rearranges from a copy of the MIC genome. (1) Nonscrambled genes rearrange simply by joining consecutive macronuclear destined sequences (MDSs, blue boxes) and removing internal eliminated sequences (IESs, thin lines). (2) Rearrangement of scrambled genes requires MDS translocation and/or inversion. Pointers are microhomologous sequences (colored vertical bars) present in two copies in the MIC and only one copy in the MAC where consecutive MDSs recombine. (<bold>B</bold>) Comparison of genome rearrangement features of representative ciliates and the non-ciliate <italic>Plasmodium falciparum</italic> as an outgroup (phylogenetic information is based on <xref ref-type="bibr" rid="bib82">Parfrey et al., 2011</xref>; <xref ref-type="bibr" rid="bib11">Bracht et al., 2013</xref>). Conclusions from this study are shown in bold. * indicates that some scrambled pointers in <italic>Euplotes woodruffi</italic> are much longer, as discussed in the results. Statistics for pointers ≤30 bp in <italic>E. woodruffi</italic> are shown. Table information derives from the following sources: 1 - <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>; 2 - <xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>; 3 - <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; 4 - <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>; 5 - <xref ref-type="bibr" rid="bib91">Sheng et al., 2020</xref>; 6 - <xref ref-type="bibr" rid="bib30">Eisen et al., 2006</xref>; 7 - <xref ref-type="bibr" rid="bib48">Hamilton et al., 2016</xref>; 8 - <xref ref-type="bibr" rid="bib4">Aury et al., 2006</xref>; 9 - <xref ref-type="bibr" rid="bib44">Guérin et al., 2017</xref>; 10 - <xref ref-type="bibr" rid="bib3">Arnaiz et al., 2012</xref>; 11 - <xref ref-type="bibr" rid="bib85">Riley and Katz, 2001</xref>; 12 - <xref ref-type="bibr" rid="bib69">Maurer-Alcalá et al., 2018a</xref>; 13 - <xref ref-type="bibr" rid="bib54">Katz and Kovner, 2010</xref>; 14 - <xref ref-type="bibr" rid="bib39">Gao et al., 2014</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig1-v2.tif"/></fig><p>Numerous studies have inferred the possible scope of genome rearrangement in different ciliate species using partial genome surveys. In <italic>Paramecium</italic>, PiggyMac-depleted cells fail to remove MIC-limited regions properly, which provided a resource to annotate ~45,000 IESs prior to assembly of a draft MIC genome (<xref ref-type="bibr" rid="bib3">Arnaiz et al., 2012</xref>). The use of single-cell sequencing has allowed pilot studies to sample partial MIC genomes of diverse species (<xref ref-type="bibr" rid="bib23">Chen et al., 2019</xref>; <xref ref-type="bibr" rid="bib69">Maurer-Alcalá et al., 2018a</xref>; <xref ref-type="bibr" rid="bib70">Maurer-Alcalá et al., 2018b</xref>; <xref ref-type="bibr" rid="bib98">Smith et al., 2020</xref>). Alignment of tentative MIC reads to either assembled MAC genomes or single-cell transcriptome data predicts over 20 candidate scrambled loci in two basal ciliates, <italic>Loxodes sp</italic>. and <italic>Blepharisma americanum</italic> (<xref ref-type="bibr" rid="bib70">Maurer-Alcalá et al., 2018b</xref>) and hundreds of candidate loci in the tintinnid <italic>Schmidingerella arcuata</italic> (<xref ref-type="bibr" rid="bib98">Smith et al., 2020</xref>). Nearly one-third (31%) of approximately 5000 surveyed transcripts may be scrambled in <italic>Chilodonella uncinata</italic> (<xref ref-type="bibr" rid="bib69">Maurer-Alcalá et al., 2018a</xref>, <xref ref-type="fig" rid="fig1">Figure 1B</xref>), which has four confirmed cases of scrambled genes (<xref ref-type="bibr" rid="bib54">Katz and Kovner, 2010</xref>; <xref ref-type="bibr" rid="bib39">Gao et al., 2014</xref>). Transcriptome-based surveys offer less precise estimates and cannot distinguish RNA splicing. Several computational pipelines have been developed to facilitate the inference of genome rearrangement features by split-read mapping in the absence of complete MIC or MAC reference genomes (<xref ref-type="bibr" rid="bib26">Denby Wilkes et al., 2016</xref>; <xref ref-type="bibr" rid="bib112">Zheng et al., 2020</xref>; <xref ref-type="bibr" rid="bib33">Feng et al., 2020</xref>; <xref ref-type="bibr" rid="bib89">Seah et al., 2021</xref>). By surveying lighter genome coverage prior to full sequencing, these tools provide partial insight into germline architecture. This helps guide selection of species for full genome sequencing and subsequent construction of complete rearrangement maps between the MIC and MAC genomes. High-quality MIC genome reference assemblies are only currently available for three ciliate genera: <italic>Oxytricha</italic> (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>), <italic>Tetrahymena</italic> (<xref ref-type="bibr" rid="bib48">Hamilton et al., 2016</xref>), and <italic>Paramecium</italic> (<xref ref-type="bibr" rid="bib44">Guérin et al., 2017</xref>; <xref ref-type="bibr" rid="bib90">Sellis et al., 2021</xref>).</p><p>Programmed genome rearrangements in <italic>Oxytricha</italic> exhibit the highest accuracy and largest scale of any known natural gene-editing system, with exquisite control over hundreds of thousands of precise DNA cleavage/joining events. Accordingly, its germline genome structure is arguably the most complex of any model organism (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>), requiring programmed deletion of over 90% of the germline DNA during development and massive descrambling of the resulting fragments to construct a new MAC genome of over 18,000 chromosomes (<xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>). This differs from the distantly related <italic>Tetrahymena</italic> and <italic>Paramecium</italic> that both eliminate ~30% of the germline genome (<xref ref-type="bibr" rid="bib48">Hamilton et al., 2016</xref>; <xref ref-type="bibr" rid="bib44">Guérin et al., 2017</xref>). <italic>Paramecium</italic> uses exclusively 2 bp pointers and lacks evidence of any scrambled loci. A small number of scrambled loci (4 confirmed out of 2711 candidates) have been reported in <italic>Tetrahymena</italic> (<xref ref-type="bibr" rid="bib91">Sheng et al., 2020</xref>, <xref ref-type="fig" rid="fig1">Figure 1B</xref>). <italic>Tetrahymena and Paramecium</italic> diverged from <italic>Oxytricha</italic> over 1 billion years ago (<xref ref-type="bibr" rid="bib82">Parfrey et al., 2011</xref>; <xref ref-type="bibr" rid="bib11">Bracht et al., 2013</xref>), which leaves a large gap in our understanding of the emergence of complex DNA rearrangements in the <italic>Oxytricha</italic> lineage.</p><p>Open questions include how did the <italic>Oxytricha</italic> germline genome acquire its high number of IES insertions and how do scrambled loci arise and evolve. Three previous studies tackled these questions at the level of single genes and orthologs, including DNA polymerase α, actin I, and Telomere end-binding protein subunit α (<xref ref-type="bibr" rid="bib50">Hogan et al., 2001</xref>; <xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>; <xref ref-type="bibr" rid="bib110">Wong and Landweber, 2006</xref>; <xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>). Here, we provide the first comparative genomic analysis of <italic>Oxytricha trifallax</italic> and two other spirotrichous ciliates, <italic>Tetmemena sp</italic>. and <italic>Euplotes woodruffi. Tetmemena sp</italic>. is a hypotrich similar to <italic>Tetmemena pustulata</italic>, formerly <italic>Stylonychia pustulata</italic> (<xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>)<italic>,</italic> in the same family as <italic>O. trifallax</italic> (<xref ref-type="fig" rid="fig1">Figure 1B</xref>; <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>). Hypotrichs are noted for the presence of scrambled genes, based on previous ortholog comparisons (<xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>; <xref ref-type="bibr" rid="bib50">Hogan et al., 2001</xref>; <xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>; <xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>; <xref ref-type="fig" rid="fig1">Figure 1B</xref>). <italic>E. woodruffi,</italic> together with the hypotrichous ciliates, belong to the class Spirotrichea (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). Like hypotrichs, <italic>Euplotes</italic> also has gene-sized nanochromosomes in the MAC genome (<xref ref-type="bibr" rid="bib108">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="bib23">Chen et al., 2019</xref>; <xref ref-type="bibr" rid="bib24">Chen et al., 2021</xref>), but this outgroup uses a different genetic code (UGA is reassigned to cysteine, <xref ref-type="bibr" rid="bib72">Meyer et al., 1991</xref>), and little is known about its MIC genome. A partial MIC genome of <italic>Euplotes vannus</italic> was previously assembled, and it contains highly conserved TA pointers (<xref ref-type="bibr" rid="bib23">Chen et al., 2019</xref>), consistent with previous observations in <italic>Euplotes crassus</italic> (<xref ref-type="bibr" rid="bib56">Klobutcher and Herrick, 1995</xref>). This differs from <italic>O. trifallax</italic>, which uses longer pointers of varying lengths, with scrambled pointers typically longer than nonscrambled ones (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>, <xref ref-type="fig" rid="fig1">Figure 1B</xref>). This observation suggests that longer pointers may supply more information to facilitate MDS descrambling, sometimes over great distances. Therefore, the preponderance of 2 bp pointers in the other <italic>Euplotes</italic> species could indicate limited capacity to support scrambled genes, and a partial genome survey of <italic>E. vannus</italic> concluded that at least 97% of loci are nonscrambled (<xref ref-type="bibr" rid="bib23">Chen et al., 2019</xref>). Early studies of <italic>Euplotes octocarinatus</italic>, on the other hand, demonstrated its use of longer pointers (that usually contain TA) (<xref ref-type="bibr" rid="bib103">Tan et al., 1999</xref>; <xref ref-type="bibr" rid="bib107">Wang et al., 2005</xref>), suggesting that some members of the <italic>Euplotes</italic> genus may have the capacity to support complex genome reorganization. To investigate the origin of scrambled genomes, we choose <italic>E. woodruffi</italic> as an outgroup<italic>,</italic> because it is closely related to <italic>E. octocarinatus</italic> (<xref ref-type="bibr" rid="bib102">Syberg-Olsen et al., 2016</xref>) and feasible to culture in the lab.</p><p>This study includes the de novo assemblies of the micronuclear genome of <italic>Tetmemena sp</italic>. and both genomes of <italic>E. woodruffi</italic>. The availability of MIC and MAC genomes for both species allows us to annotate and compare their genome rearrangement maps and other key features to each other and to <italic>O. trifallax</italic>. The MIC genome of <italic>Tetmemena</italic> is extremely interrupted, like <italic>Oxytricha</italic>. While the <italic>E. woodruffi</italic> MIC genome is much more IES-sparse, it contains thousands of scrambled genes, whose architecture we compare to orthologous loci in the other species. We infer that the evolutionary origin of scrambled genes is associated with local duplications, providing strong support for a previously proposed simple evolutionary model requiring only duplication and decay (<xref ref-type="bibr" rid="bib40">Gao et al., 2015</xref>) that allows for the evolutionary expansion of extremely rearranged chromosome architectures.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Germline genome expansion via repetitive elements</title><p><italic>Tetmemena sp</italic>. and <italic>E. woodruffi</italic> were both propagated in laboratory culture from single cells. The <italic>E. woodruffi</italic> MAC genome was sequenced and assembled from paired-end Illumina reads from whole cell DNA, which is mostly MAC-derived. For comparative analysis, the MAC genome of <italic>E. woodruffi</italic> was assembled using the same pipeline previously used for <italic>Tetmemena sp</italic>. (<xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>). Because MIC DNA is significantly more sparse than MAC DNA in individual cells (<xref ref-type="bibr" rid="bib83">Prescott, 1994</xref>), MIC DNA was enriched before sequencing (see Methods); however, this leads to much lower sequence coverage of the MIC than the MAC. Third-generation long reads (Pacific Biosciences and Oxford Nanopore Technologies) were combined with Illumina paired-end reads (Methods, see genome coverage in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>) to construct hybrid genome assemblies for <italic>Tetmemena sp</italic>. and <italic>E. woodruffi</italic>. Though the final genome assemblies are still fragmented, often due to transposon or other repetitive insertions at boundaries (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>), the current draft assemblies cover most (&gt;90%) MDSs for 89.1% of MAC nanochromosomes in <italic>Tetmemena</italic>, and for 90.0% of MAC nanochromosomes in <italic>E. woodruffi</italic>. This allowed us to establish near-complete rearrangement maps for the newly assembled genomes of <italic>Tetmemena</italic> and <italic>E. woodruffi,</italic> at a level comparable to the published reference for <italic>O. trifallax</italic> (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>), which is appropriate for comparative analysis.</p><p><xref ref-type="table" rid="table1">Table 1</xref> shows a comparison of genome features for the three species. The three MAC genomes are similar in size, with most nanochromosomes bearing only one gene. The size distributions of MAC chromosomes are similar for the three species, though slightly shorter for <italic>E. woodruffi</italic>, consistent with prior observation via gel electrophoresis (<xref ref-type="bibr" rid="bib83">Prescott, 1994</xref>, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). Like <italic>O. trifallax</italic> (<xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>), the maximum number of genes encoded on one chromosome is 7–8 (<xref ref-type="table" rid="table1">Table 1</xref>). Surprisingly, the MIC genome sizes differ substantially: the <italic>Tetmemena</italic> MIC genome assembly is 237 Mbp, nearly half that of <italic>Oxytricha</italic>. The <italic>E. woodruffi</italic> MIC genome assembly is even smaller, approximately 172 Mbp (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Statistics of somatic macronucleus (MAC) and germline micronucleus (MIC) genomes in three species.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom" colspan="2"><italic>Oxytricha trifallax</italic></th><th align="left" valign="bottom" colspan="2"><italic>Tetmemena sp</italic>.</th><th align="left" valign="bottom" colspan="2"><italic>Euplotes woodruffi</italic></th></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top">MAC<sup>a,<xref ref-type="table-fn" rid="table1fn2">*</xref></sup></td><td align="left" valign="top">MIC<sup>b</sup></td><td align="left" valign="top">MAC<sup>c</sup></td><td align="left" valign="top">MIC<sup><xref ref-type="table-fn" rid="table1fn3">†</xref></sup></td><td align="left" valign="top">MAC<xref ref-type="table-fn" rid="table1fn3"><sup>†</sup></xref></td><td align="left" valign="top">MIC<xref ref-type="table-fn" rid="table1fn3"><sup>†</sup></xref></td></tr><tr><td align="left" valign="bottom">Genome size (Mbp)</td><td align="char" char="." valign="bottom">67.1</td><td align="char" char="." valign="bottom">496</td><td align="char" char="." valign="bottom">60.6</td><td align="char" char="." valign="bottom">237</td><td align="char" char="." valign="bottom">72.2</td><td align="char" char="." valign="bottom">172</td></tr><tr><td align="left" valign="bottom">N50 (bp)</td><td align="char" char="." valign="bottom">3745</td><td align="char" char="." valign="bottom">27,807</td><td align="char" char="." valign="bottom">3339</td><td align="char" char="." valign="bottom">14,722</td><td align="char" char="." valign="bottom">2702</td><td align="char" char="." valign="bottom">44,656</td></tr><tr><td align="left" valign="bottom">GC%</td><td align="char" char="." valign="bottom">31.36</td><td align="char" char="." valign="bottom">28.44</td><td align="char" char="." valign="bottom">37.05</td><td align="char" char="." valign="bottom">32.17</td><td align="char" char="." valign="bottom">36.56</td><td align="char" char="." valign="bottom">35.31</td></tr><tr><td align="left" valign="bottom">Number of contigs<xref ref-type="table-fn" rid="table1fn4"><sup>‡</sup></xref></td><td align="char" char="." valign="bottom">22,426</td><td align="char" char="." valign="bottom">25,720</td><td align="char" char="." valign="bottom">25,206</td><td align="char" char="." valign="bottom">28,446</td><td align="char" char="." valign="bottom">35,099</td><td align="char" char="." valign="bottom">17,655</td></tr><tr><td align="left" valign="bottom">Two-telomere contigs</td><td align="char" char="." valign="bottom">14,225</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">15,802</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">19,061</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">Telomeric contigs</td><td align="char" char="." valign="bottom">20,336</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">21,165</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">28,294</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">Single-gene telomeric contigs</td><td align="char" char="." valign="bottom">76.1%</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">75.5%</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">68.5%</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">Maximum number of genes on a telomeric contig</td><td align="char" char="." valign="bottom">8</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">7</td><td align="left" valign="bottom">-</td><td align="char" char="." valign="bottom">8</td><td align="left" valign="bottom">-</td></tr></tbody></table><table-wrap-foot><fn><p>a - <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>; b - <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; c - <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>.</p></fn><fn id="table1fn2"><label>*</label><p>This study used the MAC genome of <italic>Oxytricha</italic> from <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref> instead of the long-read assembly in <xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>, because the short MAC genomes in the present study were primarily assembled from Illumina reads, as in <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>. <xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref> updated <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref> by including nanochromosomes captured in single long reads, which are currently not available for the other two species. The MIC genomes of <italic>Tetmemena</italic> and <italic>E. woodruffi</italic> were assembled to a similar N50 as the reference <italic>O. trifallax</italic> genome (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>) for comparative analysis.</p></fn><fn id="table1fn3"><label>†</label><p>Data from this study.</p></fn><fn id="table1fn4"><label>‡</label><p>Telomere-bearing element (TBE) transposon contaminants in MAC contigs were removed (Methods). Therefore, 24 <italic>Oxytricha</italic> MAC contigs and 13 <italic>Tetmemena</italic> MAC contigs were removed from the published versions.</p></fn></table-wrap-foot></table-wrap><p>The expansion of repetitive elements in the <italic>Oxytricha</italic> lineage may contribute to the difference in MIC genome sizes (<xref ref-type="fig" rid="fig2">Figure 2A–C</xref>). <italic>Oxytricha</italic> has a variety of tranposable elements (TEs) in the MIC, with telomere-bearing elements (TBEs) of the Tc1/<italic>mariner</italic> family the most abundant (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>, <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). A complete TBE transposon contains three open reading frames (ORFs). ORF1 encodes a 42kD transposase with a DDE-catalytic motif. Though present only in the germline, TBEs are so abundant in hypotrichs that some were partially recovered and assembled from whole cell DNA (<xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). The <italic>Oxytricha</italic> MIC genome contains ~10,000 complete TBEs and ~24,000 partial TBEs, which occupy approximately 15.20% (75 Mbp) of the genome (<xref ref-type="fig" rid="fig2">Figure 2A</xref>, <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>; <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). <italic>Tetmemena</italic>, on the other hand, has many fewer TBE ORFs and only 48 complete TBEs (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>), comprising 1.83% (4.3 Mbp) of its MIC genome (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). <italic>E. crassus</italic> has also been reported to have an abundant transposon family called Tec elements (<underline>T</underline>ransposon of <italic><underline>E</underline>uplotes <underline>c</underline>rassus</italic>). Like TBEs, each Tec consists of three ORFs, and ORF1 also encodes a transposase from the Tc1/<italic>mariner</italic> family (<xref ref-type="bibr" rid="bib5">Baird et al., 1989</xref>; <xref ref-type="bibr" rid="bib59">Krikau and Jahn, 1991</xref>; <xref ref-type="bibr" rid="bib53">Jahn et al., 1993</xref>; <xref ref-type="bibr" rid="bib52">Jahn et al., 1989</xref>; <xref ref-type="bibr" rid="bib57">Klobutcher and Herrick, 1997</xref>). The ~57 kD ORF2 encodes a tyrosine-type recombinase (<xref ref-type="bibr" rid="bib27">Doak et al., 2003</xref>), and the 20kD ORF3 has unknown function (<xref ref-type="bibr" rid="bib53">Jahn et al., 1993</xref>). Using the three ORFs of Tec1 and Tec2 as queries for search, we identified 74 complete Tec elements in <italic>E. woodruffi</italic>. Collectively, Tec ORFs occupy 3.6 Mbp, corresponding to only 2.1% of the MIC genome (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). Notably, the transposase-encoding ORF1 is more abundant than the other two TBE/Tec ORFs in all three ciliates (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>), consistent with its proposed role in DNA cleavage during genome rearrangement in <italic>Oxytricha</italic> (<xref ref-type="bibr" rid="bib80">Nowacki et al., 2009</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>The three germline micronucleus (MIC) genomes differ in repeat content, especially transposable elements.</title><p>(<bold>A–C</bold>) MIC genome categories for (<bold>A</bold>) <italic>Oxytricha trifallax</italic>, (<bold>B</bold>) <italic>Tetmemena sp</italic>., and (<bold>C</bold>) <italic>Euplotes woodruffi. Oxytricha</italic> displays the greatest proportion of repetitive elements (telomere-bearing elements [TBE], other repeats, and tandem repeats) relative to the other species. <italic>Oxytricha</italic> MIC-specific genes were annotated in <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib73">Miller et al., 2021</xref>. (<bold>D–F</bold>) Phylogenetic analysis of the three TBE open reading frames (ORFs) in <italic>Oxytricha</italic> and <italic>Tetmemena</italic>: (<bold>D</bold>) 42 kD, (<bold>E</bold>) 22 kD, and (<bold>F</bold>) 57 kD, suggest that TBE3 (green) is the ancestral transposon family in <italic>Oxytricha</italic>. For each ORF, 30 protein sequences from each species were randomly subsampled and maximum likelihood trees constructed using PhyML (<xref ref-type="bibr" rid="bib45">Guindon et al., 2010</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Comparison of germline micronuclear (MIC) genome context of (<bold>A</bold>) Telomere-Bearing Element (TBE) and Transposon of <italic>Euplotes crassus</italic> (Tec) transposons and (<bold>B</bold>) other transposable elements in the three species.</title><p>Complete and partial TBE/Tec elements were annotated by MIC context. Other transposable elements include all subcategories shown in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>. Boundary (light blue): edges of assembled MIC contigs. MIC-specific contig (orange): no macronuclear destined sequence (MDS) identified on the MIC contig so it cannot be annotated as intergenic or a long internally eliminated sequence (IES). Intergenic (green): MIC regions between MDSs for different MAC contigs. IES paralogous (yellow): transposable element (TE) insertions between duplicate (paralogous) MDSs, so they are neither scrambled nor nonscrambled. IES nonscrambled (dark blue): TE insertions that map between consecutive, nonscrambled MDSs for the same MAC contigs. IES scrambled (magenta): MIC regions between nonconsecutive (scrambled) MDSs for the same MAC contig. Note that TEs in IESs or intergenic regions could be flanked by other MIC-limited sequences extending beyond the TE ends.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Length distribution of assembled somatic macronuclear (MAC) nanochromosomes in the three species.</title><p>Chromosomes over 11 kb are excluded from the plot.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig2-figsupp2-v2.tif"/></fig></fig-group><p><italic>Oxytricha</italic> contains three families of TBEs. TBE3 appears to be the most ancient among hypotrichs, based on previous analysis of limited MIC genome data (<xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). We constructed phylogenetic trees using randomly subsampled TBE sequences for all three ORFs from <italic>Oxytricha</italic> and <italic>Tetmemena</italic> (<xref ref-type="fig" rid="fig2">Figure 2D–F</xref>). This confirmed that only TBE3 is present in the <italic>Tetmemena</italic> MIC genome, as proposed in <xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>. This also suggests that TBE1 and TBE2 expanded in <italic>Oxytricha</italic> after its divergence from other hypotrichous ciliates. As illustrated in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>, the MIC genome contexts of TBEs in <italic>Oxytricha</italic> and <italic>Tetmemena</italic> are similar, with many TE insertions within IESs, consistent with either IESs as hotspots for TE insertion or with the model (<xref ref-type="bibr" rid="bib57">Klobutcher and Herrick, 1997</xref>) that some TE insertions may have generated IESs, as demonstrated in <italic>Paramecium</italic> (<xref ref-type="bibr" rid="bib90">Sellis et al., 2021</xref>; <xref ref-type="bibr" rid="bib34">Feng and Landweber, 2021</xref>). Subsequent sequence evolution at the edges of IES/MDS pointers (<xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>) can give rise to boundaries that no longer correspond precisely to TBE ends. For further discussion of the conservation of TBE locations, see the section, ‘<italic>Oxytricha</italic> and <italic>Tetmemena</italic> share conserved rearrangement junctions’ below.</p><p>Additionally, Repeatmodeler/Repeatmasker identified that <italic>Oxytricha</italic> has more MIC repeats in the ‘Other’ category than <italic>Tetmemena</italic> or <italic>E. woodruffi</italic> (<xref ref-type="fig" rid="fig2">Figure 2</xref>, subcategories of repeat content in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). 214 Mbp of the <italic>Oxytricha</italic> MIC genome (43%, which is greater than 35.9% reported in <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref> that used earlier versions of the software) is considered repetitive (including TBEs, tandem repeats, and other repeats in <xref ref-type="fig" rid="fig2">Figure 2</xref>), versus 31.7 Mbp for <italic>Tetmemena</italic> (13.4%) and 28.5 Mbp (16.8%) for <italic>E. woodruffi. Oxytricha</italic>’s additional ~180 Mbp in repeat content partially explains the significantly larger MIC genome size of <italic>Oxytricha</italic> versus the other spirotrich ciliates.</p></sec><sec id="s2-2"><title>The <italic>E. woodruffi</italic> genome has fewer IESs</title><p>We used the genome rearrangement annotation tool, Scrambled DNA Rearrangement Annotation Protocol (SDRAP, <xref ref-type="bibr" rid="bib13">Braun et al., 2022</xref>) to annotate the MIC genomes of <italic>Oxytricha, Tetmemena,</italic> and <italic>E. woodruffi</italic> (Methods). Consistent with their close genetic distance, the genomes of <italic>O. trifallax</italic> and <italic>Tetmemena</italic> have similarly high levels of discontinuity (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). We annotated over 215,299 MDSs in <italic>Oxytricha</italic> and over 215,624 in <italic>Tetmemena</italic> with similar MDS length distributions (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). By contrast, <italic>E. woodruffi</italic> MDSs are typically longer, which indicates a less interrupted genome (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). We compared the number of MDSs between single-copy orthologs for single-gene MAC chromosomes across the three species and found that the orthologs have similar coding sequence (CDS) lengths (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A–B</xref>). There is a strong positive correlation between number of MDSs for orthologous genes in <italic>Oxytricha</italic> and <italic>Tetmemena</italic> (R<sup>2</sup>=0.75, <xref ref-type="fig" rid="fig3">Figure 3B</xref>). There is no correlation among number of MDSs between orthologs of <italic>E. woodruffi</italic> and <italic>Oxytricha</italic> (R<sup>2</sup>=0.003, <xref ref-type="fig" rid="fig3">Figure 3C</xref>), since <italic>E. woodruffi</italic> orthologs typically contain fewer MDSs.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>The three germline micronucleus genomes are interrupted by internally eliminated sequences (IESs) at different levels.</title><p>(<bold>A</bold>) Macronuclear destined sequences (MDSs) of <italic>Euplotes woodruffi</italic> are longer compared to <italic>Oxytricha</italic> or <italic>Tetmemena</italic>. (<bold>B</bold>) Positive correlation between the numbers of MDSs for orthologous genes in <italic>Tetmemena</italic> and in <italic>Oxytricha</italic> for 903 single-gene orthologs. Black line is the function of linear regression (R<sup>2</sup>=0.75). Red line is y=x. (<bold>C</bold>) Orthologs in <italic>E. woodruffi</italic> have fewer MDSs compared to <italic>Oxytricha</italic>, with no correlation (R<sup>2</sup>=0.003). Note that many highly discontinuous genes in <italic>Oxytricha</italic> are IES-less in <italic>E. woodruffi</italic> (present on one MDS). 917 single-gene orthologs are shown. (<bold>D</bold>) Distribution of pointers on single-gene somatic macronucleus (MAC) chromosomes in <italic>Oxytricha vs</italic>. (<bold>E</bold>) <italic>E. woodruffi</italic>, with MAC chromosomes oriented in gene direction. Pointers significantly accumulate at the 5’ end of single-gene MAC chromosomes in <italic>E. woodruffi</italic>. (<bold>F</bold>) Pointer positions on 3684 two-MDS MAC chromosomes demonstrate a preference upstream of the start codon.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Lengths of orthologs in <italic>Oxytricha, Tetmemena</italic> and <italic>Euplotes woodruffi,</italic> and the distribution of pointers on <italic>Tetmemena</italic> chromosomes.</title><p>(<bold>A and B</bold>) Coding sequence (CDS) lengths correlate for <italic>Oxytricha</italic>, <italic>Tetmemena,</italic> and <italic>E</italic><italic>. woodruffi</italic> orthologs (related to <xref ref-type="fig" rid="fig3">Figure 3</xref>). (<bold>A</bold>) <italic>Tetmemena</italic> CDS length positively correlates with that of <italic>Oxytricha</italic> orthologs (R<sup>2</sup>=0.96). Black line is the linear regression fitting function. Red line shows y=x. (<bold>B</bold>) <italic>E. woodruffi</italic> CDS length positively correlates with that of <italic>Oxytricha</italic> orthologs (R<sup>2</sup>=0.83). (<bold>C</bold>) The distribution of pointers on single-gene somatic macronucleus (MAC) chromosomes in <italic>Tetmemena</italic> displays a weak 5’ bias (related to <xref ref-type="fig" rid="fig3">Figure 3</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Scrambled and nonscrambled loci have distinct length distributions of internally eliminated sequences (IESs) and pointers.</title><p>(<bold>A–C</bold>) Length distribution of scrambled and nonscrambled pointers ≤30 bp in (<bold>A</bold>) <italic>Oxytricha</italic>, (<bold>B</bold>) <italic>Tetmemena,</italic> and (<bold>C</bold>) <italic>Euplotes woodruffi</italic>. (<bold>D–F</bold>) Length distribution of scrambled and nonscrambled IESs in (<bold>D</bold>) <italic>Oxytricha</italic> (≤100 bp), (<bold>E</bold>) <italic>Tetmemena</italic> (≤100 bp), and (F) <italic>E. woodruffi</italic> (≤300 bp).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig3-figsupp2-v2.tif"/></fig></fig-group><p>The <italic>E. woodruffi</italic> genome is generally much less interrupted than that of <italic>Oxytricha</italic> or <italic>Tetmemena</italic>. 39.9% of MAC nanochromosomes in <italic>E. woodruffi</italic> lack IESs (IES-less nanochromosomes) compared to only 4.1 and 4.4% in <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, respectively. The sparse IES distribution (as measured by plotting pointer distributions) in <italic>E. woodruffi</italic> displays a curious 5’ end bias on single-gene MAC chromosomes, oriented in gene direction (<xref ref-type="fig" rid="fig3">Figure 3E</xref>). A weak 5’ bias is also present in <italic>Oxytricha</italic> (<xref ref-type="fig" rid="fig3">Figure 3D</xref>) and <italic>Tetmemena</italic> (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1C</xref>). In addition, <italic>E. woodruffi</italic> IESs preferentially accumulate in the 5’ UTR, a short distance upstream of start codons (<xref ref-type="fig" rid="fig3">Figure 3F</xref>). Notably, the median distance between the 5’ telomere addition site and the start codon in <italic>E. woodruffi</italic> is just 54 bp for single-gene chromosomes, approximately half that of <italic>Oxytricha</italic> (<xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref>).</p></sec><sec id="s2-3"><title><italic>E. woodruffi</italic> has an intermediate level of genome scrambling</title><p>Scrambled genome rearrangements exist in all three species, which we report here for the first time in <italic>Tetmemena</italic> and the early diverged <italic>E. woodruffi</italic>. Previous studies have described scrambled genes with confirmed MIC-MAC rearrangement maps for a limited species of hypotrichs (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>; <xref ref-type="bibr" rid="bib50">Hogan et al., 2001</xref>; <xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>; <xref ref-type="bibr" rid="bib110">Wong and Landweber, 2006</xref>; <xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>) and <italic>Chilodonella</italic> (<xref ref-type="bibr" rid="bib54">Katz and Kovner, 2010</xref>; <xref ref-type="bibr" rid="bib39">Gao et al., 2014</xref>) but not in <italic>Euplotes</italic>. Consistent with the phylogenetic placement of <italic>Euplotes</italic> as an earlier diverged outgroup to hypotrichs (<xref ref-type="bibr" rid="bib67">Lynn, 2008</xref>; <xref ref-type="bibr" rid="bib41">Gao et al., 2016</xref>), the <italic>E. woodruffi</italic> genome is scrambled, but it contains approximately half as many scrambled genes (2429 genes encoded on 1913 chromosomes, or 7.3% of genes), versus 15.6% scrambled in <italic>O. trifallax</italic> (3613 genes encoded on 2852 chromosomes) and 13.6% in <italic>Tetmemena</italic> (3371 genes encoded on 2556 chromosomes). The <italic>E. woodruffi</italic> lineage may therefore reflect an evolutionary intermediate stage between ancestral genomes with only modest levels of genome scrambling and the more massively scrambled genomes of hypotrichs.</p><p>We infer that many genes were likely scrambled in the last common ancestor of <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, because these two species share approximately half of their scrambled genes (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). Furthermore, most scrambled genes are not new genes, since they possess at least one ortholog in other ciliate species (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>, <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>).</p></sec><sec id="s2-4"><title>Scrambled genes are associated with local paralogy</title><p>Notably, scrambled genes in all three species generally have more paralogs (<xref ref-type="fig" rid="fig4">Figure 4</xref>). We identified orthogroups containing genes derived from the same gene in the last common ancestor of the three species (Methods). For each species, orthogroups with at least one scrambled gene are significantly larger than those containing no scrambled genes (p-value &lt;1e−5, Mann-Whitney U test, <xref ref-type="fig" rid="fig4">Figure 4A–C</xref>). This association suggests a possible role of gene duplication in the origin of scrambled genes.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Scrambled genes have more paralogs than nonscrambled genes in the three species.</title><p>Orthogroups containing at least one scrambled gene (‘scrambled’) are larger than orthogroups that lack scrambled genes (‘nonscrambled’) in (<bold>A</bold>) <italic>Oxytricha</italic>, (<bold>B</bold>) <italic>Tetmemena,</italic> and (<bold>C</bold>) <italic>Euplotes woodruffi</italic>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>An example of a <italic>Euplotes</italic> <italic>woodruffi</italic> scrambled gene locus containing paralogous macronuclear destined sequences (MDSs).</title><p>(<bold>A</bold>) The upper panel is the map of a scrambled germline micronucleus (MIC) locus (EUPWOO_MIC_17325). Below is the corresponding map of the somatic macronucleus (MAC) chromosome (EUPWOO_MAC_29939). Pointers between MDSs are labeled above or below the MAC contig (nonscrambled pointer length in blue and scrambled pointers labeled in red). (<bold>B</bold>) A model for the evolutionary origin of this scrambled MIC locus by partial duplication and subsequent decay. Stage 1: The ancestral MIC locus contains three nonscrambled MDSs (labeled proto-MDSs because they are precursors for the modern state). Stage 2: The region containing two proto-MDSs duplicated in the MIC genome. Stage 3: Nucleotide substitutions accumulated in both paralogous copies at different positions (shown in gray dashed boxes) leading to the fixation of some regions as MDSs, while the regions that accumulated more mutations decayed into internally eliminated sequences, which are removed during genome rearrangement. (<bold>B</bold>) Has been adapted from a general model in Figure 3 from <xref ref-type="bibr" rid="bib40">Gao et al., 2015</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>The trend of scrambled loci to contain odd-even patterns may arise from partial duplication followed by mutation accumulation.</title><p>(<bold>A</bold>) A diagram describing a typical scrambled region with an odd-even pattern. We propose that the internally eliminated sequence (IES) (<bold>S1</bold>) between macronuclear destined sequence (MDS) <italic>n</italic> and MDS n+<italic>2</italic> may be ancestrally paralogous to MDS n+<italic>1</italic> (<bold>S2</bold>) which evolved by duplication of MDS n+<italic>1</italic> before it was scrambled. S1 and S2 would therefore be homologous in this model. (<bold>B</bold>) The lengths of modern IES (<bold>S1</bold>) and MDS (<bold>S2</bold>) display a strong positive correlation in <italic>Euplotes woodruffi</italic> (504 pairs). Many data points fall on the y=x (red line). All MDS and IES pairs were only considered if they are on the same germline micronucleus contig, to exclude alleles. (<bold>C</bold>) Character mapping of scrambled loci onto a phylogeny: (1) examples of scrambled loci uniquely present in one species (only showing for <italic>Oxytricha</italic> and <italic>Tetmemena</italic>; most scrambled genes in <italic>E. woodruffi</italic> have no ortholog detectable in the other two species, possibly because the long genetic distance obscured homology, see main text and <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>); (2) scrambled loci shared between <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, but not <italic>E. woodruffi</italic>; and (3) scrambled loci shared in three species. The lengths of IES (<bold>S1</bold>) and MDS (<bold>S2</bold>) in typical odd-even regions display a moderately positive correlation in <italic>Oxytricha</italic> (<bold>D</bold>) and <italic>Tetmemena</italic> (<bold>E</bold>). Newer scrambled loci correlate more strongly. Red line represents y=x. Note that S1 and S2 are flanked by identical pointers, <italic>a</italic> and <italic>b</italic>, in all annotated pairs.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig4-figsupp2-v2.tif"/></fig><fig id="fig4s3" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 3.</label><caption><title>Expression level of scrambled and nonscrambled genes in (<bold>A</bold>) <italic>Oxytricha</italic>, (<bold>B</bold>) <italic>Tetmemena,</italic> and (<bold>C</bold>) <italic>Euplotes</italic> <italic>woodruffi</italic>.</title><p>p-Values of Mann-Whitney U tests are shown in blue. The line in orange shows the median. The box shows the range between the first and third quartiles. The upper whisker represents the third quartile + 1.5 × interquartile range (IQR), and the lower whisker shows the first quartile – 1.5 × IQR. Numbers in brackets indicate genes which have a coefficient of variation of TPM (transcripts per million) less than 1.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig4-figsupp3-v2.tif"/></fig></fig-group><p>Scrambled pointers are generally longer than nonscrambled ones in all three species (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>), consistent with prior observations (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>) and the possibility that longer pointers participate in more complex rearrangements, including recombination between MDSs separated by greater distances (<xref ref-type="bibr" rid="bib60">Landweber et al., 2000</xref>). Scrambled and nonscrambled IESs also differ in their length distribution (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). Curiously, scrambled ‘pointers’ in <italic>E. woodruffi</italic> can be as long as several hundred base pairs (median 48 bp, average 212 bp) unlike the more typical 2–20 bp canonical pointers. These long ‘pointers’ in <italic>E. woodruffi</italic> are more likely partial MDS duplications (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref>). We also identified MDSs that map to two or more paralogous regions within the same MIC contig (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>), therefore representing MDS duplications and not alleles. Such paralogous regions could be alternatively incorporated into the rearranged MAC product. Moreover, we find that, for all three species, there are significantly more scrambled chromosomes than nonscrambled MAC chromosomes that contain at least one paralogous MDS (chi-square test, p-value &lt;1e−10; <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>). An example is shown in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref> (MDS 7 and 7').</p><p>The presence of paralogous MDSs can contribute to the origin of scrambled rearrangements, as proposed in an elegant model by <xref ref-type="bibr" rid="bib40">Gao et al., 2015</xref>; illustrated in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1B</xref>. The model proposes that initial MDS duplications permit alternative use of either MDS copy into the mature MAC chromosome. As mutations accumulate in redundant paralogs, cells that incorporate the least decayed MDS regions into the MAC gene would have both a fitness advantage and a better match to the template RNA (<xref ref-type="bibr" rid="bib79">Nowacki et al., 2008</xref>) that guides rearrangement, thus increasing the likelihood of incorporation into the MAC chromosome. The paralogous regions containing more mutations would gradually decay into IESs, and scrambled pointers eventually be reduced to a shorter length. The extended length ‘pointers’ that we identified in <italic>E. woodruffi</italic> may reflect an intermediate stage in the origin of scrambled genes (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1B</xref>).</p><p>This model may generally explain the abundance and expansion of ‘odd-even’ patterns in ciliate scrambled genes (<xref ref-type="bibr" rid="bib60">Landweber et al., 2000</xref>; <xref ref-type="bibr" rid="bib15">Burns et al., 2016</xref>). As illustrated in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref>, the even- and odd-numbered MDSs for many scrambled genes derive from different MIC genome clusters. The model predicts that the IES between MDS <italic>n</italic>−1 and <italic>n</italic>+1 often derives from ancestral duplication of a region containing MDS <italic>n</italic> (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>). To test this hypothesis explicitly, we extracted from all odd-even scrambled loci in the three species all sets of corresponding MDS/IES pairs that are flanked by identical pointers on both sides, i.e., all pairs of scrambled MDSs and IESs, where the IES between MDS <italic>n−1</italic> and n+<italic>1</italic> is directly exchanged for MDS <italic>n</italic> during DNA rearrangement (S1 and S2 in <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>). To exclude the possibility of alleles confounding this analysis, MDS and IES pairs were only considered if they map to the same MIC contig. In <italic>E. woodruffi</italic>, the lengths of these MDS/IES pairs strongly correlate (Spearman correlation ρ=0.755, p&lt;1e−5, <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2B</xref>). Moreover, many MDS and IES sequence pairs also share sequence similarity, consistent with paralogy: for 248 MDS-IES pairs of similar length, 90.3% share a core sequence with ~97.5% identity across 8–100% of both the IES and MDS length. The lowest end of these observations is also compatible with an alternative model (<xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>) in which direct recombination between IESs and MDSs at short repeats can lead to expansion of odd-even patterns. For <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, the MDS and IES lengths for such MDS/IES pairs also display a weakly-positive correlation (p-values and Spearman correlation ρ shown in <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2D–E</xref>). Remarkably, the odd-even-containing loci that are species-specific, and therefore became scrambled more recently, have the strongest length correlation (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2C–E</xref>) and more pairs that display sequence similarity (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>) relative to older loci (scrambled in two or more species). This result is consistent with an evolutionary process in which mutations accumulate in one copy of the MDS, gradually obscuring its sequence homology and ability to be incorporated as a functional MDS, and eventually its ability to be recognized by the template RNAs that guide DNA rearrangement. This analysis also suggests that most of the odd-even scrambled loci in <italic>E. woodruffi</italic> arose recently, because there is greater sequence similarity between MDSs and the corresponding IESs that they replace. Conversely, we infer that most loci that are scrambled in both <italic>Oxytricha</italic> and <italic>Tetmemena</italic> became scrambled earlier in evolution, since they display weaker sequence similarity between exchanged MDS and IES regions.</p><p>Scrambled and nonscrambled genes display nearly identical expression support (the presence of at least one read in all three replicates) in both <italic>Oxytricha</italic> (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>) and <italic>Tetmemena. E. woodruffi</italic> has slightly more expression support for nonscrambled vs. scrambled genes (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>), which could be explained by more recent acquisition of thousands of scrambled loci in <italic>E. woodruffi</italic>. In some of those cases the nonscrambled paralogs may still contribute the major function. The distribution of expression levels is similar for scrambled vs. nonscrambled genes in all three species, supporting their authenticity (<xref ref-type="fig" rid="fig4s3">Figure 4—figure supplement 3</xref>), although in a Mann-Whitney U test, the average expression level of three replicates is significantly higher in nonscrambled genes for <italic>Oxytricha</italic> and <italic>E. woodruffi</italic>, but not significant for <italic>Tetmemena</italic>.</p></sec><sec id="s2-5"><title><italic>Oxytricha</italic> and <italic>Tetmemena</italic> share conserved DNA rearrangement junctions</title><p>To understand the conservation of genome rearrangement patterns, we developed a pipeline guided by protein sequence alignment to compare pointer positions for orthologous genes between any two species (Methods, <xref ref-type="fig" rid="fig5">Figure 5A</xref>). We compared pointers for 2503 three-species single-copy orthologs. 4448 pointer locations are conserved between <italic>Oxytricha</italic> and <italic>Tetmemena</italic> on 1345 ortholog pairs (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>), representing 38.3% of pointers in these orthologs in <italic>Oxytricha</italic> and 30.9% in <italic>Tetmemena</italic>. For <italic>Oxytricha</italic>/<italic>E. woodruffi</italic> and <italic>Tetmemena</italic>/<italic>E. woodruffi</italic> comparisons, 56 and 58 pointer pairs are conserved, respectively. We also identified 23 pointer locations shared among all three species (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>, <xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="supplementary-material" rid="fig5sdata1">Figure 5—source data 1</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Identification and examples of conserved pointers.</title><p>(<bold>A</bold>) Pipeline for comparison of pointer positions in orthologs. Orthologs are first grouped by OrthoFinder (<xref ref-type="bibr" rid="bib32">Emms and Kelly, 2019</xref>), and protein sequences of single-copy orthologs aligned by Clustal Omega (<xref ref-type="bibr" rid="bib92">Sievers et al., 2011</xref>). Then the protein alignments are reverse translated to coding sequence (CDS) alignments by a modified script of pal2nal (105, Methods). Pointers are annotated on the CDS alignments for comparison between any two orthologs. (<bold>B</bold>) Two examples of pointer conservation across three species. Gray lines represent the alignment of orthologous CDS regions, and boxes show magnified regions containing conserved pointers. The top panel shows a conserved scrambled pointer (<italic>Oxytricha</italic>: Contig889.1.g68; <italic>Tetmemena</italic>: LASU02015390.1.g1; <italic>Euplotes woodruffi</italic>: EUPWOO_MAC_30,105 .g1). The bottom panel shows a conserved nonscrambled pointer (<italic>Oxytricha</italic>: Contig19750.0.g98; <italic>Tetmemena</italic>: LASU02002033.1.g1; <italic>E. woodruffi</italic>: EUPWOO_MAC_31,621 .g1). Pointer sequences are noted, and commas indicate reading frame. Protein domains detected by HMMER (<xref ref-type="bibr" rid="bib37">Finn et al., 2011</xref>) are marked in purple. (<bold>C</bold>) Examples of telomere-bearing element (TBE) insertions in nonscrambled internally eliminated sequences. The upper pair of sequences shows an <italic>Oxytricha</italic> TBE pointer (orange insertion of an incomplete TBE2 transposon containing the 42-kD and 57-kD open reading frames) conserved with a <italic>Tetmemena</italic> non-TBE pointer (<italic>Oxytricha</italic>: Contig736.1.g130; <italic>Tetmemena</italic>: LASU02012221.1.g1). Both species have a TA pointer at this junction. The bottom pair of sequences illustrates a case of nonconserved TBE pointers (<italic>Oxytricha</italic>: Contig17579.0.g71; <italic>Tetmemena</italic>: LASU02007616.1.g1).</p><p><supplementary-material id="fig5sdata1"><label>Figure 5—source data 1.</label><caption><title>Pointers conserved in all three species.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-82979-fig5-data1-v2.xlsx"/></supplementary-material></p><p><supplementary-material id="fig5sdata2"><label>Figure 5—source data 2.</label><caption><title>The telomere-bearing element (TBE) pointers in <italic>Oxytricha</italic> that are conserved with non-TBE pointers in <italic>Tetmemena</italic>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-82979-fig5-data2-v2.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Examples of intron-internally eliminated sequence (IES) conversion across three species.</title><p>(<bold>A</bold>) Four intron positions in <italic>Euplotes woodruffi</italic> (orange boxes in magnified regions) overlap locations of nonscrambled pointers in the orthologous genes in <italic>Oxytricha</italic> and <italic>Tetmemena</italic> (<italic>Oxytricha</italic>: Contig13378.0.g40; <italic>Tetmemena</italic>: LASU02004100.1.g1; <italic>E. woodruffi</italic>: EUPWOO_MAC_08,218 .g1) consistent with a possible trend of some ancestral introns becoming IESs in the hypotrich lineage. Two positions fall within a conserved protein domain of unknown function (DUF3591). (<bold>B</bold>) An orthologous gene with two intron-IES conversions in reciprocal directions (<italic>Oxytricha</italic>: Contig16930.0.g77; <italic>Tetmemena</italic>: LASU02013377.1.g1; <italic>E. woodruffi</italic>: EUPWOO_MAC_15,089 .g1). Colors and annotation as in <xref ref-type="fig" rid="fig5">Figure 5</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig5-figsupp1-v2.tif"/></fig></fig-group><p>To test if these pointer locations are genuinely conserved versus coincidental matching by chance, we performed a Monte Carlo simulation, as also used to study intron conservation (<xref ref-type="bibr" rid="bib86">Rogozin et al., 2003</xref>). We randomly shuffled pointer positions on CDS regions 1000 times and counted the number of conserved pointer pairs expected for each simulation (Methods). Of the 1000 simulations, none exceeded the observed number of conserved pointer pairs between <italic>Oxytricha</italic> and <italic>Tetmemena</italic> (p-value &lt;0.001), suggesting evolutionary conservation of pointer positions (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>). A similar result was obtained for pointers conserved in all three species (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>). However, the numbers of pointer pairs conserved between <italic>Oxytricha</italic>/<italic>E. woodruffi</italic> and <italic>Tetmemena</italic>/<italic>E. woodruffi</italic> is similar to the expectations by chance (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>). The low level of pointer conservation of either hypotrichs with <italic>E. woodruffi</italic> may reflect the smaller number of IESs in <italic>E. woodruffi</italic>; hence, most pointers would have arisen in the hypotrich lineage. Furthermore, <italic>E. woodruffi</italic> is genetically more distant from the two hypotrichs; hence, the accumulation of substitutions would obscure protein sequence homology, which we used to compare pointer locations. For ortholog pairs between <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, scrambled pointers are significantly more conserved than nonscrambled ones (chi-square test, p-value &lt;1e−10, <xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>). We also find that most pointer sequences differ even if the positions are conserved (<xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="supplementary-material" rid="fig5sdata1">Figure 5—source data 1</xref>, <xref ref-type="supplementary-material" rid="supp11">Supplementary file 11</xref>), suggesting that substitutions may accumulate in pointers without substantially altering rearrangement boundaries.</p><p><italic>Oxytricha</italic> and <italic>Tetmemena</italic> both contain a high copy number of TBE transposons (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>; <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). We investigated the level of TBE conservation between these two species. To identify orthologous insertions, we focus on TBE insertions in nonscrambled IESs on single-copy orthologs, which include 1706 <italic>Oxytricha</italic> TBEs inserted in 1296 nonscrambled IESs (multiple TBEs can be inserted into an IES) and 180 <italic>Tetmemena</italic> TBEs inserted into 170 nonscrambled IESs. We refer to the pointer flanking a TBE-containing IES as a <italic>TBE pointer</italic>. No TBE pointer locations are conserved between two species. This suggests that TBEs might invade the genomes of <italic>Oxytricha</italic> and <italic>Tetmemena</italic> independently, or still be actively mobile in the genome. Only 27 <italic>Oxytricha</italic> TBE pointers (containing 36 TBEs) are conserved with non-TBE pointers in <italic>Tetmemena</italic> (<xref ref-type="supplementary-material" rid="fig5sdata2">Figure 5—source data 2</xref>, <xref ref-type="fig" rid="fig5">Figure 5C</xref>). No <italic>Tetmemena</italic> TBE pointer is conserved with an <italic>Oxytricha</italic> non-TBE pointer. This suggests that TBE insertions may preferentially produce new rearrangement junctions instead of inserting into an existing IES.</p></sec><sec id="s2-6"><title>Intron locations sometimes coincide with DNA rearrangement junctions</title><p>Ciliate genomes are generally intron-poor. <italic>Oxytricha</italic> averages 1.7 introns/gene, <italic>Tetmemena</italic> has 1.1, and <italic>E. woodruffi</italic> has 2.2. Among three-species orthologs, intron locations sometimes map near pointer positions (within a 20-bp window, <xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). IESs and introns are both noncoding regions that are removed from mature transcripts, though at different stages. A previous single-gene study observed that an IES in <italic>Paraurostyla</italic> overlaps the position of an intron in <italic>Uroleptus</italic>, <italic>Urostyla,</italic> and also the human homolog (<xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>). This observation suggested an intron-IES conversion model in which the ability to eliminate non-CDS regions as either DNA or RNA provides a potential backup mechanism. Such interconversion has also been observed between two strains of <italic>Stylonychia</italic> (<xref ref-type="bibr" rid="bib76">Möllenbeck et al., 2006</xref>). In the present study, we identified 174 potential cases of intron-IES conversion in the three species (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>, <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>): 103 (59.2%) <italic>E. woodruffi</italic> introns map near <italic>Oxytricha</italic>/<italic>Tetmemena</italic> pointers. We used a 20-bp window for this analysis, since one would only expect the boundaries of introns and IESs to coincide precisely if they were recent evolutionary conversions. A Monte Carlo simulation for these intron-IES comparisons (<xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>) revealed that p&lt;0.001 for most three-species comparisons. For two-species comparisons, we identify 306 cases where an intron boundary in one species precisely coincides with a pointer sequence in another species, with strongest statistical support for the comparison between <italic>Oxytricha</italic> intron positions and <italic>Tetmemena</italic> IES junctions (p<italic>=</italic>0.008) (<xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>). Notably, <italic>Tetmemena</italic> intron locations rarely coincide with <italic>Oxytricha</italic> IESs (<xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>), suggesting a possible bias in the direction of intron-IES conversion during evolution.</p><p>The observation that <italic>E. woodruffi</italic> has the most introns but the smallest number of IESs per gene (<xref ref-type="fig" rid="fig3">Figure 3</xref>) is consistent with removal of intragenic non-CDS regions as either DNA or RNA. The intron-sparseness of ciliates is compatible with a hypothesis that it is advantageous to eliminate noncoding regions earlier at the DNA level, with intron deletion sometimes providing an opportunity for repair if they fail to be excised as IESs (<xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>).</p></sec><sec id="s2-7"><title>Evolution of complex genome rearrangements: Russian doll genes</title><p>Genome rearrangements in the <italic>Oxytricha</italic> lineage can include overlapping and nested loci, with MDSs for different MAC loci embedded in each other (<xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="bib12">Braun et al., 2018</xref>). When multiple gene loci are nested in each other, these have been called Russian doll loci (<xref ref-type="bibr" rid="bib12">Braun et al., 2018</xref>). <italic>Oxytricha</italic> contains two loci with five or more layers of nested genes (<xref ref-type="bibr" rid="bib12">Braun et al., 2018</xref>). <italic>Oxytricha</italic> and <italic>Tetmemena</italic> display a high degree of synteny and conservation in both Russian doll loci. In the first Russian doll gene cluster, one nested gene (green) is present in <italic>Oxytricha</italic> but absent in <italic>Tetmemena</italic> (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>, <xref ref-type="fig" rid="fig6s2">Figure 6—figure supplement 2</xref>), confirmed by PCR (Methods). <italic>Oxytricha</italic> also has a complete TBE3 insertion in the green gene (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1A</xref>), hinting at a possible link between transposition and new gene insertion. In addition, a two-gene chromosome in <italic>Oxytricha</italic> (orange) is present as two single-gene chromosomes in <italic>Tetmemena</italic> (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>). In <italic>Oxytricha</italic>, seven orange MDSs ligate across two other loci via an 18-bp pointer (<named-content content-type="sequence">TATATCTATACTAAACTT</named-content>) to form a two-gene nanochromosome. However, in <italic>Tetmemena</italic>, telomeres are added to the ends of both gene loci instead, forming two independent MAC chromosomes (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>). The second Russian doll locus has an example of a long, conserved pointer (orange dotted line) that bridges three other loci (the green and blue scrambled loci and one nonscrambled locus, <xref ref-type="fig" rid="fig6">Figure 6B</xref>). Close to this region is a decayed TBE insertion (769 bp) in <italic>Oxytricha.</italic> None of the <italic>E. woodruffi</italic> orthologs of both Russian doll loci maps to the same MIC contig, which suggests that the Russian doll clusters arose after the divergence of <italic>Euplotes</italic> from the common ancestor of <italic>Oxytricha</italic> and <italic>Tetmemena.</italic></p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Synteny in ‘Russian doll’ loci in <italic>Oxytricha</italic> and <italic>Tetmemena</italic>.</title><p>(<bold>A</bold>) Schematic comparison of the Russian doll gene cluster on <italic>Oxytricha</italic> germline micronucleus (MIC) contig OXYTRI_MIC_87484 vs. <italic>Tetmemena</italic> MIC contig TMEMEN_MIC_21461. Boxes of the same color represent clusters of macronuclear destined sequences (MDSs) for orthologous genes (detailed map in <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref> and <xref ref-type="fig" rid="fig6s2">Figure 6—figure supplement 2</xref>). Numbers in brackets indicate the number of MDSs in each cluster, grouped by somatic macronucleus (MAC) chromosome. One nested gene (green) in <italic>Oxytricha</italic> is absent from <italic>Tetmemena</italic>. A two-gene chromosome (orange) that derives from seven MDSs in <italic>Oxytricha</italic> is processed as two single-gene chromosomes in <italic>Tetmemena</italic> instead (indicated by black border around orange boxes). The purple gene in <italic>Oxytricha</italic> has two paralogs in <italic>Tetmemena</italic>. Black triangles represent conserved, orthologous, and nonscrambled gene loci inserted between nested Russian doll genes. Empty triangle represents scrambled MDSs for other loci. Gray triangles, complete nonscrambled MAC loci embedded between gene layers in one species with no orthologous gene detected in the other species. Black star, a complete telomere-bearing element (TBE) transposon insertion. Gray star, a partial TBE insertion. (<bold>B</bold>) <italic>Oxytricha</italic> MIC contig OXYTRI_MIC_69233 vs. <italic>Tetmemena</italic> MIC contig TMEMEN_MIC_22886. Pointer sequences bridging the nested MDSs of orange and green genes are highlighted. The underlined pointer portions are conserved between species, e.g., the last 8 bp of the <italic>Oxytricha</italic> pointer, TAAGTT<underline>CAAAGTAG</underline>, is identical to the first 8 bp of <named-content content-type="sequence"><underline>CAAAGTAG</underline>CTCAATC</named-content> in <italic>Tetmemena</italic>, illustrating pointer sliding (<xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>), or gradual shifting of MDS/IES boundaries. White star indicates a decayed TBE with no open reading frame identified.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig6-v2.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Detailed illustration of both Russian doll regions in <xref ref-type="fig" rid="fig6">Figure 6</xref>.</title><p>Macronuclear destined sequence (MDS) indices are annotated here for each somatic macronucleus (MAC) locus. Overlined numbers represent inverted MDSs. MAC contig numbers for the MDSs are listed below and shown in corresponding color patterns (the <italic>Oxytricha</italic> loci were previously characterized in <xref ref-type="bibr" rid="bib12">Braun et al., 2018</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig6-figsupp1-v2.tif"/></fig><fig id="fig6s2" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 2.</label><caption><title>Details of the Russian doll region in <italic>Tetmemena</italic> (TMEMEN_MIC_21461, <xref ref-type="fig" rid="fig6">Figure 6A</xref>).</title><p>The whole region (~50 kb) was validated by 11 PCRs. The two black arrows indicate the absence of a Russian doll gene (green in <xref ref-type="fig" rid="fig6">Figure 6A</xref>) that is present in <italic>Oxytricha</italic>. Legend lists the 20 <italic>Tetmemena</italic> somatic macronucleus contigs that contain the corresponding macronuclear destined sequences.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-fig6-figsupp2-v2.tif"/></fig></fig-group></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>The highly diverse ciliate clade provides a valuable resource for evolutionary studies of genome rearrangement. However, full assembly and annotation of germline MIC genomes have concentrated on the model ciliates <italic>Tetrahymena, Paramecium,</italic> and <italic>Oxytricha</italic>. To provide insight into genome evolution in this lineage, we assembled and compared germline and somatic genomes of <italic>Tetmemena sp</italic>. and an outgroup, <italic>E. woodruffi</italic>, to that of <italic>O. trifallax</italic>. This expands our knowledge of the diversity of ciliate genome structures and the evolutionary origin of complex genome rearrangements.</p><p>Dramatic variation in transposon copy number (TBE and Tec elements) from the Tc1/<italic>mariner</italic> family appears to explain most of the variation in MIC genome size. In many eukaryotic taxa, genome size can differ dramatically even for closely related species, a phenomenon known as the ‘C-value paradox’ (<xref ref-type="bibr" rid="bib104">Thomas, 1971</xref>). Our present observations are compatible with previous reports that the repeat content of the genome, especially transposon content, positively correlates with genome size (<xref ref-type="bibr" rid="bib31">Elliott and Gregory, 2015</xref>).</p><p><italic>Oxytricha</italic> has three TBE families in the MIC genome, but only TBE3 is present in <italic>Tetmemena</italic>, consistent with our previous conclusion that TBE3 is ancestral to the base of the transposon lineage in hypotrichous ciliates (<xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). Tens of thousands of TBE1/2 transposons then expanded specifically in <italic>Oxytricha</italic>. Despite a high copy number of TBEs in both <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, we find no identical TBE locations in nonscrambled IESs, even among syntenic Russian doll regions. These observations suggest that TBEs may be active in these genomes and contribute to the evolution of genome structure.</p><p>In the relatively IES-poor genome of <italic>E. woodruffi</italic>, IESs accumulate upstream of start codons, similar to the 5’ bias of introns in intron-poor organisms (<xref ref-type="bibr" rid="bib78">Mourier and Jeffares, 2003</xref>). The simplest model to explain 5’ intron bias is homologous recombination between a reverse transcript of an intron-lacking mRNA and the original DNA locus to erase introns in the coding region (<xref ref-type="bibr" rid="bib78">Mourier and Jeffares, 2003</xref>). A similar mechanism could simultaneously erase IESs in coding regions via germline recombination between the MIC chromosome and a reverse transcript; however, they are usually in different subcellular locations. More plausibly, a source for DNA recombination could be a MAC nanochromosome, since they are already abundant at high copy number, but another source could be by capture of a reverse transcript of a long non-coding template RNA that guides DNA rearrangement (<xref ref-type="bibr" rid="bib79">Nowacki et al., 2008</xref>; <xref ref-type="bibr" rid="bib64">Lindblad et al., 2017</xref>). Either recombination event in the germline would lead to loss of IESs, while retaining introns, but neither would necessarily provide a bias for IES-loss in coding regions. Any of these infrequent events would be meaningful on an evolutionary time scale, even if developmentally rare. The 5’ bias of IESs could also reflect an evolutionary bias for continuous coding regions. Alternatively, upstream IESs might regulate gene expression or cell growth (<xref ref-type="bibr" rid="bib90">Sellis et al., 2021</xref>), like some introns (<xref ref-type="bibr" rid="bib81">Parenteau et al., 2019</xref>; <xref ref-type="bibr" rid="bib77">Morgan et al., 2019</xref>).</p><p>This study investigated the evolution of scrambled genes by comparing <italic>Oxytricha</italic> and <italic>Tetmemena</italic> to <italic>E. woodruffi</italic>, as an earlier diverged representative of the spirotrich lineage. While <italic>E. woodruffi</italic> has approximately half as many scrambled genes as <italic>Tetmemena</italic> and <italic>Oxytricha</italic>, its genes are also much more continuous. For example, the most scrambled gene in <italic>E. woodruffi,</italic> encoding a DNA replication licensing factor (EUPWOO_MAC_28518, 3 kb), has only 20 scrambled junctions. The most scrambled gene in <italic>Tetmemena</italic> (LASU02015934.1, 14.7 kb, encoding a hydrocephalus-inducing-like protein) has 204 scrambled pointers, and the most scrambled gene in <italic>Oxytricha</italic> (Contig17454.0, 13.7 kb, encoding a dynein heavy chain family protein, <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>) is similarly complex, with 195 scrambled junctions. Together, these observations are consistent with our interpretation that <italic>E. woodruffi</italic> reflects an evolutionary intermediate stage, as it contains both fewer scrambled loci and fewer scrambled junctions within its scrambled loci. The observation that the most scrambled locus differs in each species is also consistent with the conclusion that complex gene architectures may continue to elaborate independently.</p><p>We observed that scrambled genes in each species tend to have more paralogs than nonscrambled genes. Similarly, in <italic>C. uncinata</italic> (<xref ref-type="bibr" rid="bib69">Maurer-Alcalá et al., 2018a</xref>), a distantly related ciliate in the class <italic>Phyllopharyngea</italic> that also has scrambled genes, scrambled gene families (orthogroups) contain more genes (~2.9) than nonscrambled gene families (~1.3) (<xref ref-type="bibr" rid="bib69">Maurer-Alcalá et al., 2018a</xref>). Apart from duplications at the gene level, <italic>E. woodruffi</italic> often contains partial MDS duplications at scrambled junctions, annotated as unusually long ‘pointers’ (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). We also demonstrate that odd-even scrambled patterns could readily arise from local duplications (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>). These observations are most consistent with a simple model (<xref ref-type="bibr" rid="bib40">Gao et al., 2015</xref>) in which local duplications permit combinatorial DNA recombination between paralogous germline regions, and mutation accumulation in either paralogs establishes an odd-even scrambled pattern that can propagate by weaving together segments from paralogous sources. Other proposed models include <xref ref-type="bibr" rid="bib49">Hoffman and Prescott, 1997</xref> IES-invasion model that suggested that pairs of IESs could invade an MDS, and then subsequently recombine with another IES to yield odd-even scrambled regions; however, a previous examination did not find support for this model (<xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>). <xref ref-type="bibr" rid="bib84">Prescott et al., 1998</xref> also proposed that some odd-even scrambled loci could arise suddenly via reciprocal recombination with loops of A/T-rich DNA, but this does not exploit paralogy, only the high A/T content in the MIC. We previously proposed a gradual model (<xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>; <xref ref-type="bibr" rid="bib60">Landweber et al., 2000</xref>) in which MDS/IES recombination at short AT-rich repeats (precursors to pointers) could generate and propagate odd-even scrambled patterns. While limited comparisons of orthologs favored the stepwise recombination models (<xref ref-type="bibr" rid="bib50">Hogan et al., 2001</xref>; <xref ref-type="bibr" rid="bib19">Chang et al., 2005</xref>; <xref ref-type="bibr" rid="bib110">Wong and Landweber, 2006</xref>; <xref ref-type="bibr" rid="bib28">DuBois and Prescott, 1995</xref>), none of the earlier models accounted for the widespread existence of partial paralogy, revealed by genome assemblies.</p><p>Local duplications provide a buffer against mutations, allowing paralogous MDSs to repair the MAC locus during assembly of odd/even scrambled genes. Therefore, once an odd/even scrambled locus is established, a consequence is that evolution can only proceed in the direction of accumulating more scrambled junctions, as each new mutation in one paralog necessitates repair via incorporation of the other paralog (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1B</xref>). This shortens the length of the respective MDSs and increases the number of recombination junctions, creating an evolutionary ratchet that drives the increase in scrambling. The lack of the presence of an error-free, continuous version of this locus in the germline reduces the possibility of losing the scrambled pattern from the MIC genome, relative to the trend toward decreasing MDS lengths as more mutations accumulate in either paralogs, with a resulting increase in the levels of scrambling and fragmentation (<xref ref-type="bibr" rid="bib61">Landweber, 2007</xref>; <xref ref-type="bibr" rid="bib99">Speijer, 2008</xref>). The only opportunity to repair a scrambled locus in the MIC would be a rare event that replaces the locus via recombination with a continuous version from the parental MAC, with the source being either parental MAC DNA or a reverse transcript of a template RNA (<xref ref-type="bibr" rid="bib79">Nowacki et al., 2008</xref>; <xref ref-type="bibr" rid="bib64">Lindblad et al., 2017</xref>), as discussed above.</p><p>Recent exciting reports have also described scrambled genomes in metazoa, including cephalopods (<xref ref-type="bibr" rid="bib88">Schmidbaur et al., 2022</xref>; <xref ref-type="bibr" rid="bib2">Albertin et al., 2022</xref>), but those events entail primarily evolutionary shuffling of gene order, without accompanying genome editing or repair. The ciliate lineage is remarkable in having evolved a sophisticated mechanism of RNA-guided genome editing that allows accurate and precise DNA repair of translocations and inversions. The future opportunity to harness this system to develop novel tools for genome editing outside of <italic>Oxytricha</italic> offers exciting directions.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><sec id="s4-1"><title>DNA collection and sequencing of <italic>Tetmemena sp.</italic></title><p><italic>Tetmemena sp.</italic> (strain SeJ-2015; <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>) was isolated as a single cell from a stock culture and propagated as a clonal strain via vegetative (asexual) cell culture. Cells were cultured in Pringsheim media (0.11 mM Na<sub>2</sub>HPO<sub>4</sub>, 0.08 mM MgSO<sub>4</sub>, 0.85 mM Ca(NO<sub>3</sub>)<sub>2</sub>, 0.35 mM KCl, pH 7.0) and fed with <italic>Chlamydomonas reinhardtii</italic>, together with 0.1%(v/v) of an overnight culture of non-virulent <italic>Klebsiella pneumoniae</italic>. Macronuclei and micronuclei were isolated using sucrose gradient centrifugation (<xref ref-type="bibr" rid="bib63">Lauth et al., 1976</xref>). Genomic DNA was subsequently purified using the Nucleospin Tissue Kit (Takara Bio USA, Inc). Macronuclear DNA was sequenced and assembled in <xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>. Micronuclear DNA was further size-selected via BluePippin (Sage Science) for PacBio sequencing, or via 0.6% (w/v) SeaKem Gold agarose electrophoresis (Lonza) for Illumina sequencing. Micronuclear DNA purification and sequencing protocols are described in <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>.</p></sec><sec id="s4-2"><title>DNA collection and sequencing for <italic>E. woodruffi</italic></title><p><italic>E. woodruffi</italic> (strain Iz01) was cultured in Volvic water at room temperature and fed with green algae every 2–3 days. We fed cells with <italic>C. reinhardtii</italic> for MAC DNA collection, and switched to <italic>Chlorogonium capillatum</italic> for MIC DNA collection. In order to remove algal contamination, cells were starved for at least 2–3 days before collection. Cells were washed and concentrated as in <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>. Because MAC DNA is predominant in whole cell DNA, we used whole cell DNA (purified via NucleoSpin Tissue kit, Takara Bio USA, Inc) for MAC genome sequencing. Paired-end sequencing was performed on an Illumina Hiseq2000 at the Princeton University Genomics Core Facility.</p><p>MIC DNA was enriched from whole cell DNA and sequenced via three sequencing platforms (Illumina, Pacific Biosciences, and Oxford Nanopore Technologies). We used conventional and pulse-field gel electrophoresis (PFGE) to enrich MIC DNA:</p><list list-type="order"><list-item><p>High-molecular-weight DNA was separated from whole cell DNA by gel-electrophoresis (0.25% agarose gel at 4°C, 120 V for 4 hr). The top band was cut from the gel and purified with the QIAGEN QIAquick kit. The purified high-molecular-weight DNA was directly sent to the group of Dr. Robert Sebra at the Icahn School of Medicine at Mount Sinai for library construction and sequencing. BluePippin (Sage Science) separation was used before sequencing to select DNA &gt;10 kb. DNA was sequenced on two platforms: Illumina HiSeq2500 (150 bp paired-end reads) and PacBio Sequel (SMRT reads).</p></list-item><list-item><p>High-molecular-weight DNA was also enriched by PFGE. <italic>E. woodruffi</italic> cells were mixed with 1% low-melt agarose to form plugs according to <xref ref-type="bibr" rid="bib1">Akematsu et al., 2017</xref>, with addition of 1 hr incubation with 50 μg/ml RNase (Invitrogen AM2288) in 10 mM Tris-HCl (pH7.5) at 37°C for RNA depletion. After three washes of 1 hr with 1× TE buffer, the DNA plugs were incubated in 1 mM phenylmethylsulfonyl fluoride (PMSF) to inactivate proteinase K, followed by MspJI (New England Biolabs) digestion at <sup>m</sup>CNNR(9/13) sites to remove contaminant DNA (<sup>m</sup>C indicates C5-methylation or C5- hydroxymethylation). Previous reports have shown that no methylcytosine is detectable in vegetative cells of <italic>Oxytricha</italic> (<xref ref-type="bibr" rid="bib10">Bracht et al., 2012</xref>), <italic>Tetrahymena</italic> (<xref ref-type="bibr" rid="bib42">Gorovsky et al., 1973</xref>), and <italic>Paramecium</italic> (<xref ref-type="bibr" rid="bib25">Cummings et al., 1974</xref>), suggesting that C5-methylation and C5-hydroxymethylation are rarely involved in the vegetative growth of the ciliate lineage. We also validated by qPCR that the quantity of two randomly selected MIC loci is not changed after the MspJI digestion. On the contrary, algal genomic DNA is significantly digested by MspJI. Based on these results, we conclude that MspJI digestion can be used to remove bacterial and algal DNA with C5-methylation and C5-hydroxymethylation, leaving <italic>E. woodruffi</italic> MIC DNA intact. The agarose plugs containing digested DNA were then inserted into wells of 1.0% Certified Megabase agarose gel (Bio-Rad) for PFGE (CHEF-DR II System, Bio-Rad). The DNA was separated at 6 V, 14°C with 0.5× TBE buffer at a 120° angle for 24 hr with switch time of 60–120 s. We validated by qPCR that the <italic>E. woodruffi</italic> MIC chromosomes were not mobilized from the well, while the MAC DNA migrated into the gel. The MIC DNA was then extracted by phenol-chloroform purification. Library preparation and sequencing were performed at Oxford Nanopore Technologies (New York, NY).</p></list-item></list></sec><sec id="s4-3"><title>MAC genome assembly of <italic>E. woodruffi</italic></title><p>We assembled the MAC genome of <italic>E. woodruffi</italic> using the same pipeline for <italic>Tetmemena sp.</italic> (<xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>) for comparative analysis: two draft genomes were assembled by SPAdes (<xref ref-type="bibr" rid="bib6">Bankevich et al., 2012</xref>) and Trinity (<xref ref-type="bibr" rid="bib43">Grabherr et al., 2011</xref>), and were then merged by CAP3 (<xref ref-type="bibr" rid="bib51">Huang and Madan, 1999</xref>). Trinity, which is a software developed for de novo transcriptome assembly (<xref ref-type="bibr" rid="bib43">Grabherr et al., 2011</xref>), has been used to assemble hypotrich MAC genomes (<xref ref-type="bibr" rid="bib21">Chen et al., 2015</xref>) because their nanochromosome genome structure is similar to transcriptomes, including properties such as variable copy number and alternative isoforms (<xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>). Telomeric reads were mapped to contigs by BLAT (<xref ref-type="bibr" rid="bib55">Kent, 2002</xref>), and contigs were further extended and capped by telomeres when at least five reads pile up at a position near ends by custom python scripts (<ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/MAC_genome_telomere_capping">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/MAC_genome_telomere_capping</ext-link>) (<xref ref-type="bibr" rid="bib35">Feng, 2022a</xref>). The mitochondrial DNA was removed if the contig has a TBLASTX (<xref ref-type="bibr" rid="bib16">Camacho et al., 2009</xref>) hit on the <italic>Oxytricha</italic> mitochondrial genome (Genbank accession JN383842.1 and JN383843.1) or two <italic>Euplotes</italic> mitochondrial genomes (<italic>Euplotes minuta</italic> GQ903130.1, <italic>E. crassus</italic> GQ903131.1). Algal contigs were removed by BLASTN to all <italic>C. reinhardtii</italic> nucleotide sequences downloaded from Genbank. Non-telomeric contigs were mapped to bacterial NR by BLASTX to remove bacterial contaminations. The genome was further compressed by CD-HIT (<xref ref-type="bibr" rid="bib38">Fu et al., 2012</xref>) in two steps: (1) contigs &lt;500 bp were removed if 90% of the short contig can be aligned to a contig ≥ 500 bp with 90% similarity (-c 0.9 -aS 0.9 -uS 0.1); (2) then the genome was compressed by 95% similarity (-c 0.95 -aS 0.9 -uS 0.1). Contigs shorter than 500 bp without telomeres were removed. Nine contigs, likely Tec contaminants from the MIC genome, were also excluded (Tblastn, ‘-db_gencode 10 -evalue 1e-5’), and they could be assembled due to the high copy number in the MIC genome (47, 48, Genbank accessions of Tec ORFs are AAA62601.1, AAA62602.1, AAA62603.1, AAA91339.1, AAA91340.1, AAA91341.1, AAA91342.1).</p></sec><sec id="s4-4"><title>RNA sequencing of <italic>E. woodruffi and Tetmemena sp.</italic></title><p>Three biological replicates of total RNA was isolated from asexually growing <italic>E. woodruffi</italic> and <italic>Tetmemena sp.</italic> cells using TRIzol reagent (Thermo Fisher Scientific) and enriched for the poly(A)+fraction using the NEBNext Poly(A) mRNA Magnetic Isolation Module (New England Biolabs). Stranded RNA-seq libraries were constructed using the ScriptSeq v2 RNA-seq library preparation kit (Epicentre) and sequenced on an Illumina Nextseq500 at the Columbia Genome Center. For <italic>E. woodruffi</italic>, the transcriptome was assembled by Trinity (<xref ref-type="bibr" rid="bib43">Grabherr et al., 2011</xref>), and transcript alignments to the MAC genome were generated by PASA (<xref ref-type="bibr" rid="bib46">Haas et al., 2003</xref>).</p></sec><sec id="s4-5"><title>Gene prediction of the <italic>E. woodruffi</italic> MAC genome and validation of MAC genome completeness</title><p>We followed the gene prediction pipeline developed by the Broad institute (<ext-link ext-link-type="uri" xlink:href="https://github.com/PASApipeline/PASApipeline/wiki">https://github.com/PASApipeline/PASApipeline/wiki</ext-link>); using EVidenceModeler (EVM, <xref ref-type="bibr" rid="bib47">Haas et al., 2008</xref>) to generate the final gene predictions. EVM produced gene structures by weighted combination of evidence from three resources: <italic>ab initio</italic> prediction, protein alignments, and transcript alignments (the weight was 3, 3, and 10 respectively). <italic>Ab initio</italic> prediction was generated by BRAKER2 pipeline (<xref ref-type="bibr" rid="bib14">Brůna et al., 2021</xref>). Protein alignments for EVM were generated by mapping <italic>Oxytricha</italic> proteins to the <italic>E. woodruffi</italic> MAC genome by Exonerate (<xref ref-type="bibr" rid="bib94">Slater and Birney, 2005</xref>). EVM predicted 33,379 genes on MAC chromosomes with at least one telomere.</p><p>We assessed MAC genome completeness using three methods: (1) 28,294 (80.6%) of the 35,099 <italic>E. woodruffi</italic> MAC contigs have at least one telomere. (2) In the <italic>E. woodruffi</italic> genes predicted on telomeric contigs, 88.8% of BUSCO (<xref ref-type="bibr" rid="bib93">Simão et al., 2015</xref>; <xref ref-type="bibr" rid="bib68">Manni et al., 2021</xref>) genes in the lineage database alveolata_odb10 were identified as complete. Within the 171 BUSCO genes, 135 are complete and single-copy, 17 are complete and duplicated, 7 are fragmented, and 12 are missing. This represents the best <italic>Euplotes</italic> MAC genome assembly available. (3) We identified 51 tRNA genes encoding all 20 amino acids by tRNAscan-SE (<xref ref-type="bibr" rid="bib66">Lowe and Eddy, 1997</xref>) in the MAC genome, including two suppressor tRNAs of UAA and UAG.</p></sec><sec id="s4-6"><title>MIC genome assembly of <italic>Tetmemena sp.</italic></title><p>The MIC genome of <italic>Tetmemena</italic> was assembled with a hybrid approach to combine reads from different sequencing platforms. <italic>Tetmemena</italic> Illumina reads were first assembled by SPAdes (77, parameters ‘-k 21,33,55,77,99,127 –careful’). PacBio reads were error corrected by FMLRC (<xref ref-type="bibr" rid="bib109">Wang et al., 2018</xref>) using Illumina reads with default parameters. Corrected PacBio reads were aligned to both the MAC genome and the Illumina MIC assembly with BLASTN. Reads were removed if they start or end with telomeres or are aligned better to the MAC. The remaining reads were assembled with wtdbg2 (<xref ref-type="bibr" rid="bib87">Ruan and Li, 2020</xref>, parameters ‘-x rs’). The PacBio assembly was polished by Pilon (<xref ref-type="bibr" rid="bib106">Walker et al., 2014</xref>) with the ‘--diploid’ option. The Illumina and PacBio assemblies were merged by quickmerge (<xref ref-type="bibr" rid="bib18">Chakraborty et al., 2016</xref>) with the ‘-l 5000’ option.</p></sec><sec id="s4-7"><title>MIC genome assembly of <italic>E. woodruffi</italic></title><p>The MIC genome of <italic>E. woodruffi</italic> was assembled using a similar procedure as described above for <italic>Tetmemena. E. woodruffi</italic> reads were filtered to remove bacterial contamination, including abundant high-GC-content contaminants, possibly endosymbionts (<xref ref-type="bibr" rid="bib9">Boscaro et al., 2019</xref>). Nanopore reads with GC content ≥55% were assembled by Flye (<xref ref-type="bibr" rid="bib58">Kolmogorov et al., 2019</xref>) with the parameter ‘--meta’ for metagenomic assembly of bacterial contigs. We used kaiju (<xref ref-type="bibr" rid="bib71">Menzel et al., 2016</xref>) to identify bacteria taxa for these contigs. 9 of 10 top-covered contigs derive from Proteobacteria, from which many <italic>Euplotes</italic> symbionts derive (<xref ref-type="bibr" rid="bib9">Boscaro et al., 2019</xref>). Bacterial contamination was removed from Illumina reads if perfectly mapping to these metagenomic contigs by Bowtie2 (<xref ref-type="bibr" rid="bib62">Langmead and Salzberg, 2012</xref>). The cleaned Illumina reads were then assembled by SPAdes with ‘-k 21,33,55,77,99,127’ (<xref ref-type="bibr" rid="bib6">Bankevich et al., 2012</xref>). Pacbio raw reads and Nanopore raw reads with GC content &lt;55% were aligned to a concatenated database containing both the MAC genome and the Illumina MIC assembly with BLASTN. Reads were removed if they start or end with telomeres or align better to the MAC. Remaining PacBio/Nanopore reads were assembled by Flye with ‘--meta’ mode. The PacBio-Nanopore assembly was polished by Pilon with the ‘--diploid’ option. Illumina and PacBio-Nanopore assemblies were merged by quickmerge with the ‘-l 10000’ option. Contigs shorter than 1 kb were removed.</p></sec><sec id="s4-8"><title>MIC genome decontamination</title><p>The draft MIC genome of <italic>Tetmemena</italic> was first mapped to telomeric MAC contigs by BLASTN. MIC contigs containing MDSs were included in the final assembly. The rest of the MIC contigs were filtered by a decontamination pipeline: (1) contigs were aligned to the <italic>K. pneumoniae</italic> genome, <italic>C. reinhardtii</italic> genome, and the <italic>Oxytricha</italic> mitochondrial genome by BLASTN to remove contaminants; (2) the remaining contigs were then searched against the bacteria NR database and a ciliate protein database (including protein sequences annotated in <italic>Tetrahymena thermophila</italic>: <ext-link ext-link-type="uri" xlink:href="http://www.ciliate.org/system/downloads/tet-latest/4-Protein%20fasta.fasta">http://www.ciliate.org/system/downloads/tet-latest/4-Protein%20fasta.fasta</ext-link>; <italic>Paramecium tetraurelia</italic>: <ext-link ext-link-type="uri" xlink:href="http://paramecium.cgm.cnrs-gif.fr">http://paramecium.cgm.cnrs-gif.fr</ext-link>; and <italic>O. trifallax</italic>: <ext-link ext-link-type="uri" xlink:href="https://oxy.ciliate.org">https://oxy.ciliate.org</ext-link>) by BLASTX. Contigs with higher bit score to bacteria NR or G+C &gt;45% were removed. The <italic>E. woodruffi</italic> MIC genome was decontaminated, similarly, with addition of all <italic>Chlorogonium</italic> sequences (the algal food source) on NCBI and the two <italic>Euplotes</italic> mitochondrial genomes (<italic>E. minuta</italic> GQ903130.1, <italic>E. crassus</italic> GQ903131.1) to filter contaminants.</p></sec><sec id="s4-9"><title>Repeat identification</title><p>The repeat content in the MIC genomes was identified by RepeatModeler 1.0.10 (<xref ref-type="bibr" rid="bib95">Smit and Hubley, 2008</xref>) and RepeatMasker 4.0.7 (<xref ref-type="bibr" rid="bib96">Smit et al., 2013</xref>) with default parameters.</p></sec><sec id="s4-10"><title>TBE/Tec detection</title><p>Representative <italic>Oxytricha</italic> TBE ORFs (Genbank accession AAB42034.1, AAB42016.1, and AAB42018.1) were used as queries to search TBEs in the <italic>Oxytricha</italic> and <italic>Tetmemena</italic> MIC genomes by TBLASTN (-db_gencode 6 -evalue 1e-7 -max_target_seqs 30000). Tec ORFs were similarly detected by using <italic>E. crassus</italic> Tec1 and Tec2 ORFs as queries (-db_gencode 10 -evalue 1e-5 -max_target_seqs 30000, Genbank accessions of Tec ORFs are AAA62601.1, AAA62602.1, AAA62603.1, AAA91339.1, AAA91340.1, AAA91341.1, AAA91342.1). Complete TBEs/Tecs were determined by custom python scripts when three ORFs are within 2000 bp from each other and in correct orientation (<ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/TBE_ORFs/TBE_to_oxy_genome_tblastn_parse.py">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/TBE_ORFs/TBE_to_oxy_genome_tblastn_parse.py</ext-link>, <xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). 30 TBE ORFs with &gt;70% completeness were subsampled from each species for phylogenetic analysis (except for the 57 kD ORF in <italic>Tetmemena</italic>, for which 21 were subsampled). The subsampled TBE ORFs were aligned using MUSCLE (<xref ref-type="bibr" rid="bib29">Edgar, 2004</xref>), and the alignments were trimmed by trimAl ‘-automated1’ (<xref ref-type="bibr" rid="bib17">Capella-Gutiérrez et al., 2009</xref>). Phylogenetic trees were constructed using PhyML 3.3 (<xref ref-type="bibr" rid="bib45">Guindon et al., 2010</xref>).</p></sec><sec id="s4-11"><title>Rearrangement annotations</title><p>SDRAP (<xref ref-type="bibr" rid="bib13">Braun et al., 2022</xref>) was used to annotate MDSs, pointers, and MIC-specific regions (minimum percent identity for preliminary match annotation = 95, minimum percent identity for additional match annotation = 90, minimum length of pointer annotation = 2). SDRAP requires MAC and MIC genomes as input. For the SDRAP annotation of <italic>Oxytricha</italic>, we used the MAC genome from <xref ref-type="bibr" rid="bib101">Swart et al., 2013</xref> instead of the latest hybrid assembly that incorporated PacBio reads (<xref ref-type="bibr" rid="bib65">Lindblad et al., 2019</xref>), because the former version was primarily based on Illumina reads, similar to the MAC genomes of <italic>Tetmemena</italic> (7, Genbank GCA_001273295.2) and <italic>E. woodruffi</italic> which are also Illumina assemblies. <italic>Oxytricha</italic> and <italic>Tetmemena</italic> MAC genomes were preprocessed by removing MAC contigs with TBE ORFs, considered MIC contaminants (<xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>). SDRAP is a new program that can output the rearrangement annotations with minor differences from <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>, but most annotations are robust (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). Scrambled and nonscrambled junctions/IESs were annotated by custom python scripts (<ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/scrambled_nonscrambled_IES_pointer">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/scrambled_nonscrambled_IES_pointer</ext-link>).</p></sec><sec id="s4-12"><title>MIC genome categories</title><p>Each MIC genome region is assigned to only one category in <xref ref-type="fig" rid="fig2">Figure 2A–C</xref>, even if it belongs to more than one category. The assignment is based on the following priority: MDS, TBE/Tec, MIC genes (only available for <italic>Oxytricha</italic>, which has developmental RNA-seq data), IES, tandem repeats, other repeats, and non-coding non-repetitive regions. For example, an MIC region can be a TBE in an IES, and it is only considered as TBE in <xref ref-type="fig" rid="fig2">Figure 2A–C</xref>.</p></sec><sec id="s4-13"><title>Ortholog comparison pipeline and Monte Carlo simulations</title><p>Orthogroups of genes on telomeric MAC contigs were detected by OrthoFinder with ‘-S blast’ (<xref ref-type="bibr" rid="bib32">Emms and Kelly, 2019</xref>). Single-copy orthologs were aligned by Clustal Omega (<xref ref-type="bibr" rid="bib92">Sievers et al., 2011</xref>). Protein alignments were reversely translated to CDS alignments by a modified script of pal2nal (<xref ref-type="bibr" rid="bib100">Suyama et al., 2006</xref>, <ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/Ortholog_comparison/pal2nal.pl">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/Ortholog_comparison/pal2nal.pl</ext-link>). Two modifications were made in the script: (1) the modified script allows pal2nal to take different genetic codes for three sequences (-codontable 6,6,10); (2) the script also fixed an error in the original pal2nal script in which codontable 10 for the Euplotid nuclear code was the same as the universal code. Visualization of pointer positions and intron locations on orthologs was implemented by a custom python script (<ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/blob/main/Ortholog_comparison/visualization_of_ortholog_comparison.py">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/blob/main/Ortholog_comparison/visualization_of_ortholog_comparison.py</ext-link>). Pointer positions or intron locations are considered conserved if they are within a 20-bp alignment window on the CDS alignment. Protein domains were annotated by HMMER (<xref ref-type="bibr" rid="bib37">Finn et al., 2011</xref>). We performed Monte Carlo simulations by randomly shuffling pointer locations on the CDS but keeping their original position distribution. This was implemented by a custom python script, which transforms the CDS to a circle, rotates pointer positions on the circle, and outputs the shuffled position on the re-linearized CDS (<ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/blob/main/Ortholog_comparison/shuffle_simulation.py">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/blob/main/Ortholog_comparison/shuffle_simulation.py</ext-link>). The null hypothesis of the Monte Carlo test is that pointer positions are conserved by chance. p-Value of Monte Carlo test is given by N<sub>expected&gt;observed</sub>/N<sub>total</sub> (N<sub>expected&gt;observed</sub> is the number of simulations when there are more conserved pointers in the simulation than the observation from real data, N<sub>total</sub> = 1000 in this study).</p></sec><sec id="s4-14"><title>PCR validation of Russian doll locus</title><p>The complex Russian doll locus on MIC contig TMEMEN_MIC_21461 in <italic>Tetmemena</italic> was validated by PCR to confirm the <italic>Tetmemena</italic> MIC genome assembly. <italic>Tetmemena</italic> micronuclear DNA was purified as described previously and used as template for PCR using PrimeSTAR Max DNA polymerase (Takara Bio). 11 primer sets (<xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>) were designed to amplify products between 3 kb and 6 kb in length, with overlapping regions between consecutive primer pairs. The resulting PCR products were visualized through agarose gel electrophoresis, and bands of the expected size were extracted using a Monarch DNA Gel Extraction Kit (New England Biolabs). The purified gel bands were cloned using a TOPO XL-2 Complete PCR Cloning Kit (Invitrogen), transformed into One Shot OmniMAX 2 T1R <italic>E. coli</italic> cells (Invitrogen), and individual clones were grown and their plasmids harvested with a QIAprep Spin Miniprep Kit (QIAGEN). The plasmid ends were Sanger sequenced, as well as the region where the <italic>Oxytricha</italic> MIC assembly contains inserted MDSs (Genewiz). Sanger sequencing reads were mapped to the <italic>Tetmemena</italic> MIC contig TMEMEN_MIC_21461 and visualized using Geneious Prime 2021.1.1 (<ext-link ext-link-type="uri" xlink:href="https://www.geneious.com">https://www.geneious.com</ext-link>).</p></sec><sec id="s4-15"><title>Availability of data and materials</title><p>Custom scripts are public on <ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes</ext-link>, (<xref ref-type="bibr" rid="bib36">Feng, 2022b</xref> copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8ad132d58c3073da701bdde6700a37e2cdc01509;origin=https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes;visit=swh:1:snp:3e53ca9f9f0b0bc48a5c56d379e0def68cce596f;anchor=swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400">swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400</ext-link>). DNA-seq reads and genome assemblies are available at GenBank under Bioprojects PRJNA694964 (<italic>Tetmemena sp.</italic>) and PRJNA781979 (<italic>E. woodruffi</italic>). Genbank accession numbers for genomes are JAJKFJ000000000 (<italic>Tetmemena sp.</italic> Micronucleus genome), JAJLLS000000000 (<italic>E. woodruffi</italic> Micronucleus genome), and JAJLLT000000000 (<italic>E. woodruffi</italic> Macronucleus genome).</p><p>Three replicates of RNA-seq reads for vegetative cells are available at GenBank under accession numbers of SRR21815378, SRR21815379, and SRR21815380 for <italic>E. woodruffi</italic> and SRR21817702, SRR21817703, and SRR21817704 for <italic>Tetmemena sp.</italic></p><p>MDS annotations for three species are available at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5061/dryad.5dv41ns96">https://doi.org/10.5061/dryad.5dv41ns96</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://knot.math.usf.edu/mds_ies_db/2022/downloads.html">https://knot.math.usf.edu/mds_ies_db/2022/downloads.html</ext-link> (please select species from the drop-down menu).</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf3"><p>The author is currently employed by Illumina</p></fn><fn fn-type="COI-statement" id="conf4"><p>employed by Pacific Biosciences</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Visualization, Methodology, Writing - original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Methodology</p></fn><fn fn-type="con" id="con3"><p>Investigation</p></fn><fn fn-type="con" id="con4"><p>Investigation</p></fn><fn fn-type="con" id="con5"><p>Software</p></fn><fn fn-type="con" id="con6"><p>Investigation, Validation</p></fn><fn fn-type="con" id="con7"><p>Conceptualization, Supervision, Funding acquisition, Investigation, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Sequencing depth statistics for germline micronucleus (MIC) genome assemblies.</title><p>*Sequencing data from <xref ref-type="bibr" rid="bib20">Chen et al., 2014</xref>.</p><p>**Raw reads were mapped to the MIC genome assembly by Minimap2 and Bowtie2 (<xref ref-type="bibr" rid="bib62">Langmead and Salzberg, 2012</xref>). Average coverage was calculated with BBmap (<ext-link ext-link-type="uri" xlink:href="http://sourceforge.net/projects/bbmap/">sourceforge.net/projects/bbmap/</ext-link>) pileup.sh for macronuclear destined sequence-containing contigs in the MIC genome assembly.</p></caption><media xlink:href="elife-82979-supp1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Subcategories of repeat content in the three species.</title><p>Repeat content of the three genomes, as annotated by Repeatmasker (<xref ref-type="bibr" rid="bib96">Smit et al., 2013</xref>) with additional manual annotation of Telomere-Bearing Element (TBE)/Transposon of <italic>Euplotes crassus</italic> (TEC) elements. The numbers may differ from <xref ref-type="fig" rid="fig2">Figure 2A–C</xref> because some repeats are assigned as other germline micronucleus (MIC) categories in the pie charts (Methods). For example, a MIC region which is both an internally eliminated sequence (IES) and satellite, is assigned as IES in <xref ref-type="fig" rid="fig2">Figure 2A–C</xref>, but is counted as a satellite in this table.</p></caption><media xlink:href="elife-82979-supp2-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Telomere-bearing elements (TBE)/transposon of <italic>Euplotes crassus</italic> (TEC) elements open reading frames in three species.</title><p>* Differs from 10,109 in Chen et al. (<xref ref-type="bibr" rid="bib22">Chen and Landweber, 2016</xref>) because we used different versions of BLAST and custom python scripts to identify complete TBEs (see Methods).</p></caption><media xlink:href="elife-82979-supp3-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Orthology among scrambled and nonscrambled genes in the three species.</title><p>* Ciliate database is generated by extracting all protein sequences in phylum Ciliophora (taxid: 5878) from NR database.</p></caption><media xlink:href="elife-82979-supp4-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Summary of orthologs in each pair of species.</title><p>The (<italic>i,j</italic>) cell shows the number of genes in species <italic>i</italic> with an ortholog in species <italic>j</italic>.</p><p>* Genes with no ortholog detected by OrthoFinder (<xref ref-type="bibr" rid="bib32">Emms and Kelly, 2019</xref>) in the other two species.</p></caption><media xlink:href="elife-82979-supp5-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>More scrambled somatic macronucleus (MAC) contigs contain at least one paralogous macronuclear destined sequence that may be involved in alternative rearrangement.</title></caption><media xlink:href="elife-82979-supp6-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp7"><label>Supplementary file 7.</label><caption><title>Macronuclear destined sequence (MDS)-internally eliminated sequence (IES) pairs share homologous sequences in the three species (related to <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>).</title></caption><media xlink:href="elife-82979-supp7-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp8"><label>Supplementary file 8.</label><caption><title>Genes with expression support in the three species.</title></caption><media xlink:href="elife-82979-supp8-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp9"><label>Supplementary file 9.</label><caption><title>Presence of conserved pointers in three species, with Monte Carlo simulations.</title></caption><media xlink:href="elife-82979-supp9-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp10"><label>Supplementary file 10.</label><caption><title>Scrambled pointers are more conserved than nonscrambled pointers.</title></caption><media xlink:href="elife-82979-supp10-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp11"><label>Supplementary file 11.</label><caption><title>Most pointers conserved in position are different in sequence.</title></caption><media xlink:href="elife-82979-supp11-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp12"><label>Supplementary file 12.</label><caption><title>Intron-IES conversion comparison in three species and Monte Carlo simulations.</title></caption><media xlink:href="elife-82979-supp12-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp13"><label>Supplementary file 13.</label><caption><title>Pairwise intron-IES conversion comparisons and Monte Carlo simulations.</title></caption><media xlink:href="elife-82979-supp13-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp14"><label>Supplementary file 14.</label><caption><title>PCR primers for validation of the Russian doll region in <italic>Tetmemena</italic> DNA (<xref ref-type="fig" rid="fig6">Figure 6A</xref>).</title></caption><media xlink:href="elife-82979-supp14-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-82979-mdarchecklist1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Custom scripts are public on <ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes</ext-link>, (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8ad132d58c3073da701bdde6700a37e2cdc01509;origin=https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes;visit=swh:1:snp:3e53ca9f9f0b0bc48a5c56d379e0def68cce596f;anchor=swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400">swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400</ext-link>). DNA-seq reads and genome assemblies are available at GenBank under Bioprojects PRJNA694964 (<italic>Tetmemena sp</italic>.) and PRJNA781979 (<italic>Euplotes woodruffi</italic>). Genbank accession numbers for genomes are JAJKFJ000000000 (<italic>Tetmemena sp.</italic> Micronucleus genome), JAJLLS000000000 (<italic>Euplotes woodruffi</italic> Micronucleus genome), and JAJLLT000000000 (<italic>Euplotes woodruffi</italic> Macronucleus genome). Three replicates of RNA-seq reads for vegetative cells are available at GenBank under accession numbers of SRR21815378, SRR21815379, SRR21815380 for <italic>E. woodruffi</italic> and SRR21817702, SRR21817703 and SRR21817704 for <italic>Tetmemena sp</italic>. MDS annotations for three species are available at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5061/dryad.5dv41ns96">https://doi.org/10.5061/dryad.5dv41ns96</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://knot.math.usf.edu/mds_ies_db/2022/downloads.html">https://knot.math.usf.edu/mds_ies_db/2022/downloads.html</ext-link> (please select species from the drop-down menu).</p><p>The following datasets were generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>MW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Euplotes woodruffi genome sequencing and assembly</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA781979">PRJNA781979</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset2"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>MW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Tetmemena sp. micronucleus genome sequencing and assembly</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA694964">PRJNA694964</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset3"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>MW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Euplotes woodruffi strain:Iz01</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA781602">PRJNA781602</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset4"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>MW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>RNA-seq of Tetmemena sp</data-title><source>NCBI BioProject</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA887426">PRJNA887426</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset5"><person-group person-group-type="author"><name><surname>Channagiri</surname><given-names>T</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>MDS-IES database</data-title><source>MDSIESDB</source><pub-id pub-id-type="accession" xlink:href="https://knot.math.usf.edu/mds_ies_db/2022/downloads.html">db/2022/downloads</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset6"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>L</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>M</given-names></name><name><surname>Landweber</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>MDS and IES annotations for Euplotes woodruff, Tetmemena sp. and Oxytricha trifallax</data-title><source>Dryad Digital Repository</source><pub-id pub-id-type="doi">10.5061/dryad.5dv41ns96</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset7"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>MW</surname><given-names>Lu</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Tetmemena sp. Micronucleus genome</data-title><source>NCBI GenBank</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/nuccore/JAJKFJ000000000">JAJKFJ000000000</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset8"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>MW</surname><given-names>Lu</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Euplotes woodruffi Micronucleus genome</data-title><source>NCBI GenBank</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/nuccore/JAJLLS000000000">JAJLLS000000000</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset9"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>MW</surname><given-names>Lu</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Euplotes woodruffi Macronucleus genome</data-title><source>NCBI GenBank</source><comment>JAJLLT000000000</comment></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset10"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Goldman</surname><given-names>AD</given-names></name><name><surname>Dolzhenko</surname><given-names>E</given-names></name><name><surname>Clay</surname><given-names>DM</given-names></name><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Perlman</surname><given-names>DH</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Stuart</surname><given-names>A</given-names></name><name><surname>Amemiya</surname><given-names>CT</given-names></name><name><surname>Sebra</surname><given-names>RP</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2014">2014</year><data-title>Oxytricha trifallax micronucleus genome</data-title><source>NCBI Assembly</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/assembly/GCA_000711775.1">GCA_000711775.1</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset11"><person-group person-group-type="author"><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Magrini</surname><given-names>V</given-names></name><name><surname>Minx</surname><given-names>P</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Zhou</surname><given-names>Y</given-names></name><name><surname>Khurana</surname><given-names>JS</given-names></name><name><surname>Goldman</surname><given-names>AD</given-names></name><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Schotanus</surname><given-names>K</given-names></name><name><surname>Jung</surname><given-names>S</given-names></name><name><surname>Ly</surname><given-names>A</given-names></name><name><surname>McGrath</surname><given-names>S</given-names></name><name><surname>Haub</surname><given-names>K</given-names></name><name><surname>Wiggins</surname><given-names>JL</given-names></name><name><surname>Storton</surname><given-names>D</given-names></name><name><surname>Matese</surname><given-names>JC</given-names></name><name><surname>Parsons</surname><given-names>L</given-names></name><name><surname>Chang</surname><given-names>WJ</given-names></name><name><surname>Bowen</surname><given-names>MS</given-names></name><name><surname>Stover</surname><given-names>NA</given-names></name><name><surname>Jones</surname><given-names>TA</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Herrick</surname><given-names>GA</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Mardis</surname><given-names>ER</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2013">2013</year><data-title>Oxytricha trifallax macronucleus genome</data-title><source>NCBI Assembly</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/assembly/GCA_000295675.1/">GCA_000295675.1</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset12"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Jung</surname><given-names>S</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2015">2015</year><data-title>Tetmemena sp. macronucleus genome</data-title><source>NCBI Assembly</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/assembly/GCA_001273295.2">GCA_001273295.2</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset13"><person-group person-group-type="author"><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Debelouchina</surname><given-names>GT</given-names></name><name><surname>Clay</surname><given-names>DM</given-names></name><name><surname>Thompson</surname><given-names>RE</given-names></name><name><surname>Lindblad</surname><given-names>KA</given-names></name><name><surname>Hutton</surname><given-names>ER</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Sebra</surname><given-names>RP</given-names></name><name><surname>Muir</surname><given-names>TW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Genome-wide analysis of chromatin and transcription in the ciliates Oxytricha trifallax and Tetrahymena thermophila</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE94421">GSE94421</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Toshinobu Suzaki (Kobe University) for the gift of <italic>E. woodruffi</italic> (strain Iz01 from Shizuoka Prefecture) and <italic>Chlorogonium capillatum</italic>. We thank Sheela George for laboratory support and help with cell collection. We thank David Dai, Eoghan Harrington, John Beaulaurier, and Sissel Juul at Oxford Nanopore Technologies in New York for providing sequencing and advice. We thank Robert Sebra and Melissa Smith for advice and PacBio sequencing. We thank Takahiko Akematsu, Lorraine Symington, and Lea Marie for helping with PFGE. We thank Kaiyi Zhu, Shaojie He, Molly Przeworski, Harmen Bussemaker, and Nataša Jonoska for advice on Monte Carlo simulations. We also thank Scott Roy, Samuel Sternberg, Bill Jack, and all current and past Landweber lab members for discussion about the origin of scrambled genes, as well as David Prescott and Klaus Heckmann for inspiration, and Margarita T Angelova, Sindhuja Devanapally, Danylo Villano and Kehan Bao for comments on the manuscript. This work was supported by the National Institutes of Health, R35GM122555, and National Science Foundation, DMS1764366, and the National Center for Genome Analysis Support computing resources (supported by National Science Foundation DBI1062432, ABI1458641, and ABI1759906 to Indiana University). Rafik Neme was supported by the Pew Latin American Fellows Program.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akematsu</surname><given-names>T</given-names></name><name><surname>Fukuda</surname><given-names>Y</given-names></name><name><surname>Garg</surname><given-names>J</given-names></name><name><surname>Fillingham</surname><given-names>JS</given-names></name><name><surname>Pearlman</surname><given-names>RE</given-names></name><name><surname>Loidl</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Post-meiotic DNA double-strand breaks occur in <italic>Tetrahymena</italic>, and require topoisomerase II and SPO11</article-title><source>eLife</source><volume>6</volume><elocation-id>e26176</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.26176</pub-id><pub-id pub-id-type="pmid">28621664</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albertin</surname><given-names>CB</given-names></name><name><surname>Medina-Ruiz</surname><given-names>S</given-names></name><name><surname>Mitros</surname><given-names>T</given-names></name><name><surname>Schmidbaur</surname><given-names>H</given-names></name><name><surname>Sanchez</surname><given-names>G</given-names></name><name><surname>Wang</surname><given-names>ZY</given-names></name><name><surname>Grimwood</surname><given-names>J</given-names></name><name><surname>Rosenthal</surname><given-names>JJC</given-names></name><name><surname>Ragsdale</surname><given-names>CW</given-names></name><name><surname>Simakov</surname><given-names>O</given-names></name><name><surname>Rokhsar</surname><given-names>DS</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Genome and transcriptome mechanisms driving cephalopod evolution</article-title><source>Nature Communications</source><volume>13</volume><elocation-id>2427</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-022-29748-w</pub-id><pub-id pub-id-type="pmid">35508532</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arnaiz</surname><given-names>O</given-names></name><name><surname>Mathy</surname><given-names>N</given-names></name><name><surname>Baudry</surname><given-names>C</given-names></name><name><surname>Malinsky</surname><given-names>S</given-names></name><name><surname>Aury</surname><given-names>JM</given-names></name><name><surname>Denby Wilkes</surname><given-names>C</given-names></name><name><surname>Garnier</surname><given-names>O</given-names></name><name><surname>Labadie</surname><given-names>K</given-names></name><name><surname>Lauderdale</surname><given-names>BE</given-names></name><name><surname>Le Mouël</surname><given-names>A</given-names></name><name><surname>Marmignon</surname><given-names>A</given-names></name><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Poulain</surname><given-names>J</given-names></name><name><surname>Prajer</surname><given-names>M</given-names></name><name><surname>Wincker</surname><given-names>P</given-names></name><name><surname>Meyer</surname><given-names>E</given-names></name><name><surname>Duharcourt</surname><given-names>S</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Bétermier</surname><given-names>M</given-names></name><name><surname>Sperling</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The <italic>Paramecium</italic> germline genome provides a niche for intragenic parasitic DNA: evolutionary dynamics of internal eliminated sequences</article-title><source>PLOS Genetics</source><volume>8</volume><elocation-id>e1002984</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1002984</pub-id><pub-id pub-id-type="pmid">23071448</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aury</surname><given-names>JM</given-names></name><name><surname>Jaillon</surname><given-names>O</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Noel</surname><given-names>B</given-names></name><name><surname>Jubin</surname><given-names>C</given-names></name><name><surname>Porcel</surname><given-names>BM</given-names></name><name><surname>Ségurens</surname><given-names>B</given-names></name><name><surname>Daubin</surname><given-names>V</given-names></name><name><surname>Anthouard</surname><given-names>V</given-names></name><name><surname>Aiach</surname><given-names>N</given-names></name><name><surname>Arnaiz</surname><given-names>O</given-names></name><name><surname>Billaut</surname><given-names>A</given-names></name><name><surname>Beisson</surname><given-names>J</given-names></name><name><surname>Blanc</surname><given-names>I</given-names></name><name><surname>Bouhouche</surname><given-names>K</given-names></name><name><surname>Câmara</surname><given-names>F</given-names></name><name><surname>Duharcourt</surname><given-names>S</given-names></name><name><surname>Guigo</surname><given-names>R</given-names></name><name><surname>Gogendeau</surname><given-names>D</given-names></name><name><surname>Katinka</surname><given-names>M</given-names></name><name><surname>Keller</surname><given-names>AM</given-names></name><name><surname>Kissmehl</surname><given-names>R</given-names></name><name><surname>Klotz</surname><given-names>C</given-names></name><name><surname>Koll</surname><given-names>F</given-names></name><name><surname>Le Mouël</surname><given-names>A</given-names></name><name><surname>Lepère</surname><given-names>G</given-names></name><name><surname>Malinsky</surname><given-names>S</given-names></name><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Nowak</surname><given-names>JK</given-names></name><name><surname>Plattner</surname><given-names>H</given-names></name><name><surname>Poulain</surname><given-names>J</given-names></name><name><surname>Ruiz</surname><given-names>F</given-names></name><name><surname>Serrano</surname><given-names>V</given-names></name><name><surname>Zagulski</surname><given-names>M</given-names></name><name><surname>Dessen</surname><given-names>P</given-names></name><name><surname>Bétermier</surname><given-names>M</given-names></name><name><surname>Weissenbach</surname><given-names>J</given-names></name><name><surname>Scarpelli</surname><given-names>C</given-names></name><name><surname>Schächter</surname><given-names>V</given-names></name><name><surname>Sperling</surname><given-names>L</given-names></name><name><surname>Meyer</surname><given-names>E</given-names></name><name><surname>Cohen</surname><given-names>J</given-names></name><name><surname>Wincker</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Global trends of whole-genome duplications revealed by the ciliate <italic>Paramecium tetraurelia</italic></article-title><source>Nature</source><volume>444</volume><fpage>171</fpage><lpage>178</lpage><pub-id pub-id-type="doi">10.1038/nature05230</pub-id><pub-id pub-id-type="pmid">17086204</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baird</surname><given-names>SE</given-names></name><name><surname>Fino</surname><given-names>GM</given-names></name><name><surname>Tausta</surname><given-names>SL</given-names></name><name><surname>Klobutcher</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="1989">1989</year><article-title>Micronuclear genome organization in <italic>Euplotes crassus</italic>: a transposonlike element is removed during macronuclear development</article-title><source>Molecular and Cellular Biology</source><volume>9</volume><fpage>3793</fpage><lpage>3807</lpage><pub-id pub-id-type="doi">10.1128/mcb.9.9.3793-3807.1989</pub-id><pub-id pub-id-type="pmid">2550802</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bankevich</surname><given-names>A</given-names></name><name><surname>Nurk</surname><given-names>S</given-names></name><name><surname>Antipov</surname><given-names>D</given-names></name><name><surname>Gurevich</surname><given-names>AA</given-names></name><name><surname>Dvorkin</surname><given-names>M</given-names></name><name><surname>Kulikov</surname><given-names>AS</given-names></name><name><surname>Lesin</surname><given-names>VM</given-names></name><name><surname>Nikolenko</surname><given-names>SI</given-names></name><name><surname>Pham</surname><given-names>S</given-names></name><name><surname>Prjibelski</surname><given-names>AD</given-names></name><name><surname>Pyshkin</surname><given-names>AV</given-names></name><name><surname>Sirotkin</surname><given-names>AV</given-names></name><name><surname>Vyahhi</surname><given-names>N</given-names></name><name><surname>Tesler</surname><given-names>G</given-names></name><name><surname>Alekseyev</surname><given-names>MA</given-names></name><name><surname>Pevzner</surname><given-names>PA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>SPAdes: a new genome assembly algorithm and its applications to single-cell sequencing</article-title><source>Journal of Computational Biology</source><volume>19</volume><fpage>455</fpage><lpage>477</lpage><pub-id pub-id-type="doi">10.1089/cmb.2012.0021</pub-id><pub-id pub-id-type="pmid">22506599</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Debelouchina</surname><given-names>GT</given-names></name><name><surname>Clay</surname><given-names>DM</given-names></name><name><surname>Thompson</surname><given-names>RE</given-names></name><name><surname>Lindblad</surname><given-names>KA</given-names></name><name><surname>Hutton</surname><given-names>ER</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Sebra</surname><given-names>RP</given-names></name><name><surname>Muir</surname><given-names>TW</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Identification of a DNA N6-adenine methyltransferase complex and its impact on chromatin organization</article-title><source>Cell</source><volume>177</volume><fpage>1781</fpage><lpage>1796</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2019.04.028</pub-id><pub-id pub-id-type="pmid">31104845</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Biederman</surname><given-names>MK</given-names></name><name><surname>Nelson</surname><given-names>MM</given-names></name><name><surname>Asalone</surname><given-names>KC</given-names></name><name><surname>Pedersen</surname><given-names>AL</given-names></name><name><surname>Saldanha</surname><given-names>CJ</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Discovery of the first germline-restricted gene by subtractive transcriptomic analysis in the zebra finch, <italic>Taeniopygia guttata</italic></article-title><source>Current Biology</source><volume>28</volume><fpage>1620</fpage><lpage>1627</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2018.03.067</pub-id><pub-id pub-id-type="pmid">29731307</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boscaro</surname><given-names>V</given-names></name><name><surname>Husnik</surname><given-names>F</given-names></name><name><surname>Vannini</surname><given-names>C</given-names></name><name><surname>Keeling</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Symbionts of the ciliate <italic>Euplotes</italic>: diversity, patterns and potential as models for bacteria-eukaryote endosymbioses</article-title><source>Proceedings of the Royal Society B: Biological Sciences</source><volume>286</volume><elocation-id>20190693</elocation-id><pub-id pub-id-type="doi">10.1098/rspb.2019.0693</pub-id><pub-id pub-id-type="pmid">31311477</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Perlman</surname><given-names>DH</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Cytosine methylation and hydroxymethylation mark DNA for elimination in <italic>Oxytricha trifallax</italic></article-title><source>Genome Biology</source><volume>13</volume><fpage>1</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1186/gb-2012-13-10-r99</pub-id><pub-id pub-id-type="pmid">23075511</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Fang</surname><given-names>W</given-names></name><name><surname>Goldman</surname><given-names>AD</given-names></name><name><surname>Dolzhenko</surname><given-names>E</given-names></name><name><surname>Stein</surname><given-names>EM</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Genomes on the edge: programmed genome instability in ciliates</article-title><source>Cell</source><volume>152</volume><fpage>406</fpage><lpage>416</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.01.005</pub-id><pub-id pub-id-type="pmid">23374338</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Nabergall</surname><given-names>L</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name><name><surname>Saito</surname><given-names>M</given-names></name><name><surname>Jonoska</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Russian doll genes and complex chromosome rearrangements in <italic>Oxytricha trifallax</italic></article-title><source>G3: Genes, Genomes, Genetics</source><volume>8</volume><fpage>1669</fpage><lpage>1674</lpage><pub-id pub-id-type="doi">10.1534/g3.118.200176</pub-id><pub-id pub-id-type="pmid">29545465</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Braun</surname><given-names>J</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name><name><surname>Jonoska</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>SDRAP for Annotating Scrambled or Rearranged Genomes</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.10.24.513505</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brůna</surname><given-names>T</given-names></name><name><surname>Hoff</surname><given-names>KJ</given-names></name><name><surname>Lomsadze</surname><given-names>A</given-names></name><name><surname>Stanke</surname><given-names>M</given-names></name><name><surname>Borodovsky</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>BRAKER2: automatic eukaryotic genome annotation with genemark-EP+ and AUGUSTUS supported by a protein database</article-title><source>NAR Genomics and Bioinformatics</source><volume>3</volume><elocation-id>lqaa108</elocation-id><pub-id pub-id-type="doi">10.1093/nargab/lqaa108</pub-id><pub-id pub-id-type="pmid">33575650</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burns</surname><given-names>J</given-names></name><name><surname>Kukushkin</surname><given-names>D</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name><name><surname>Saito</surname><given-names>M</given-names></name><name><surname>Jonoska</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Recurring patterns among scrambled genes in the encrypted genome of the ciliate <italic>Oxytricha trifallax</italic></article-title><source>Journal of Theoretical Biology</source><volume>410</volume><fpage>171</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1016/j.jtbi.2016.08.038</pub-id><pub-id pub-id-type="pmid">27593332</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Camacho</surname><given-names>C</given-names></name><name><surname>Coulouris</surname><given-names>G</given-names></name><name><surname>Avagyan</surname><given-names>V</given-names></name><name><surname>Ma</surname><given-names>N</given-names></name><name><surname>Papadopoulos</surname><given-names>J</given-names></name><name><surname>Bealer</surname><given-names>K</given-names></name><name><surname>Madden</surname><given-names>TL</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>BLAST+: architecture and applications</article-title><source>BMC Bioinformatics</source><volume>10</volume><fpage>1</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1186/1471-2105-10-421</pub-id><pub-id pub-id-type="pmid">20003500</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Capella-Gutiérrez</surname><given-names>S</given-names></name><name><surname>Silla-Martínez</surname><given-names>JM</given-names></name><name><surname>Gabaldón</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>TrimAl: a tool for automated alignment trimming in large-scale phylogenetic analyses</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1972</fpage><lpage>1973</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp348</pub-id><pub-id pub-id-type="pmid">19505945</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname><given-names>M</given-names></name><name><surname>Baldwin-Brown</surname><given-names>JG</given-names></name><name><surname>Long</surname><given-names>AD</given-names></name><name><surname>Emerson</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Contiguous and accurate de novo assembly of metazoan genomes with modest long read coverage</article-title><source>Nucleic Acids Research</source><volume>44</volume><elocation-id>e147</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkw654</pub-id><pub-id pub-id-type="pmid">27458204</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>WJ</given-names></name><name><surname>Bryson</surname><given-names>PD</given-names></name><name><surname>Liang</surname><given-names>H</given-names></name><name><surname>Shin</surname><given-names>MK</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The evolutionary origin of a complex scrambled gene</article-title><source>PNAS</source><volume>102</volume><fpage>15149</fpage><lpage>15154</lpage><pub-id pub-id-type="doi">10.1073/pnas.0507682102</pub-id><pub-id pub-id-type="pmid">16217011</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Goldman</surname><given-names>AD</given-names></name><name><surname>Dolzhenko</surname><given-names>E</given-names></name><name><surname>Clay</surname><given-names>DM</given-names></name><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Perlman</surname><given-names>DH</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Stuart</surname><given-names>A</given-names></name><name><surname>Amemiya</surname><given-names>CT</given-names></name><name><surname>Sebra</surname><given-names>RP</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The architecture of a scrambled genome reveals massive levels of genomic rearrangement during development</article-title><source>Cell</source><volume>158</volume><fpage>1187</fpage><lpage>1198</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.07.034</pub-id><pub-id pub-id-type="pmid">25171416</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Jung</surname><given-names>S</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Combinatorial DNA rearrangement facilitates the origin of new genes in ciliates</article-title><source>Genome Biology and Evolution</source><volume>7</volume><fpage>2859</fpage><lpage>2870</lpage><pub-id pub-id-type="doi">10.1093/gbe/evv172</pub-id><pub-id pub-id-type="pmid">26338187</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Phylogenomic analysis reveals genome-wide purifying selection on TBE transposons in the ciliate <italic>Oxytricha</italic></article-title><source>Mobile DNA</source><volume>7</volume><elocation-id>2</elocation-id><pub-id pub-id-type="doi">10.1186/s13100-016-0057-9</pub-id><pub-id pub-id-type="pmid">26811739</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Jiang</surname><given-names>Y</given-names></name><name><surname>Gao</surname><given-names>F</given-names></name><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Krock</surname><given-names>TJ</given-names></name><name><surname>Stover</surname><given-names>NA</given-names></name><name><surname>Lu</surname><given-names>C</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name><name><surname>Song</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Genome analyses of the new model protist <italic>Euplotes vannus</italic> focusing on genome rearrangement and resistance to environmental stressors</article-title><source>Molecular Ecology Resources</source><volume>19</volume><fpage>1292</fpage><lpage>1308</lpage><pub-id pub-id-type="doi">10.1111/1755-0998.13023</pub-id><pub-id pub-id-type="pmid">30985983</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>W</given-names></name><name><surname>Zuo</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>C</given-names></name><name><surname>Zhang</surname><given-names>T</given-names></name><name><surname>Lyu</surname><given-names>L</given-names></name><name><surname>Qiao</surname><given-names>Y</given-names></name><name><surname>Zhao</surname><given-names>F</given-names></name><name><surname>Miao</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The hidden genomic diversity of ciliated protists revealed by single-cell genome sequencing</article-title><source>BMC Biology</source><volume>19</volume><elocation-id>264</elocation-id><pub-id pub-id-type="doi">10.1186/s12915-021-01202-1</pub-id><pub-id pub-id-type="pmid">34903227</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cummings</surname><given-names>DJ</given-names></name><name><surname>Tait</surname><given-names>A</given-names></name><name><surname>Goddard</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="1974">1974</year><article-title>Methylated bases in DNA from <italic>Paramecium aurelia</italic></article-title><source>Biochimica et Biophysica Acta - Nucleic Acids and Protein Synthesis</source><volume>374</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1016/0005-2787(74)90194-4</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Denby Wilkes</surname><given-names>C</given-names></name><name><surname>Arnaiz</surname><given-names>O</given-names></name><name><surname>Sperling</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>ParTIES: a toolbox for <italic>Paramecium</italic> interspersed DNA elimination studies</article-title><source>Bioinformatics</source><volume>32</volume><fpage>599</fpage><lpage>601</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv691</pub-id><pub-id pub-id-type="pmid">26589276</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Witherspoon</surname><given-names>DJ</given-names></name><name><surname>Jahn</surname><given-names>CL</given-names></name><name><surname>Herrick</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Selection on the genes of <italic>Euplotes crassus</italic> Tec1 and Tec2 transposons: evolutionary appearance of a programmed frameshift in a Tec2 gene encoding a tyrosine family site-specific recombinase</article-title><source>Eukaryotic Cell</source><volume>2</volume><fpage>95</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1128/EC.2.1.95-102.2003</pub-id><pub-id pub-id-type="pmid">12582126</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>DuBois</surname><given-names>MI</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Scrambling of the actin I gene in two <italic>Oxytricha</italic> species</article-title><source>PNAS</source><volume>92</volume><fpage>3888</fpage><lpage>3892</lpage><pub-id pub-id-type="doi">10.1073/pnas.92.9.3888</pub-id><pub-id pub-id-type="pmid">7732002</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>1792</fpage><lpage>1797</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id><pub-id pub-id-type="pmid">15034147</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eisen</surname><given-names>JA</given-names></name><name><surname>Coyne</surname><given-names>RS</given-names></name><name><surname>Wu</surname><given-names>M</given-names></name><name><surname>Wu</surname><given-names>D</given-names></name><name><surname>Thiagarajan</surname><given-names>M</given-names></name><name><surname>Wortman</surname><given-names>JR</given-names></name><name><surname>Badger</surname><given-names>JH</given-names></name><name><surname>Ren</surname><given-names>Q</given-names></name><name><surname>Amedeo</surname><given-names>P</given-names></name><name><surname>Jones</surname><given-names>KM</given-names></name><name><surname>Tallon</surname><given-names>LJ</given-names></name><name><surname>Delcher</surname><given-names>AL</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name><name><surname>Silva</surname><given-names>JC</given-names></name><name><surname>Haas</surname><given-names>BJ</given-names></name><name><surname>Majoros</surname><given-names>WH</given-names></name><name><surname>Farzad</surname><given-names>M</given-names></name><name><surname>Carlton</surname><given-names>JM</given-names></name><name><surname>Smith</surname><given-names>RK</given-names></name><name><surname>Garg</surname><given-names>J</given-names></name><name><surname>Pearlman</surname><given-names>RE</given-names></name><name><surname>Karrer</surname><given-names>KM</given-names></name><name><surname>Sun</surname><given-names>L</given-names></name><name><surname>Manning</surname><given-names>G</given-names></name><name><surname>Elde</surname><given-names>NC</given-names></name><name><surname>Turkewitz</surname><given-names>AP</given-names></name><name><surname>Asai</surname><given-names>DJ</given-names></name><name><surname>Wilkes</surname><given-names>DE</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Cai</surname><given-names>H</given-names></name><name><surname>Collins</surname><given-names>K</given-names></name><name><surname>Stewart</surname><given-names>BA</given-names></name><name><surname>Lee</surname><given-names>SR</given-names></name><name><surname>Wilamowska</surname><given-names>K</given-names></name><name><surname>Weinberg</surname><given-names>Z</given-names></name><name><surname>Ruzzo</surname><given-names>WL</given-names></name><name><surname>Wloga</surname><given-names>D</given-names></name><name><surname>Gaertig</surname><given-names>J</given-names></name><name><surname>Frankel</surname><given-names>J</given-names></name><name><surname>Tsao</surname><given-names>CC</given-names></name><name><surname>Gorovsky</surname><given-names>MA</given-names></name><name><surname>Keeling</surname><given-names>PJ</given-names></name><name><surname>Waller</surname><given-names>RF</given-names></name><name><surname>Patron</surname><given-names>NJ</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Stover</surname><given-names>NA</given-names></name><name><surname>Krieger</surname><given-names>CJ</given-names></name><name><surname>del Toro</surname><given-names>C</given-names></name><name><surname>Ryder</surname><given-names>HF</given-names></name><name><surname>Williamson</surname><given-names>SC</given-names></name><name><surname>Barbeau</surname><given-names>RA</given-names></name><name><surname>Hamilton</surname><given-names>EP</given-names></name><name><surname>Orias</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Macronuclear genome sequence of the ciliate <italic>Tetrahymena thermophila</italic>, a model eukaryote</article-title><source>PLOS Biology</source><volume>4</volume><elocation-id>e286</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0040286</pub-id><pub-id pub-id-type="pmid">16933976</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Elliott</surname><given-names>TA</given-names></name><name><surname>Gregory</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>What’s in a genome? the C-value enigma and the evolution of eukaryotic genome content</article-title><source>Philosophical Transactions of the Royal Society of London. Series B, Biological Sciences</source><volume>370</volume><elocation-id>20140331</elocation-id><pub-id pub-id-type="doi">10.1098/rstb.2014.0331</pub-id><pub-id pub-id-type="pmid">26323762</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Emms</surname><given-names>DM</given-names></name><name><surname>Kelly</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>OrthoFinder: phylogenetic orthology inference for comparative genomics</article-title><source>Genome Biology</source><volume>20</volume><elocation-id>238</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id><pub-id pub-id-type="pmid">31727128</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Beh</surname><given-names>LY</given-names></name><name><surname>Chang</surname><given-names>WJ</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SIGAR: inferring features of genome architecture and DNA rearrangements by split-read mapping</article-title><source>Genome Biology and Evolution</source><volume>12</volume><fpage>1711</fpage><lpage>1718</lpage><pub-id pub-id-type="doi">10.1093/gbe/evaa147</pub-id><pub-id pub-id-type="pmid">32790832</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Transposon debris in ciliate genomes</article-title><source>PLOS Biology</source><volume>19</volume><elocation-id>e3001354</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.3001354</pub-id><pub-id pub-id-type="pmid">34428213</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2022">2022a</year><data-title>MAC genome telomere capping script</data-title><version designator="871eb00">871eb00</version><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/MAC_genome_telomere_capping">https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes/tree/main/MAC_genome_telomere_capping</ext-link></element-citation></ref><ref id="bib36"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2022">2022b</year><data-title>Oxytricha_Tetmemena_Euplotes</data-title><version designator="swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400">swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8ad132d58c3073da701bdde6700a37e2cdc01509;origin=https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes;visit=swh:1:snp:3e53ca9f9f0b0bc48a5c56d379e0def68cce596f;anchor=swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400">https://archive.softwareheritage.org/swh:1:dir:8ad132d58c3073da701bdde6700a37e2cdc01509;origin=https://github.com/yifeng-evo/Oxytricha_Tetmemena_Euplotes;visit=swh:1:snp:3e53ca9f9f0b0bc48a5c56d379e0def68cce596f;anchor=swh:1:rev:fd66a0efeaf9feb2d79e183313192d641b4e5400</ext-link></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Finn</surname><given-names>RD</given-names></name><name><surname>Clements</surname><given-names>J</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>HMMER web server: interactive sequence similarity searching</article-title><source>Nucleic Acids Research</source><volume>39</volume><fpage>W29</fpage><lpage>W37</lpage><pub-id pub-id-type="doi">10.1093/nar/gkr367</pub-id><pub-id pub-id-type="pmid">21593126</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>L</given-names></name><name><surname>Niu</surname><given-names>B</given-names></name><name><surname>Zhu</surname><given-names>Z</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>CD-HIT: accelerated for clustering the next-generation sequencing data</article-title><source>Bioinformatics</source><volume>28</volume><fpage>3150</fpage><lpage>3152</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id><pub-id pub-id-type="pmid">23060610</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>F</given-names></name><name><surname>Song</surname><given-names>W</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genome structure drives patterns of gene family evolution in ciliates, a case study using <italic>Chilodonella uncinata</italic> (Protista, Ciliophora, Phyllopharyngea)</article-title><source>Evolution; International Journal of Organic Evolution</source><volume>68</volume><fpage>2287</fpage><lpage>2295</lpage><pub-id pub-id-type="doi">10.1111/evo.12430</pub-id><pub-id pub-id-type="pmid">24749903</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>F</given-names></name><name><surname>Roy</surname><given-names>SW</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Analyses of alternatively processed genes in ciliates provide insights into the origins of scrambled genomes and may provide a mechanism for speciation</article-title><source>MBio</source><volume>6</volume><elocation-id>e01998-14</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.01998-14</pub-id><pub-id pub-id-type="pmid">25650397</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>F</given-names></name><name><surname>Warren</surname><given-names>A</given-names></name><name><surname>Zhang</surname><given-names>Q</given-names></name><name><surname>Gong</surname><given-names>J</given-names></name><name><surname>Miao</surname><given-names>M</given-names></name><name><surname>Sun</surname><given-names>P</given-names></name><name><surname>Xu</surname><given-names>D</given-names></name><name><surname>Huang</surname><given-names>J</given-names></name><name><surname>Yi</surname><given-names>Z</given-names></name><name><surname>Song</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The all-data-based evolutionary hypothesis of ciliated protists with a revised classification of the phylum Ciliophora (eukaryota, alveolata)</article-title><source>Scientific Reports</source><volume>6</volume><elocation-id>24874</elocation-id><pub-id pub-id-type="doi">10.1038/srep24874</pub-id><pub-id pub-id-type="pmid">27126745</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gorovsky</surname><given-names>MA</given-names></name><name><surname>Hattman</surname><given-names>S</given-names></name><name><surname>Pleger</surname><given-names>GL</given-names></name></person-group><year iso-8601-date="1973">1973</year><article-title>(6 N) methyl adenine in the nuclear DNA of a eucaryote, <italic>Tetrahymena pyriformis</italic></article-title><source>The Journal of Cell Biology</source><volume>56</volume><fpage>697</fpage><lpage>701</lpage><pub-id pub-id-type="doi">10.1083/jcb.56.3.697</pub-id><pub-id pub-id-type="pmid">4631666</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grabherr</surname><given-names>MG</given-names></name><name><surname>Haas</surname><given-names>BJ</given-names></name><name><surname>Yassour</surname><given-names>M</given-names></name><name><surname>Levin</surname><given-names>JZ</given-names></name><name><surname>Thompson</surname><given-names>DA</given-names></name><name><surname>Amit</surname><given-names>I</given-names></name><name><surname>Adiconis</surname><given-names>X</given-names></name><name><surname>Fan</surname><given-names>L</given-names></name><name><surname>Raychowdhury</surname><given-names>R</given-names></name><name><surname>Zeng</surname><given-names>Q</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name><name><surname>Mauceli</surname><given-names>E</given-names></name><name><surname>Hacohen</surname><given-names>N</given-names></name><name><surname>Gnirke</surname><given-names>A</given-names></name><name><surname>Rhind</surname><given-names>N</given-names></name><name><surname>di Palma</surname><given-names>F</given-names></name><name><surname>Birren</surname><given-names>BW</given-names></name><name><surname>Nusbaum</surname><given-names>C</given-names></name><name><surname>Lindblad-Toh</surname><given-names>K</given-names></name><name><surname>Friedman</surname><given-names>N</given-names></name><name><surname>Regev</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Full-Length transcriptome assembly from RNA-Seq data without a reference genome</article-title><source>Nature Biotechnology</source><volume>29</volume><fpage>644</fpage><lpage>652</lpage><pub-id pub-id-type="doi">10.1038/nbt.1883</pub-id><pub-id pub-id-type="pmid">21572440</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guérin</surname><given-names>F</given-names></name><name><surname>Arnaiz</surname><given-names>O</given-names></name><name><surname>Boggetto</surname><given-names>N</given-names></name><name><surname>Denby Wilkes</surname><given-names>C</given-names></name><name><surname>Meyer</surname><given-names>E</given-names></name><name><surname>Sperling</surname><given-names>L</given-names></name><name><surname>Duharcourt</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Flow cytometry sorting of nuclei enables the first global characterization of <italic>Paramecium</italic> germline DNA and transposable elements</article-title><source>BMC Genomics</source><volume>18</volume><elocation-id>327</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-017-3713-7</pub-id><pub-id pub-id-type="pmid">28446146</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guindon</surname><given-names>S</given-names></name><name><surname>Dufayard</surname><given-names>JF</given-names></name><name><surname>Lefort</surname><given-names>V</given-names></name><name><surname>Anisimova</surname><given-names>M</given-names></name><name><surname>Hordijk</surname><given-names>W</given-names></name><name><surname>Gascuel</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>New algorithms and methods to estimate maximum-likelihood phylogenies: assessing the performance of PhyML 3.0</article-title><source>Systematic Biology</source><volume>59</volume><fpage>307</fpage><lpage>321</lpage><pub-id pub-id-type="doi">10.1093/sysbio/syq010</pub-id><pub-id pub-id-type="pmid">20525638</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haas</surname><given-names>BJ</given-names></name><name><surname>Delcher</surname><given-names>AL</given-names></name><name><surname>Mount</surname><given-names>SM</given-names></name><name><surname>Wortman</surname><given-names>JR</given-names></name><name><surname>Smith</surname><given-names>RK</given-names></name><name><surname>Hannick</surname><given-names>LI</given-names></name><name><surname>Maiti</surname><given-names>R</given-names></name><name><surname>Ronning</surname><given-names>CM</given-names></name><name><surname>Rusch</surname><given-names>DB</given-names></name><name><surname>Town</surname><given-names>CD</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name><name><surname>White</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Improving the <italic>Arabidopsis</italic> genome annotation using maximal transcript alignment assemblies</article-title><source>Nucleic Acids Research</source><volume>31</volume><fpage>5654</fpage><lpage>5666</lpage><pub-id pub-id-type="doi">10.1093/nar/gkg770</pub-id><pub-id pub-id-type="pmid">14500829</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haas</surname><given-names>BJ</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name><name><surname>Zhu</surname><given-names>W</given-names></name><name><surname>Pertea</surname><given-names>M</given-names></name><name><surname>Allen</surname><given-names>JE</given-names></name><name><surname>Orvis</surname><given-names>J</given-names></name><name><surname>White</surname><given-names>O</given-names></name><name><surname>Buell</surname><given-names>CR</given-names></name><name><surname>Wortman</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Automated eukaryotic gene structure annotation using EVidenceModeler and the program to assemble spliced alignments</article-title><source>Genome Biology</source><volume>9</volume><fpage>1</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1186/gb-2008-9-1-r7</pub-id><pub-id pub-id-type="pmid">18190707</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hamilton</surname><given-names>EP</given-names></name><name><surname>Kapusta</surname><given-names>A</given-names></name><name><surname>Huvos</surname><given-names>PE</given-names></name><name><surname>Bidwell</surname><given-names>SL</given-names></name><name><surname>Zafar</surname><given-names>N</given-names></name><name><surname>Tang</surname><given-names>H</given-names></name><name><surname>Hadjithomas</surname><given-names>M</given-names></name><name><surname>Krishnakumar</surname><given-names>V</given-names></name><name><surname>Badger</surname><given-names>JH</given-names></name><name><surname>Caler</surname><given-names>EV</given-names></name><name><surname>Russ</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Structure of the germline genome of <italic>Tetrahymena thermophila</italic> and relationship to the massively rearranged somatic genome</article-title><source>eLife</source><volume>5</volume><elocation-id>e19090</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.19090</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hoffman</surname><given-names>DC</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Evolution of internal eliminated segments and scrambling in the micronuclear gene encoding DNA polymerase alpha in two <italic>Oxytricha</italic> species</article-title><source>Nucleic Acids Research</source><volume>25</volume><fpage>1883</fpage><lpage>1889</lpage><pub-id pub-id-type="doi">10.1093/nar/25.10.1883</pub-id><pub-id pub-id-type="pmid">9115353</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hogan</surname><given-names>DJ</given-names></name><name><surname>Hewitt</surname><given-names>EA</given-names></name><name><surname>Orr</surname><given-names>KE</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name><name><surname>Müller</surname><given-names>KM</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Evolution of IESs and scrambling in the actin I gene in hypotrichous ciliates</article-title><source>PNAS</source><volume>98</volume><fpage>15101</fpage><lpage>15106</lpage><pub-id pub-id-type="doi">10.1073/pnas.011578598</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Madan</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>CAP3: a DNA sequence assembly program</article-title><source>Genome Research</source><volume>9</volume><fpage>868</fpage><lpage>877</lpage><pub-id pub-id-type="doi">10.1101/gr.9.9.868</pub-id><pub-id pub-id-type="pmid">10508846</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jahn</surname><given-names>CL</given-names></name><name><surname>Krikau</surname><given-names>MF</given-names></name><name><surname>Shyman</surname><given-names>S</given-names></name></person-group><year iso-8601-date="1989">1989</year><article-title>Developmentally coordinated en masse excision of a highly repetitive element in <italic>E. crassus</italic></article-title><source>Cell</source><volume>59</volume><fpage>1009</fpage><lpage>1018</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(89)90757-5</pub-id><pub-id pub-id-type="pmid">2513126</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jahn</surname><given-names>CL</given-names></name><name><surname>Doktor</surname><given-names>SZ</given-names></name><name><surname>Frels</surname><given-names>JS</given-names></name><name><surname>Jaraczewski</surname><given-names>JW</given-names></name><name><surname>Krikau</surname><given-names>MF</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Structures of the <italic>Euplotes crassus</italic> Tec1 and Tec2 elements: identification of putative transposase coding regions</article-title><source>Gene</source><volume>133</volume><fpage>71</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.1016/0378-1119(93)90226-s</pub-id><pub-id pub-id-type="pmid">8224896</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Katz</surname><given-names>LA</given-names></name><name><surname>Kovner</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Alternative processing of scrambled genes generates protein diversity in the ciliate <italic>Chilodonella uncinata</italic></article-title><source>Journal of Experimental Zoology. Part B, Molecular and Developmental Evolution</source><volume>314</volume><fpage>480</fpage><lpage>488</lpage><pub-id pub-id-type="doi">10.1002/jez.b.21354</pub-id><pub-id pub-id-type="pmid">20700892</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kent</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>BLAT -- the BLAST-like alignment tool</article-title><source>Genome Research</source><volume>12</volume><fpage>656</fpage><lpage>664</lpage><pub-id pub-id-type="doi">10.1101/gr.229202</pub-id><pub-id pub-id-type="pmid">11932250</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klobutcher</surname><given-names>LA</given-names></name><name><surname>Herrick</surname><given-names>G</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Consensus inverted terminal repeat sequence of <italic>Paramecium</italic> IESs: resemblance to termini of Tc1-related and <italic>Euplotes</italic> Tec transposons</article-title><source>Nucleic Acids Research</source><volume>23</volume><fpage>2006</fpage><lpage>2013</lpage><pub-id pub-id-type="doi">10.1093/nar/23.11.2006</pub-id><pub-id pub-id-type="pmid">7596830</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klobutcher</surname><given-names>LA</given-names></name><name><surname>Herrick</surname><given-names>GL</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Developmental genome reorganization in ciliated protozoa: the transposon link</article-title><source>Progress in Nucleic Acid Research and Molecular Biology</source><volume>56</volume><fpage>1</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1016/s0079-6603(08)61001-6</pub-id><pub-id pub-id-type="pmid">9187050</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kolmogorov</surname><given-names>M</given-names></name><name><surname>Yuan</surname><given-names>J</given-names></name><name><surname>Lin</surname><given-names>Y</given-names></name><name><surname>Pevzner</surname><given-names>PA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Assembly of long, error-prone reads using repeat graphs</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>540</fpage><lpage>546</lpage><pub-id pub-id-type="doi">10.1038/s41587-019-0072-8</pub-id><pub-id pub-id-type="pmid">30936562</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krikau</surname><given-names>MF</given-names></name><name><surname>Jahn</surname><given-names>CL</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Tec2, a second transposon-like element demonstrating developmentally programmed excision in <italic>Euplotes crassus</italic></article-title><source>Molecular and Cellular Biology</source><volume>11</volume><fpage>4751</fpage><lpage>4759</lpage><pub-id pub-id-type="doi">10.1128/mcb.11.9.4751-4759.1991</pub-id><pub-id pub-id-type="pmid">1652062</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landweber</surname><given-names>LF</given-names></name><name><surname>Kuo</surname><given-names>TC</given-names></name><name><surname>Curtis</surname><given-names>EA</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Evolution and assembly of an extremely scrambled gene</article-title><source>PNAS</source><volume>97</volume><fpage>3298</fpage><lpage>3303</lpage><pub-id pub-id-type="doi">10.1073/pnas.97.7.3298</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Why genomes in pieces?</article-title><source>Science</source><volume>318</volume><fpage>405</fpage><lpage>407</lpage><pub-id pub-id-type="doi">10.1126/science.1150280</pub-id><pub-id pub-id-type="pmid">17947572</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname><given-names>B</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Fast gapped-read alignment with Bowtie 2</article-title><source>Nature Methods</source><volume>9</volume><fpage>357</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id><pub-id pub-id-type="pmid">22388286</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lauth</surname><given-names>MR</given-names></name><name><surname>Spear</surname><given-names>BB</given-names></name><name><surname>Heumann</surname><given-names>J</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1976">1976</year><article-title>DNA of ciliated protozoa: DNA sequence diminution during macronuclear development of <italic>Oxytricha</italic></article-title><source>Cell</source><volume>7</volume><fpage>67</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(76)90256-7</pub-id><pub-id pub-id-type="pmid">820431</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lindblad</surname><given-names>KA</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Williams</surname><given-names>AE</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Thousands of RNA-cached copies of whole chromosomes are present in the ciliate <italic>Oxytricha</italic> during development</article-title><source>RNA</source><volume>23</volume><fpage>1200</fpage><lpage>1208</lpage><pub-id pub-id-type="doi">10.1261/rna.058511.116</pub-id><pub-id pub-id-type="pmid">28450531</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lindblad</surname><given-names>KA</given-names></name><name><surname>Pathmanathan</surname><given-names>JS</given-names></name><name><surname>Moreira</surname><given-names>S</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Sebra</surname><given-names>RP</given-names></name><name><surname>Hutton</surname><given-names>ER</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Capture of complete ciliate chromosomes in single sequencing reads reveals widespread chromosome isoforms</article-title><source>BMC Genomics</source><volume>20</volume><elocation-id>1037</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-019-6189-9</pub-id><pub-id pub-id-type="pmid">31888453</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lowe</surname><given-names>TM</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>tRNAscan-SE: a program for improved detection of transfer RNA genes in genomic sequence</article-title><source>Nucleic Acids Research</source><volume>25</volume><fpage>955</fpage><lpage>964</lpage><pub-id pub-id-type="doi">10.1093/nar/25.5.955</pub-id><pub-id pub-id-type="pmid">9023104</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Lynn</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><source>The Ciliated Protozoa: Characterization, Classification, and Guide to the Literature</source><publisher-name>Springer Science &amp; Business Media</publisher-name></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Manni</surname><given-names>M</given-names></name><name><surname>Berkeley</surname><given-names>MR</given-names></name><name><surname>Seppey</surname><given-names>M</given-names></name><name><surname>Simão</surname><given-names>FA</given-names></name><name><surname>Zdobnov</surname><given-names>EM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>BUSCO update: novel and streamlined workflows along with broader and deeper phylogenetic coverage for scoring of eukaryotic, prokaryotic, and viral genomes</article-title><source>Molecular Biology and Evolution</source><volume>38</volume><fpage>4647</fpage><lpage>4654</lpage><pub-id pub-id-type="doi">10.1093/molbev/msab199</pub-id><pub-id pub-id-type="pmid">34320186</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maurer-Alcalá</surname><given-names>XX</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2018">2018a</year><article-title>Exploration of the germline genome of the ciliate <italic>Chilodonella uncinata</italic> through single-cell omics (transcriptomics and genomics)</article-title><source>MBio</source><volume>9</volume><elocation-id>e01836-17</elocation-id><pub-id pub-id-type="doi">10.1128/mBio.01836-17</pub-id><pub-id pub-id-type="pmid">29317511</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maurer-Alcalá</surname><given-names>XX</given-names></name><name><surname>Yan</surname><given-names>Y</given-names></name><name><surname>Pilling</surname><given-names>OA</given-names></name><name><surname>Knight</surname><given-names>R</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2018">2018b</year><article-title>Twisted tales: insights into genome diversity of ciliates using single-cell ’omics</article-title><source>Genome Biology and Evolution</source><volume>10</volume><fpage>1927</fpage><lpage>1939</lpage><pub-id pub-id-type="doi">10.1093/gbe/evy133</pub-id><pub-id pub-id-type="pmid">29945193</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Menzel</surname><given-names>P</given-names></name><name><surname>Ng</surname><given-names>KL</given-names></name><name><surname>Krogh</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Fast and sensitive taxonomic classification for metagenomics with Kaiju</article-title><source>Nature Communications</source><volume>7</volume><elocation-id>11257</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms11257</pub-id><pub-id pub-id-type="pmid">27071849</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Meyer</surname><given-names>F</given-names></name><name><surname>Schmidt</surname><given-names>HJ</given-names></name><name><surname>Plümper</surname><given-names>E</given-names></name><name><surname>Hasilik</surname><given-names>A</given-names></name><name><surname>Mersmann</surname><given-names>G</given-names></name><name><surname>Meyer</surname><given-names>HE</given-names></name><name><surname>Engström</surname><given-names>A</given-names></name><name><surname>Heckmann</surname><given-names>K</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>UGA is translated as cysteine in pheromone 3 of <italic>Euplotes octocarinatus</italic></article-title><source>PNAS</source><volume>88</volume><fpage>3758</fpage><lpage>3761</lpage><pub-id pub-id-type="doi">10.1073/pnas.88.9.3758</pub-id><pub-id pub-id-type="pmid">1902568</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname><given-names>RV</given-names></name><name><surname>Neme</surname><given-names>R</given-names></name><name><surname>Clay</surname><given-names>DM</given-names></name><name><surname>Pathmanathan</surname><given-names>JS</given-names></name><name><surname>Lu</surname><given-names>MW</given-names></name><name><surname>Yerlici</surname><given-names>VT</given-names></name><name><surname>Khurana</surname><given-names>JS</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Transcribed germline-limited coding sequences in <italic>Oxytricha trifallax</italic></article-title><source>G3</source><volume>11</volume><elocation-id>jkab092</elocation-id><pub-id pub-id-type="doi">10.1093/g3journal/jkab092</pub-id><pub-id pub-id-type="pmid">33772542</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mitcham</surname><given-names>JL</given-names></name><name><surname>Lynn</surname><given-names>AJ</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Analysis of a scrambled gene: the gene encoding alpha-telomere-binding protein in <italic>Oxytricha nova</italic></article-title><source>Genes &amp; Development</source><volume>6</volume><fpage>788</fpage><lpage>800</lpage><pub-id pub-id-type="doi">10.1101/gad.6.5.788</pub-id><pub-id pub-id-type="pmid">1577273</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mitreva</surname><given-names>M</given-names></name><name><surname>Blaxter</surname><given-names>ML</given-names></name><name><surname>Bird</surname><given-names>DM</given-names></name><name><surname>McCarter</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Comparative genomics of nematodes</article-title><source>Trends in Genetics</source><volume>21</volume><fpage>573</fpage><lpage>581</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2005.08.003</pub-id><pub-id pub-id-type="pmid">16099532</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Möllenbeck</surname><given-names>M</given-names></name><name><surname>Cavalcanti</surname><given-names>ARO</given-names></name><name><surname>Jönsson</surname><given-names>F</given-names></name><name><surname>Lipps</surname><given-names>HJ</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Interconversion of germline-limited and somatic DNA in a scrambled gene</article-title><source>Journal of Molecular Evolution</source><volume>63</volume><fpage>69</fpage><lpage>73</lpage><pub-id pub-id-type="doi">10.1007/s00239-005-0166-4</pub-id><pub-id pub-id-type="pmid">16755354</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morgan</surname><given-names>JT</given-names></name><name><surname>Fink</surname><given-names>GR</given-names></name><name><surname>Bartel</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Excised linear introns regulate growth in yeast</article-title><source>Nature</source><volume>565</volume><fpage>606</fpage><lpage>611</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0828-1</pub-id><pub-id pub-id-type="pmid">30651636</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mourier</surname><given-names>T</given-names></name><name><surname>Jeffares</surname><given-names>DC</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Eukaryotic intron loss</article-title><source>Science</source><volume>300</volume><elocation-id>1393</elocation-id><pub-id pub-id-type="doi">10.1126/science.1080559</pub-id><pub-id pub-id-type="pmid">12775832</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Vijayan</surname><given-names>V</given-names></name><name><surname>Zhou</surname><given-names>Y</given-names></name><name><surname>Schotanus</surname><given-names>K</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>RNA-Mediated epigenetic programming of a genome-rearrangement pathway</article-title><source>Nature</source><volume>451</volume><fpage>153</fpage><lpage>158</lpage><pub-id pub-id-type="doi">10.1038/nature06452</pub-id><pub-id pub-id-type="pmid">18046331</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Higgins</surname><given-names>BP</given-names></name><name><surname>Maquilan</surname><given-names>GM</given-names></name><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>A functional role for transposases in a large eukaryotic genome</article-title><source>Science</source><volume>324</volume><fpage>935</fpage><lpage>938</lpage><pub-id pub-id-type="doi">10.1126/science.1170023</pub-id><pub-id pub-id-type="pmid">19372392</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parenteau</surname><given-names>J</given-names></name><name><surname>Maignon</surname><given-names>L</given-names></name><name><surname>Berthoumieux</surname><given-names>M</given-names></name><name><surname>Catala</surname><given-names>M</given-names></name><name><surname>Gagnon</surname><given-names>V</given-names></name><name><surname>Abou Elela</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Introns are mediators of cell response to starvation</article-title><source>Nature</source><volume>565</volume><fpage>612</fpage><lpage>617</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0859-7</pub-id><pub-id pub-id-type="pmid">30651641</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parfrey</surname><given-names>LW</given-names></name><name><surname>Lahr</surname><given-names>DJG</given-names></name><name><surname>Knoll</surname><given-names>AH</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Estimating the timing of early eukaryotic diversification with multigene molecular clocks</article-title><source>PNAS</source><volume>108</volume><fpage>13624</fpage><lpage>13629</lpage><pub-id pub-id-type="doi">10.1073/pnas.1110633108</pub-id><pub-id pub-id-type="pmid">21810989</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>The DNA of ciliated protozoa</article-title><source>Microbiological Reviews</source><volume>58</volume><fpage>233</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.1128/mr.58.2.233-267.1994</pub-id><pub-id pub-id-type="pmid">8078435</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Prescott</surname><given-names>JD</given-names></name><name><surname>DuBois</surname><given-names>ML</given-names></name><name><surname>Prescott</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Evolution of the scrambled germline gene encoding alpha-telomere binding protein in three hypotrichous ciliates</article-title><source>Chromosoma</source><volume>107</volume><fpage>293</fpage><lpage>303</lpage><pub-id pub-id-type="doi">10.1007/s004120050311</pub-id><pub-id pub-id-type="pmid">9880762</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Riley</surname><given-names>JL</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Widespread distribution of extensive chromosomal fragmentation in ciliates</article-title><source>Molecular Biology and Evolution</source><volume>18</volume><fpage>1372</fpage><lpage>1377</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a003921</pub-id><pub-id pub-id-type="pmid">11420375</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rogozin</surname><given-names>IB</given-names></name><name><surname>Wolf</surname><given-names>YI</given-names></name><name><surname>Sorokin</surname><given-names>AV</given-names></name><name><surname>Mirkin</surname><given-names>BG</given-names></name><name><surname>Koonin</surname><given-names>EV</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Remarkable interkingdom conservation of intron positions and massive, lineage-specific intron loss and gain in eukaryotic evolution</article-title><source>Current Biology</source><volume>13</volume><fpage>1512</fpage><lpage>1517</lpage><pub-id pub-id-type="doi">10.1016/s0960-9822(03)00558-x</pub-id><pub-id pub-id-type="pmid">12956953</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ruan</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Fast and accurate long-read assembly with wtdbg2</article-title><source>Nature Methods</source><volume>17</volume><fpage>155</fpage><lpage>158</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0669-3</pub-id><pub-id pub-id-type="pmid">31819265</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmidbaur</surname><given-names>H</given-names></name><name><surname>Kawaguchi</surname><given-names>A</given-names></name><name><surname>Clarence</surname><given-names>T</given-names></name><name><surname>Fu</surname><given-names>X</given-names></name><name><surname>Hoang</surname><given-names>OP</given-names></name><name><surname>Zimmermann</surname><given-names>B</given-names></name><name><surname>Ritschard</surname><given-names>EA</given-names></name><name><surname>Weissenbacher</surname><given-names>A</given-names></name><name><surname>Foster</surname><given-names>JS</given-names></name><name><surname>Nyholm</surname><given-names>SV</given-names></name><name><surname>Bates</surname><given-names>PA</given-names></name><name><surname>Albertin</surname><given-names>CB</given-names></name><name><surname>Tanaka</surname><given-names>E</given-names></name><name><surname>Simakov</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Emergence of novel cephalopod gene regulation and expression through large-scale genome reorganization</article-title><source>Nature Communications</source><volume>13</volume><elocation-id>2172</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-022-29694-7</pub-id><pub-id pub-id-type="pmid">35449136</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Seah</surname><given-names>BKB</given-names></name><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Alkan</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>BleTIES: annotation of natural genome editing in ciliates using long read sequencing</article-title><source>Bioinformatics</source><volume>37</volume><fpage>3929</fpage><lpage>3931</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btab613</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sellis</surname><given-names>D</given-names></name><name><surname>Guérin</surname><given-names>F</given-names></name><name><surname>Arnaiz</surname><given-names>O</given-names></name><name><surname>Pett</surname><given-names>W</given-names></name><name><surname>Lerat</surname><given-names>E</given-names></name><name><surname>Boggetto</surname><given-names>N</given-names></name><name><surname>Krenek</surname><given-names>S</given-names></name><name><surname>Berendonk</surname><given-names>T</given-names></name><name><surname>Couloux</surname><given-names>A</given-names></name><name><surname>Aury</surname><given-names>JM</given-names></name><name><surname>Labadie</surname><given-names>K</given-names></name><name><surname>Malinsky</surname><given-names>S</given-names></name><name><surname>Bhullar</surname><given-names>S</given-names></name><name><surname>Meyer</surname><given-names>E</given-names></name><name><surname>Sperling</surname><given-names>L</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Duharcourt</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Massive colonization of protein-coding exons by selfish genetic elements in <italic>Paramecium</italic> germline genomes</article-title><source>PLOS Biology</source><volume>19</volume><elocation-id>e3001309</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.3001309</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sheng</surname><given-names>Y</given-names></name><name><surname>Duan</surname><given-names>L</given-names></name><name><surname>Cheng</surname><given-names>T</given-names></name><name><surname>Qiao</surname><given-names>Y</given-names></name><name><surname>Stover</surname><given-names>NA</given-names></name><name><surname>Gao</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The completed macronuclear genome of a model ciliate <italic>Tetrahymena thermophila</italic> and its application in genome scrambling and copy number analyses</article-title><source>Science China. Life Sciences</source><volume>63</volume><fpage>1534</fpage><lpage>1542</lpage><pub-id pub-id-type="doi">10.1007/s11427-020-1689-4</pub-id><pub-id pub-id-type="pmid">32297047</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sievers</surname><given-names>F</given-names></name><name><surname>Wilm</surname><given-names>A</given-names></name><name><surname>Dineen</surname><given-names>D</given-names></name><name><surname>Gibson</surname><given-names>TJ</given-names></name><name><surname>Karplus</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>W</given-names></name><name><surname>Lopez</surname><given-names>R</given-names></name><name><surname>McWilliam</surname><given-names>H</given-names></name><name><surname>Remmert</surname><given-names>M</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name><name><surname>Thompson</surname><given-names>JD</given-names></name><name><surname>Higgins</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Fast, scalable generation of high-quality protein multiple sequence alignments using Clustal Omega</article-title><source>Molecular Systems Biology</source><volume>7</volume><elocation-id>539</elocation-id><pub-id pub-id-type="doi">10.1038/msb.2011.75</pub-id><pub-id pub-id-type="pmid">21988835</pub-id></element-citation></ref><ref id="bib93"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simão</surname><given-names>FA</given-names></name><name><surname>Waterhouse</surname><given-names>RM</given-names></name><name><surname>Ioannidis</surname><given-names>P</given-names></name><name><surname>Kriventseva</surname><given-names>EV</given-names></name><name><surname>Zdobnov</surname><given-names>EM</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>BUSCO: assessing genome assembly and annotation completeness with single-copy orthologs</article-title><source>Bioinformatics</source><volume>31</volume><fpage>3210</fpage><lpage>3212</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id><pub-id pub-id-type="pmid">26059717</pub-id></element-citation></ref><ref id="bib94"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Slater</surname><given-names>GSC</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Automated generation of heuristics for biological sequence comparison</article-title><source>BMC Bioinformatics</source><volume>6</volume><elocation-id>1</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-6-31</pub-id><pub-id pub-id-type="pmid">15713233</pub-id></element-citation></ref><ref id="bib95"><element-citation publication-type="web"><person-group person-group-type="author"><name><surname>Smit</surname><given-names>AF</given-names></name><name><surname>Hubley</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>RepeatModeler Open-1.0</article-title><ext-link ext-link-type="uri" xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</ext-link><date-in-citation iso-8601-date="2020-09-23">September 23, 2020</date-in-citation></element-citation></ref><ref id="bib96"><element-citation publication-type="web"><person-group person-group-type="author"><name><surname>Smit</surname><given-names>AF</given-names></name><name><surname>Hubley</surname><given-names>R</given-names></name><name><surname>Green</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>RepeatMasker Open-4.0</article-title><ext-link ext-link-type="uri" xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</ext-link><date-in-citation iso-8601-date="2020-09-23">September 23, 2020</date-in-citation></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>JJ</given-names></name><name><surname>Baker</surname><given-names>C</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Amemiya</surname><given-names>CT</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Genetic consequences of programmed genome rearrangement</article-title><source>Current Biology</source><volume>22</volume><fpage>1524</fpage><lpage>1529</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2012.06.028</pub-id><pub-id pub-id-type="pmid">22818913</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>SA</given-names></name><name><surname>Maurer-Alcalá</surname><given-names>XX</given-names></name><name><surname>Yan</surname><given-names>Y</given-names></name><name><surname>Katz</surname><given-names>LA</given-names></name><name><surname>Santoferrara</surname><given-names>LF</given-names></name><name><surname>McManus</surname><given-names>GB</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Combined genome and transcriptome analyses of the ciliate schmidingerella arcuata (Spirotrichea) reveal patterns of DNA elimination, scrambling, and inversion</article-title><source>Genome Biology and Evolution</source><volume>12</volume><fpage>1616</fpage><lpage>1622</lpage><pub-id pub-id-type="doi">10.1093/gbe/evaa185</pub-id><pub-id pub-id-type="pmid">32870974</pub-id></element-citation></ref><ref id="bib99"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Speijer</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Making sense of scrambled genomes</article-title><source>Science</source><volume>319</volume><fpage>901</fpage><lpage>902</lpage><pub-id pub-id-type="doi">10.1126/science.319.5865.901a</pub-id><pub-id pub-id-type="pmid">18276871</pub-id></element-citation></ref><ref id="bib100"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Suyama</surname><given-names>M</given-names></name><name><surname>Torrents</surname><given-names>D</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>PAL2NAL: robust conversion of protein sequence alignments into the corresponding codon alignments</article-title><source>Nucleic Acids Research</source><volume>34</volume><fpage>W609</fpage><lpage>W612</lpage><pub-id pub-id-type="doi">10.1093/nar/gkl315</pub-id><pub-id pub-id-type="pmid">16845082</pub-id></element-citation></ref><ref id="bib101"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Swart</surname><given-names>EC</given-names></name><name><surname>Bracht</surname><given-names>JR</given-names></name><name><surname>Magrini</surname><given-names>V</given-names></name><name><surname>Minx</surname><given-names>P</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Zhou</surname><given-names>Y</given-names></name><name><surname>Khurana</surname><given-names>JS</given-names></name><name><surname>Goldman</surname><given-names>AD</given-names></name><name><surname>Nowacki</surname><given-names>M</given-names></name><name><surname>Schotanus</surname><given-names>K</given-names></name><name><surname>Jung</surname><given-names>S</given-names></name><name><surname>Fulton</surname><given-names>RS</given-names></name><name><surname>Ly</surname><given-names>A</given-names></name><name><surname>McGrath</surname><given-names>S</given-names></name><name><surname>Haub</surname><given-names>K</given-names></name><name><surname>Wiggins</surname><given-names>JL</given-names></name><name><surname>Storton</surname><given-names>D</given-names></name><name><surname>Matese</surname><given-names>JC</given-names></name><name><surname>Parsons</surname><given-names>L</given-names></name><name><surname>Chang</surname><given-names>WJ</given-names></name><name><surname>Bowen</surname><given-names>MS</given-names></name><name><surname>Stover</surname><given-names>NA</given-names></name><name><surname>Jones</surname><given-names>TA</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Herrick</surname><given-names>GA</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Mardis</surname><given-names>ER</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The <italic>Oxytricha trifallax</italic> macronuclear genome: a complex eukaryotic genome with 16,000 tiny chromosomes</article-title><source>PLOS Biology</source><volume>11</volume><elocation-id>e1001473</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1001473</pub-id><pub-id pub-id-type="pmid">23382650</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Syberg-Olsen</surname><given-names>MJ</given-names></name><name><surname>Irwin</surname><given-names>NAT</given-names></name><name><surname>Vannini</surname><given-names>C</given-names></name><name><surname>Erra</surname><given-names>F</given-names></name><name><surname>Di Giuseppe</surname><given-names>G</given-names></name><name><surname>Boscaro</surname><given-names>V</given-names></name><name><surname>Keeling</surname><given-names>PJ</given-names></name><name><surname>Schubert</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Biogeography and character evolution of the ciliate genus euplotes (spirotrichea, euplotia), with description of euplotes curdsi sp. nov</article-title><source>PLOS ONE</source><volume>11</volume><elocation-id>e0165442</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0165442</pub-id><pub-id pub-id-type="pmid">27828996</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname><given-names>M</given-names></name><name><surname>Brünen-Nieweler</surname><given-names>C</given-names></name><name><surname>Heckmann</surname><given-names>K</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Isolation of micronuclei from euplotes octocarinatus and identification of an internal eliminated sequence in the micronuclear gene encoding γ-tubulin 2</article-title><source>European Journal of Protistology</source><volume>35</volume><fpage>208</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1016/S0932-4739(99)80039-X</pub-id></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thomas</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="1971">1971</year><article-title>The genetic organization of chromosomes</article-title><source>Annual Review of Genetics</source><volume>5</volume><fpage>237</fpage><lpage>256</lpage><pub-id pub-id-type="doi">10.1146/annurev.ge.05.120171.001321</pub-id><pub-id pub-id-type="pmid">16097657</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vinogradov</surname><given-names>DV</given-names></name><name><surname>Tsoĭ</surname><given-names>OV</given-names></name><name><surname>Zaika</surname><given-names>AV</given-names></name><name><surname>Lobanov</surname><given-names>AV</given-names></name><name><surname>Turanov</surname><given-names>AA</given-names></name><name><surname>Gladyshev</surname><given-names>VN</given-names></name><name><surname>Gel’fand</surname><given-names>MS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Draft macronuclear genome of a ciliate <italic>Euplotes crassus</italic></article-title><source>Molekuliarnaia Biologiia</source><volume>46</volume><fpage>361</fpage><lpage>366</lpage><pub-id pub-id-type="doi">10.1134/S0026893312020197</pub-id><pub-id pub-id-type="pmid">22670532</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname><given-names>BJ</given-names></name><name><surname>Abeel</surname><given-names>T</given-names></name><name><surname>Shea</surname><given-names>T</given-names></name><name><surname>Priest</surname><given-names>M</given-names></name><name><surname>Abouelliel</surname><given-names>A</given-names></name><name><surname>Sakthikumar</surname><given-names>S</given-names></name><name><surname>Cuomo</surname><given-names>CA</given-names></name><name><surname>Zeng</surname><given-names>Q</given-names></name><name><surname>Wortman</surname><given-names>J</given-names></name><name><surname>Young</surname><given-names>SK</given-names></name><name><surname>Earl</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Pilon: an integrated tool for comprehensive microbial variant detection and genome assembly improvement</article-title><source>PLOS ONE</source><volume>9</volume><elocation-id>e112963</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0112963</pub-id><pub-id pub-id-type="pmid">25409509</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Zhi</surname><given-names>H</given-names></name><name><surname>Chai</surname><given-names>B</given-names></name><name><surname>Liang</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Cloning and sequence analysis of the micronuclear and macronuclear gene encoding Rab protein of <italic>Euplotes octocarinatus</italic></article-title><source>Bioscience, Biotechnology, and Biochemistry</source><volume>69</volume><fpage>649</fpage><lpage>652</lpage><pub-id pub-id-type="doi">10.1271/bbb.69.649</pub-id><pub-id pub-id-type="pmid">15785000</pub-id></element-citation></ref><ref id="bib108"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>R</given-names></name><name><surname>Xiong</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Miao</surname><given-names>W</given-names></name><name><surname>Liang</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>High frequency of +1 programmed ribosomal frameshifting in <italic>Euplotes octocarinatus</italic></article-title><source>Scientific Reports</source><volume>6</volume><elocation-id>21139</elocation-id><pub-id pub-id-type="doi">10.1038/srep21139</pub-id><pub-id pub-id-type="pmid">26891713</pub-id></element-citation></ref><ref id="bib109"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>JR</given-names></name><name><surname>Holt</surname><given-names>J</given-names></name><name><surname>McMillan</surname><given-names>L</given-names></name><name><surname>Jones</surname><given-names>CD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>FMLRC: hybrid long read error correction using an FM-index</article-title><source>BMC Bioinformatics</source><volume>19</volume><elocation-id>50</elocation-id><pub-id pub-id-type="doi">10.1186/s12859-018-2051-3</pub-id><pub-id pub-id-type="pmid">29426289</pub-id></element-citation></ref><ref id="bib110"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname><given-names>LC</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Evolution of programmed DNA rearrangements in a scrambled gene</article-title><source>Molecular Biology and Evolution</source><volume>23</volume><fpage>756</fpage><lpage>763</lpage><pub-id pub-id-type="doi">10.1093/molbev/msj089</pub-id><pub-id pub-id-type="pmid">16431850</pub-id></element-citation></ref><ref id="bib111"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yerlici</surname><given-names>VT</given-names></name><name><surname>Landweber</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Programmed genome rearrangements in the ciliate <italic>Oxytricha</italic></article-title><source>Microbiology Spectrum</source><volume>2</volume><fpage>2</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1128/microbiolspec.MDNA3-0025-2014</pub-id><pub-id pub-id-type="pmid">26104449</pub-id></element-citation></ref><ref id="bib112"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Doak</surname><given-names>TG</given-names></name><name><surname>Song</surname><given-names>W</given-names></name><name><surname>Yan</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>ADFinder: accurate detection of programmed DNA elimination using NGS high-throughput sequencing data</article-title><source>Bioinformatics</source><volume>36</volume><fpage>3632</fpage><lpage>3636</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btaa226</pub-id><pub-id pub-id-type="pmid">32246828</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82979.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group></front-stub><body><p>The study marks a significant advance in the field of evolutionary genomics of ciliates, an ancient and highly diverse eukaryotic phylum with many idiosyncrasies that teach us valuable lessons, inter alia, about sex and the plasticity of genomes. By focusing on two species from the same family, plus a more distant outgroup within the same class, this valuable study provides new and compelling information on evolutionary trends of genome rearrangement among different species of this interesting group of organisms. The work will be of interest to anyone interested in genome dynamics.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82979.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p>[Editors' note: this paper was reviewed by <ext-link ext-link-type="uri" xlink:href="https://www.reviewcommons.org/">Review Commons</ext-link>.]</p><p>Thank you for submitting your article &quot;Comparative genomics reveals insight into the evolutionary origin of massively scrambled genomes.&quot; for consideration by eLife. Your article has been reviewed by 2 peer reviewers at Review Commons, and the evaluation at eLife has been overseen by Detlef Weigel as Deputy Editor in discussion with several Senior and Reviewing Editors and an outside expert.</p><p>Based on your manuscript, the reviews and your responses, we invite you to submit a revised version incorporating the revisions as outlined in your response to the reviews.</p><p>When preparing your revisions, please also address the following points:</p><p>A major concern is that the genome assemblies are poor. Much of this is probably because of the technical challenges associated with this system, but BUSCO scores of 76% are and N50 &lt;30kb for the MICs are red flags. This does make one wonder how confident one can be that some of the conclusions, such as those pertaining to paralogs, are not at least in part due to assembly artifacts. It is striking that most genes undergoing unscrambling have no orthologs in others species, and that is thus unclear what these genes actually do. Please provide additional evidence that the conclusions are unaffected by the limitations of the assemblies. Please also provide more statistics regarding the assemblies (e.g., expected lengths distribution of nanochromosomes in MACs, number of chromosomes in MICs vs contigs, etc.), and annotations (gene families -- how similar are they between species, how many &quot;species-specific&quot; genes [which the genomics community considers to be largely, though not entirely artifactual] etc.).</p><p>Please spell out more clearly what you consider the evolutionary forces that have increased scrambling. My reading is that you consider many of the scrambled genes not to be functionally very important, which would be in line with there being so few orthologs in other species. To this end, please provide some basic information as to how many of the scrambled genes have expression support, what the distribution of expression quantiles is etc. -- this could help to shore up these inferences.</p><p>Finally, the manuscript is overly long, and despite -- or perhaps because -- of this, it is not presented in the most accessible manner. For example, the abstract still struggles to explain the importance or relevance of the work. e.g. the last sentence is &quot;Scrambled loci are more often associated with local duplications, supporting a simple model for the origin of scrambled genes via DNA duplication and decay&quot;. This would need some motivation, i.e., why should we care about scrambled genes per se? Similarly, why does the Introduction have an almost 1-page summary of the results?</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82979.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Reviewer #1 (Evidence, reproducibility and clarity (Required)):</p><p>Summary:</p><p>Ciliates extensively rearrange their somatic genome every time a new somatic nucleus develops from the zygotic germline nucleus. In this manuscript, Feng et al. report the sequencing, assembly and annotation of the germline and somatic genomes of Euplotes woodruffi and the germline genome of Tetmemena sp. (whose somatic genome was sequenced and assembled by the same lab in 2015). They present a comparative analysis of developmentally programmed genome rearrangements in these two species and in the model ciliate Oxytricha trifallax. Their major findings are that:</p><p>(i) E. woodruffi and Tetmemena sp. eliminate a smaller fraction of their germline genome (~54%) from their somatic macronucleus (MAC) than O. trifallax (&gt;80%)</p><p>(ii) Transposable elements (TE) represent a smaller fraction of the germline genome (~2%) in the first two ciliates than in O. trifallax (~15%). TEs are mainly located at the boundaries of germline chromosomes and in intergenic regions, but can also be found inside IESs</p><p>(iii) Several thousands of genes are scrambled in the germline genome of all three species</p><p>The authors have also addressed the possible origin of gene scrambling. They report an interesting association with local paralogy and propose a model for the emergence of the odd-even pattern of gene unscrambling between two paralogous copies.</p><p>Major comments:</p><p>1. Based on the statistics presented in Table 1, genome assemblies are of good quality, with a reasonable N50 size of germline (MIC) contigs. It seems, however, that no entire MIC chromosome could be assembled, since no two-telomere contig is mentioned in the list. As proposed by the authors (p.7) the presence of numerous TEs at the boundaries of MIC contigs (Figure S1) may have hindered the assembly of MIC chromosome ends. I would have appreciated to have more information on the &quot;other repeats&quot; (which seem to differ from tandem repeats according to Figure 2) and their location along MIC contigs.</p></disp-quote><p>Subcategories of “other repeats” were included in Table S2 based on Repeatmasker annotations. We now analyzed the locations of other repeats in MIC contigs and include those as well in new Figure S1B. About 30% of “other” transposable elements are present at the boundaries of MIC contigs, which may also hinder the assembly. Notably, 35-45% of “other TEs” are in assembled, intergenic regions.</p><disp-quote content-type="editor-comment"><p>2. The definition of &quot;Internal Eliminated Sequences&quot; (IES) is not clear. The authors make a distinction between IESs and TEs. I understand that IESs are DNA segments that separate two macronuclear-destined sequences (MDS) in the germline genome. Thus they appear to be restricted to those regions that eventually yield gene-sized MAC chromosomes. IESs are eliminated between two pointers that may not be identical on both sides in case of scrambled genes. Some clarification is needed here.</p><p>To illustrate my point: I found the statement &quot;with many TE insertions within IESs, suggesting that TE insertions may have generated IESs&quot; particularly confusing (p. 9 lines 5-6). Does this mean that IESs extend beyond the ends of inserted TEs? The legend of Figure S1 should also be clarified.</p></disp-quote><p>We clarified the text and legend. IESs can extend beyond the ends of inserted TEs, even if the original IES is a decayed TE, due to subsequent sequence evolution at the boundaries or if the original insertion was into an existing IES. David Prescott referred to sequence evolution at the edges of IESs as “pointer sliding” (ref.36).</p><disp-quote content-type="editor-comment"><p>3. p. 10 lines 2-4 and Figure S2: Could the authors explain the difference they make between MDS (in the text) and CDS (in Figure S2)? My understanding is that a CDS is the entire gene coding sequence and may be made of multiple MDSs. If this is correct, the sentence should read &quot;We compared the number of MDSs between single-copy orthologs for single-gene MAC chromosomes across the three species and found that the orthologs have similar CDS lengths&quot;.</p></disp-quote><p>Yes, we made the correction.</p><disp-quote content-type="editor-comment"><p>4. p. 12 lines 10-15: the discovery that paralogous MDSs can be found in scrambled genomic loci is interesting. If the two paralogs can be distinguished based on the number of substitutions, it would be informative to go back to individual reads and check whether each of the two copies can be incorporated in the unscrambled CDS (and at which frequency). Would the pointers be compatible with this?</p></disp-quote><p>The paralogous MDSs in the MIC are often not identical. The copy with the highest similarity is assigned as “preliminary match” by SDRAP (ref. 52), and others are assigned as “additional matches”. To validate SDRAP assignments, we did pairwise BLASTN alignments (“-task megablast”) of paralogous MIC MDSs and their corresponding MAC MDSs. We confirmed that in the three species, the preliminary match has the best or equally best pid (percentage of identity) in most cases. Therefore, the MDS assigned as preliminary match is more likely the paralog incorporated into the MAC chromosome.</p><p>We used genome assemblies of <italic>Euplotes woodruffi</italic>, which had the highest Nanopore coverage, to further investigate the frequency of MDS incorporation. We followed the reviewer’s suggestion and called SNP variants on both MAC and MIC genomes. For MAC SNP calling, we used Illumina reads as input for freebayes (ref a). For MIC SNP calling, we used Nanopore reads, instead of Illumina reads, to avoid non-specific short-read mapping on paralogous MDSs and to avoid the presence of any contaminating MAC reads. Variants were called and phased by PEPPER-Margin-DeepVariant (ref b), a new tool published in 2021 in Nature Methods, which has been reported to have similar accuracy to Illumina read variant calling, especially at high read coverage. We used the parameter “--pepper_min_coverage_threshold 20” to call confident variants when at least 20 reads cover the position. Only 92 MIC SNPs in the paralogous MDSs passed all filters of the program. Using this small set of MIC SNPs, we were unfortunately unable to distinguish which paralogous MIC MDS was incorporated into the MAC. Therefore, we cannot infer with what frequency one paralogous MDS is incorporated over another, until they become sufficiently diverged, which is compatible with the model.</p><list list-type="alpha-lower"><list-item><p>Garrison E, Marth G. Haplotype-based variant detection from short-read sequencing. arXiv preprint arXiv:1207.3907. 2012 Jul 17.</p></list-item><list-item><p>Shafin K, Pesout T, Chang PC, Nattestad M, Kolesnikov A, Goel S, Baid G, Kolmogorov M, Eizenga JM, Miga KH, Carnevali P. Haplotype-aware variant calling with PEPPER-Margin-DeepVariant enables high accuracy in nanopore long-reads. Nature methods. 2021 Nov;18(11):1322-32.</p></list-item></list><disp-quote content-type="editor-comment"><p>5. The hypothesis that odd-even scrambled loci have evolved from paralogous genes in E. woodruffi is supported by the existence of paralogous MDSs, length conservation of MDS/IES pairs and sequence similarity between corresponding MDS and IES in a pair. The correlations presented for Oxytricha and Tetmemena are much less convincing (Figure S5D and E). I recommend that the authors are even more cautious in their statement on p.13 (&quot;For Oxytricha and Tememena, the MDS and IES lengths for such MDS/IES pairs also correlate positively, but more moderately&quot;)</p></disp-quote><p>Thank you, we rephrased the text.</p><disp-quote content-type="editor-comment"><p>6. p. 15 last paragraph: Why did the authors focus only on TBEs inserted in non-scrambled IESs to look for orthologous TBE insertions? Is there a reason to believe that no recent TBE insertion occurred at other genomic loci? Or was it only for practical reasons? It is also not clear to me whether the authors have considered full-length TBEs or the presence of at least one TBE ORF.</p></disp-quote><p>This analysis was limited for practical reasons, because we identify position conservation of TBEs by aligning protein sequences of MAC genes. We only consider TBEs inserted in non-scrambled IESs in exons. It would be difficult and less meaningful to align completely non-coding MIC-limited regions.</p><p>Partial TBEs are also included if they contain at least one TBE ORF (detected by BLAST).</p><p>Furthermore, TE insertion cannot explain the origin of scrambled IESs, and TEs rarely map to scrambled IESs (Figure S1A), but there is a clear evolutionary model for the origin of nonscrambled IESs from decay of TBEs (ref. 49). Initial purifying selection would act on the TE to maintain its ability to self-excise, whereas we advocate for a different model for the origin of scrambled IESs by decay of paralogous MDSs.</p><disp-quote content-type="editor-comment"><p>7. p. 16: the authors report that some introns of E. woodruffi map &quot;near&quot; Oxytricha/Tetmemena pointers. How near? Based on the information provided by the authors, I don't think this observation necessarily implies that IESs were converted to introns (or reciprocally) during evolution. If this were true, shouldn't at least one intron boundary coincide exactly with a pointer? The authors should clarify this (also in the discussion, on p. 20, top paragraph).</p></disp-quote><p>We used a 20bp window (~7 amino acids), as described in the Methods, and added that to the Results. Full detail is provided in the Methods section, “Ortholog comparison pipeline and Monte Carlo simulations”. 103 <italic>E. woodruffi</italic> introns are within 20bp from the midpoint of <italic>Oxytricha/Tetmemena</italic> pointers. Among these, 43 intron boundaries overlap an <italic>Oxytricha</italic> or <italic>Tetmemena</italic> pointer. We observed 306 cases of precisely matching boundaries between any two species, where the exon junction of one species maps inside the MDS/IES pointer of another species, although we would only expect the boundaries of introns and IESs to coincide so precisely if they were recent conversions. Hence we feel that a window analysis is informative.</p><disp-quote content-type="editor-comment"><p>8. p. 19 2nd paragraph: the suggested mechanism explaining the 5' bias of IESs in E. woodruffi genes is unclear. How could germline recombination take place between a MIC chromosome and a MAC reverse transcript or nanochromosome? This would imply that DNA could be imported in the MIC. Is there evidence that this might occur?</p></disp-quote><p>The ability of TEs to invade the MIC demonstrates that even foreign DNA can be incorporated into the MIC. Since MAC DNA is present at high copy number, it offers a potential source for a recombination template that could erase IESs, as could an errant reverse transcript of one of the long noncoding template RNAs. Any of these would be infrequent events that would matter on an evolutionary time scale even if developmentally rare.</p><disp-quote content-type="editor-comment"><p>9. According to Figure 1, no scrambled genes have been reported in Paramecium tetraurelia. Within the frame of the proposed model, this is somewhat unexpected because this ciliate went through several whole genome duplications during evolution and harbors many paralogous gene pairs. Is there a reason why no gene scrambling took place in Paramecium?</p></disp-quote><p><italic>Paramecium</italic> uses only TA dinucleotide pointers for IES elimination, unlike the rich diversity of pointers in spirotrichous ciliates. This limitation in its machinery may explain why no scrambled loci have been observed in <italic>Paramecium</italic>, despite the abundance of paralogs. Our model suggests that local MIC paralogy is associated with the origin of scrambling. But most of the paralogy reported in <italic>Paramecium</italic> is at the level of whole chromosomes in the MAC (ref. 104) rather than local MIC paralogy.</p><disp-quote content-type="editor-comment"><p>Minor comments:</p><p>p. 4 (4th bottom line): To my knowledge, ref #28 presents a draft (incomplete) MIC assembly of the Paramecium genome.</p></disp-quote><p>Thank you, we added reference 29 and adjusted the wording describing the quality of MIC genome draft assemblies.</p><disp-quote content-type="editor-comment"><p>p. 7 (last paragraph): &quot;encoding&quot; should be replaced by &quot;carrying&quot;</p></disp-quote><p>Thank you, we made the change.</p><disp-quote content-type="editor-comment"><p>p. 10 (2nd paragraph): insert a missing &quot;o&quot; into &quot;nanochromosomes&quot;</p></disp-quote><p>Thank you, corrected.</p><disp-quote content-type="editor-comment"><p>p. 10 (same paragraph): the weak 5' bias of IES distribution in Tetmemena should be shown (either as an additional panel in Figure 3 or in a Sup Figure).</p></disp-quote><p>Thank you, we added it as Figure S2C.</p><disp-quote content-type="editor-comment"><p>p. 24 2nd paragraph: &quot;a&quot; is missing in &quot;Trinity, which is a software…&quot;</p></disp-quote><p>Thank you, we made the correction.</p><disp-quote content-type="editor-comment"><p>Cross-Consultation Comments</p><p>I agree with most comments of reviewer 3.</p><p>The authors have actually defined &quot;TE&quot; in the introduction (p. 6). Depending on the journal's rules for abbreviation use, it may not be necessary to define it again in the Results section</p><p>Reviewer #1 (Significance (Required)):</p><p>Ciliates are unicellular models to study developmentally programmed genome rearrangements at the mechanistic, genome-wide and evolutionary levels. These aspects have so far mostly been addressed in three species: <italic>P. tetraurelia</italic> and <italic>Tetrahymena thermophila</italic> on the one hand, the spirotrichous ciliate O. trifallax on the other.</p><p>One new piece of information that can be found in the present manuscript is the assembly and annotation of the germline genome of two novel species: Tetmemena sp, closely related to Oxytricha, and the more distant E. woodruffi. Feng et al. establish that, similar to other ciliates, Tetmemena and Euplotes eliminate TEs and other germline-specific sequences during programmed genome rearrangements. They also undergo extensive gene unscrambling, which results in IES removal and MDS reordering to assemble coding sequences.</p><p>A TE origin was discussed previously for Paramecium (Arnaiz et al. PLoS Genet; Sellis et al. 2021 PLoS Biol) and Tetrahymena IESs (Hamilton et al. 2016 eLife). While this may also hold true in spirotrichous ciliatesThe present manuscript proposes a completely new evolutionary scenario for IESs from scrambled genes. Here, Feng et al. establish that scrambled genes of spirotrichous ciliates tend to be associated with local paralogy. They provide evidence supporting that IESs from scrambled genes may have evolved from paralogous MDSs.</p><p>Although I am more an expert in the molecular mechanisms involved in genome rearrangements, I feel that the work reported here should draw the attention of a broader audience interested in genome dynamics and evolution, beyond the specific field of spirotrichous ciliate biology.</p><p>Reviewer #3 (Evidence, reproducibility and clarity (Required)):</p><p>Feng et al. provide a solid analysis of the evolution of genome rearrangement in spirotrich ciliates. The authors applied a variety of state-of-the-art sequencing and bioinformatic methods to investigate the intriguing and extremely complex patterns of genome architecture in this protist lineage. Methods (including statistical analyses) are adequate and explained in detail. Results and discussions reflect careful, clever analysis of the data and excellent linkage with the literature. Figures and tables complement the text in a compelling way. I have only minor suggestions:</p><p>Summary: more gradually introduce Spirotrichea and the phylogenetic relationship among the three species analyzed. This would better position the reader to understand the evolutionary context you are working in. Also, it would be helpful to more clearly differentiate novel vs. existing data. A suggestion: &quot;This study focuses on three spirotrich species: two in the family Oxytrichidae (Oxytricha trifallax and Tetmemena sp) and Euplotes woodruffi as an outgroup. To complement existing data, we sequenced, assembled and annotated the germiline and somatic genomes of E. woodruffi and the germline genome of Tetmemena sp.&quot;</p></disp-quote><p>Thank you, we clarified the summary (abstract).</p><disp-quote content-type="editor-comment"><p>Introduction, first paragraph: Replace &quot;The species in this study…&quot; for a more precise statement, such as &quot;The three spirotrich species studied here…&quot;</p></disp-quote><p>Thank you, we have made this statement more precise.</p><disp-quote content-type="editor-comment"><p>p. 4: This sentence is unclear: &quot;These useful tools provide partial insight to guide selection of species for full genome sequencing, which allows construction of complete rearrangement maps of a MIC genome onto a MAC genome for a reference species.&quot;</p></disp-quote><p>Thank you, we have clarified this sentence.</p><disp-quote content-type="editor-comment"><p>p. 8: define TE on first mention.</p></disp-quote><p>Defined on page 6.</p><disp-quote content-type="editor-comment"><p>Table 1. Indicate which MIC and MAC data are from this study.</p></disp-quote><p>References are included for published data and a note has been added to indicate data from this study.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Significance (Required)):</p><p>The present work represents a significant advance in the field of evolutionary genomics. The focus of the paper is on ciliates, an ancient (2 billion-year old) and highly diverse eukaryotic phylum that presents many peculiarities, including sex, nuclear dimorphism, genome rearrangement, high numbers of paralogs and transposons, etc. While some data exist on a few model ciliates of disparate phylogenetic position, this work focuses on two species taxonomically placed in the same family, plus a more distant outgroup within the same class. This gives a novel dimension to this study, that goes beyond exploring genome architecture in a single clade. Instead, it allows to explore evolutionary trends in genome rearrangement among relatively closely related species. This paper should be of high interest not only for ciliate biologists (like me), but also in relation to comparative genomics of protists/eukaryotes and germ-soma biology. I highly recommend publication.</p></disp-quote><p>[Editors' note: further revisions were suggested prior to acceptance, as described below.]</p><disp-quote content-type="editor-comment"><p>A major concern is that the genome assemblies are poor. Much of this is probably because of the technical challenges associated with this system, but BUSCO scores of 76% are and N50 &lt;30kb for the MICs are red flags. This does make one wonder how confident one can be that some of the conclusions, such as those pertaining to paralogs, are not at least in part due to assembly artifacts. It is striking that most genes undergoing unscrambling have no orthologs in others species, and that is thus unclear what these genes actually do. Please provide additional evidence that the conclusions are unaffected by the limitations of the assemblies. Please also provide more statistics regarding the assemblies (e.g., expected lengths distribution of nanochromosomes in MACs, number of chromosomes in MICs vs contigs, etc.), and annotations (gene families -- how similar are they between species, how many &quot;species-specific&quot; genes [which the genomics community considers to be largely, though not entirely artifactual] etc.).</p></disp-quote><p>In this study, we compared both somatic MAC and germline MIC genomes in the three ciliate species. We would like to explain why we think all the genome assemblies are sufficiently high quality for our analysis.</p><p>The MAC genome of <italic>Euplotes woodruffi</italic> was newly assembled in this study, following the similar assembly pipeline for <italic>Oxytricha trifallax</italic> (6) and <italic>Tetmemena sp.</italic> (7). The BUSCO score of 76% that we reported in our submitted manuscript was assessed using BUSCO v3 and lineage dataset eukaryota_odb9. Fortunately, BUSCO recently updated their database (88) to sample more species including ciliates (see below). Using the latest BUSCO v5.4.3 and lineage dataset alveolata_odb10, we now find that the MAC genome of <italic>E. woodruffi</italic> has a BUSCO score of 88.8%, the highest for any published <italic>Euplotes</italic> genome. Therefore, we are confident in the quality of our MAC genome assembly.</p><p>As the authors of BUSCO describe (87), the BUSCO score is just one metric of assessing genome completeness and is strongly influenced by (1) the gene prediction model (the nonmodel ciliate <italic>Euplotes</italic> uses a different genetic code, UGA=Cys, and has fewer curated genes compared to other well-established model ciliates; therefore it has a less mature gene prediction model) and (2) the genetic distance from the sampled species in the BUSCO dataset (e.g. the <italic>C. elegans</italic> genome only contains 90% of BUSCO genes, as shown in ref. 87). In our case, the missing detection of BUSCO genes is more likely the result of the evolutionary distance between ciliates and the species used in the previous BUSCO dataset.</p><p>We also assessed other ciliate genomes for comparison (see Figure 1, below). The MAC genomes of four reference hypotrich ciliates (<italic>Tetmemena sp., Laurentiella sp., Paraurostyla sp.</italic> and <italic>Urostyla sp.,</italic> ref. 7), which were previously assessed as &quot;complete&quot; using core eukaryotic genes (CEG), have current BUSCO scores ranging from 88.3% to 94.1% (Figure 1). <italic>Urostyla sp.</italic>, which is the most diverged from <italic>Oxytricha trifallax</italic> (which is included in the current BUSCO reference dataset), has the lowest BUSCO score among those, consistent with the greater evolutionary distance, and does not reflect a less complete genome. Note that some ciliate genomes have BUSCO scores near 100% because they were included in the BUSCO dataset. Since no <italic>Euplotes</italic> species were sampled in the BUSCO dataset, we conclude that 88.8% is within the range of a high-quality ciliate MAC genome assembly.</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><caption><title>BUSCO assessments of MAC genomes in representative ciliates.</title><p>The three species in the present manuscript are shown in bold. Species with a * are expected to have high BUSCO scores because they were sampled in the BUSCO reference dataset (alveolata_odb10). MAC genomes included here: Euplotes octocarinatus (8), Euplotes vannus (9), Laurentiella sp., Paraurostyla sp., Urostyla sp. (shown to be complete based on the presence of core eukaryotic genes, 7), Halteria grandinella (Zheng et al., Genbank RRYP01000000; https://journals.asm.org/doi/10.1128/mBio.01964-20), Paramecium tetraurelia (https://paramecium.i2bc.parissaclay.fr/download/Paramecium/tetraurelia/51/annotations/ptetraurelia_mac_51/), <italic>Tetrahymena thermophila</italic> (http://ciliate.org/index.php/home/downloads).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-sa2-fig1-v2.tif"/></fig><p>We also provided in the Methods section two other commonly used metrics to assess MAC genome completeness, as in Swart et al (6) and Chen et al. (7): 80.6% of <italic>E. woodruffi</italic> MAC contigs contain at least one telomere. Furthermore, the <italic>E. woodruffi</italic> MAC genome contains tRNAs for all 20 amino acids. Both assessments support the conclusion that the <italic>E. woodruffi</italic> MAC genome assembly is of high quality.</p><p>The short N50 of MAC genomes is expected for ciliate MAC genomes with gene-sized chromosomes. We added Figure 2 - figure supplement 2 to the manuscript to show the length distribution of MAC nanochromosomes (telomere-to-telomere assembled contigs). This is itself a slight underestimate of the expected MAC length distribution, because the longest MAC chromosomes have a lower probability of complete assembly. It was also expected that <italic>Euplotes</italic> species would have shorter MAC nanochromosomes, compared to <italic>Oxytricha</italic> and <italic>Tetmemena</italic>, based on prior gel electrophoresis (Swanton, Greslin, and Prescott. 1980. <italic>Chromosoma</italic> 77:203215), which reported that the distribution of MAC DNA molecules in <italic>Euplotes aediculatus</italic> appeared to migrate faster than other hypotrichs (see below, Figure 2)<italic>.</italic></p><fig id="sa2fig2" position="float"><label>Author response image 2.</label><caption><title>Agarose gel electrophoresis visualizing the size distribution of MAC DNA from Euplotes aediculatus, Stylonychia pustulata (also known as Tetmemena pustulata) and Oxytricha nova.</title><p>Figure 8 in ref. 13 and Swanton, Greslin, and Prescott. 1980. Chromosoma 77:203-215.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82979-sa2-fig2-v2.tif"/></fig><p>MIC genomes, on the other hand, are more challenging to assemble because of the high content of repetitive elements that hinder assembly (Figure 2 - figure supplement 1). In this study, the major use of the MIC genome assembly is to infer MDS and IES annotations. We report that 90% of the MAC chromosomes are well covered (&gt;90%) in the MIC genome (reported in Results). This level of completeness is comparable to that of the <italic>Oxytricha trifallax</italic> reference genome (1), and therefore is sufficient to permit comparison of rearrangement maps, including scrambled loci.</p><p>The gene annotation and ortholog analysis in this study is only performed on MAC genes. The MIC genome is not used for gene or paralog annotation because coding regions are contained in MDSs that are interrupted by IESs and/or scrambled. Therefore, the fragmentation of MIC contigs does not influence the quality of gene annotation or paralog/ortholog assignment. Paralogs were carefully analyzed to include only genes found on telomere-terminating contigs (G<sub>4</sub>T<sub>4</sub>G<sub>4</sub> or C<sub>4</sub>A<sub>4</sub>C<sub>4</sub>), to be confident that they represent MAC chromosomes. Furthermore, the MAC genome assemblies were clustered to a level of sequence similarity of 95% in order to collapse and therefore exclude allelic differences. For the purpose of studying paralogous MDSs in the MIC genome, we only included paralogous MDSs that map to the same MIC contig. This excludes the possibility of conflating MDSs that map to different contigs. Therefore, the higher levels of paralogy that we report for scrambled genes (both orthogroup size, which entirely derives from the MAC assembly, and numbers of paralogous MDSs identified on MIC contigs) are not likely due to assembly artifacts but meet a stringent quality standard.</p><p>Details of gene families identified by OrthoFinder for scrambled and nonscrambled genes were provided in Supplementary File 4. We also provide a new Supplementary File 5 to summarize gene families detected by OrthoFinder for each pair of species. Gene annotations for the three species have been uploaded to https://knot.math.usf.edu//mds_ies_db/2022/downloads.html. Supplementary File 4 and 5 also provide an estimate of &quot;species-specific&quot; genes. We do not observe more species-specific genes among scrambled vs. nonscrambled loci (see the response to point 2 below).</p><p>We do not presently report the number of MIC chromosomes in any species, but this will be possible for <italic>Oxytricha trifallax</italic> from a new study based on Hi-C from another first author. Chen et al. (1) reported an estimate based on collapsing MIC-telomere-containing PacBio reads, but we find that Hi-C will provide a more accurate estimate.</p><disp-quote content-type="editor-comment"><p>Please spell out more clearly what you consider the evolutionary forces that have increased scrambling. My reading is that you consider many of the scrambled genes not to be functionally very important, which would be in line with there being so few orthologs in other species. To this end, please provide some basic information as to how many of the scrambled genes have expression support, what the distribution of expression quantiles is etc. -- this could help to shore up these inferences.</p></disp-quote><p>We do not suggest that scrambled genes are less important than nonscrambled ones. Supplementary File 4 shows that a similar portion of scrambled vs. nonscrambled genes have orthologs in other ciliates. The lack of orthologs in other ciliates may be due to less investigation of closely related species. This could explain why <italic>Oxytricha</italic> genes, whether scrambled or nonscrambled, have more orthologs detected in other ciliates compared to <italic>Euplotes</italic>.</p><p>Our model suggests that local duplication provides a buffer against mutations, allowing paralogous MDSs to repair the MAC locus during assembly of the scrambled genes. The increased levels of scrambling in the <italic>Oxytricha</italic> lineage do not need to invoke a fitness advantage, however. It is more likely that a neutral ratchet drives the increase in scrambling, because of the difficulty of &quot;erasing&quot; it from the germline MIC genome once a scrambled architecture has been established, relative to the trend towards shortening MDSs as more mutations accumulate in either paralog, which leads to increased levels of fragmentation (68, 69). We add this to the discussion.</p><p>We followed the editor’s suggestion to analyze gene expression data for all species. We collected poly-A enriched mRNA and sequenced three replicates from asexually growing vegetative cells. We find that scrambled and nonscrambled genes have nearly identical levels of expression support (at least one read in all 3 replicates) in both <italic>Oxytricha</italic> (Supplementary File 8) and <italic>Tetmemena</italic>. <italic>E. woodruffi</italic> has more expression support for nonscrambled vs. scrambled genes, on the other hand (Supplementary File 8), which could be due to the more recent acquisition of scrambled loci in <italic>E. woodruffi</italic>, based on the stronger correlations observed between MDS and IES length and sequence similarity for MDS/IES pairs flanked by the same pointers (Figure 4 - figure supplement 2). Thus, it is possible that their nonscrambled paralogs may still serve the major function.</p><p>We also compared the expression levels of scrambled vs. nonscrambled genes. Only genes with a low Coefficient of Variation (CV&lt;1) of TPM (transcripts per million) are included in the comparison (new Figure 4 - figure supplement 3). The distribution of expression levels is similar for scrambled vs. nonscrambled genes (Figure 4 - figure supplement 3). In a Mann-Whitney U test, the average expression level among three replicates is significantly higher in nonscrambled genes for <italic>Oxytricha</italic> and <italic>E. woodruffi</italic>, but not significant for <italic>Tetmemena</italic>.</p><disp-quote content-type="editor-comment"><p>Finally, the manuscript is overly long, and despite -- or perhaps because -- of this, it is not presented in the most accessible manner. For example, the abstract still struggles to explain the importance or relevance of the work. e.g. the last sentence is &quot;Scrambled loci are more often associated with local duplications, supporting a simple model for the origin of scrambled genes via DNA duplication and decay&quot;. This would need some motivation, i.e., why should we care about scrambled genes per se? Similarly, why does the Introduction have an almost 1-page summary of the results?</p></disp-quote><p>We provide additional motivation for the study of scrambled genomes in the beginning of the abstract and the introduction. <italic>Oxytricha</italic>'s scrambled genome and others in its lineage represent some of the most complex genome architectures known in <italic>any organism</italic>, with hundreds of thousands of precise, programmed genome editing events required to assemble coding regions. Recent exciting reports have described scrambled genomes in metazoa, including cephalopods, but those evolutionary events entail only shuffling of gene order, with no accompanying genome editing, so they are less complex, and other examples of programmed genome rearrangement in both protists and metazoa are usually simple DNA deletions.</p><p>We shortened other portions of the manuscript, eliminating or combining paragraphs where feasible, and added more motivation throughout, particularly emphasizing the compelling ways in which the <italic>Oxytricha</italic> lineage presents the most complex natural genome editing pathway known to date. Hence it is important to ask how such complex genomes have arisen, and comparative genomics provides fundamental insight into this question.</p></body></sub-article></article>