<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">88895</article-id><article-id pub-id-type="doi">10.7554/eLife.88895</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.88895.3</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Chromosomes and Gene Expression</subject></subj-group></article-categories><title-group><article-title>Transcriptional immune suppression and up-regulation of double-stranded DNA damage and repair repertoires in ecDNA-containing tumors</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-315976"><name><surname>Lin</surname><given-names>Miin S</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-2017-4246</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-316941"><name><surname>Jo</surname><given-names>Se-Young</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund10"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-316942"><name><surname>Luebeck</surname><given-names>Jens</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-4550"><name><surname>Chang</surname><given-names>Howard Y</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9459-4393</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf3"/></contrib><contrib contrib-type="author" id="author-288303"><name><surname>Wu</surname><given-names>Sihan</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-8329-7492</contrib-id><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="other" rid="fund7"/><xref ref-type="other" rid="fund8"/><xref ref-type="other" rid="fund9"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf4"/></contrib><contrib contrib-type="author" corresp="yes" id="author-256118"><name><surname>Mischel</surname><given-names>Paul S</given-names></name><email>pmischel@stanford.edu</email><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf5"/></contrib><contrib contrib-type="author" corresp="yes" id="author-316943"><name><surname>Bafna</surname><given-names>Vineet</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5810-6241</contrib-id><email>vbafna@ucsd.edu</email><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff10">10</xref><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf6"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0168r3w48</institution-id><institution>Bioinformatics and Systems Biology Graduate Program, University of California, San Diego</institution></institution-wrap><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01wjejq96</institution-id><institution>Department of Biomedical Systems Informatics and Brain Korea 21 PLUS Project for Medical Science, Yonsei University College of Medicine</institution></institution-wrap><addr-line><named-content content-type="city">Seoul</named-content></addr-line><country>Republic of Korea</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0168r3w48</institution-id><institution>Department of Computer Science and Engineering, University of California, San Diego</institution></institution-wrap><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Center for Personal Dynamic Regulomes, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Genetics, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff6"><label>6</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Howard Hughes Medical Institute, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff7"><label>7</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05byvp690</institution-id><institution>Children’s Medical Center Research Institute, University of Texas Southwestern Medical Center</institution></institution-wrap><addr-line><named-content content-type="city">Dallas</named-content></addr-line><country>United States</country></aff><aff id="aff8"><label>8</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Sarafan Chemistry, Engineering, and Medicine for Human Health (Sarafan ChEM-H), Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff9"><label>9</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Pathology, Stanford University School of Medicine</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff10"><label>10</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0168r3w48</institution-id><institution>Halıcıoğlu Data Science Institute, University of California, San Diego</institution></institution-wrap><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Gingeras</surname><given-names>Thomas R</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02qz8b764</institution-id><institution>Cold Spring Harbor Laboratory</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>White</surname><given-names>Richard M</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01e473h50</institution-id><institution>Ludwig Institute for Cancer Research, University of Oxford</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>19</day><month>06</month><year>2024</year></pub-date><volume>12</volume><elocation-id>RP88895</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-06-13"><day>13</day><month>06</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-04-24"><day>24</day><month>04</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.04.24.537925"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-08-10"><day>10</day><month>08</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88895.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-02-27"><day>27</day><month>02</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88895.2"/></event></pub-history><permissions><copyright-statement>© 2023, Lin et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Lin et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-88895-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-88895-figures-v1.pdf"/><abstract><p>Extrachromosomal DNA is a common cause of oncogene amplification in cancer. The non-chromosomal inheritance of ecDNA enables tumors to rapidly evolve, contributing to treatment resistance and poor outcome for patients. The transcriptional context in which ecDNAs arise and progress, including chromosomally-driven transcription, is incompletely understood. We examined gene expression patterns of 870 tumors of varied histological types, to identify transcriptional correlates of ecDNA. Here, we show that ecDNA-containing tumors impact four major biological processes. Specifically, ecDNA-containing tumors up-regulate DNA damage and repair, cell cycle control, and mitotic processes, but down-regulate global immune regulation pathways. Taken together, these results suggest profound alterations in gene regulation in ecDNA-containing tumors, shedding light on molecular processes that give rise to their development and progression.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>ecDNA</kwd><kwd>extrachromosomal DNA</kwd><kwd>cancer</kwd><kwd>transcriptomics</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100000289</institution-id><institution>Cancer Research UK</institution></institution-wrap></funding-source><award-id>CGCATF-2021/100012</award-id><principal-award-recipient><name><surname>Chang</surname><given-names>Howard Y</given-names></name><name><surname>Mischel</surname><given-names>Paul S</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100000289</institution-id><institution>Cancer Research UK</institution></institution-wrap></funding-source><award-id>CGCATF-2021/100025</award-id><principal-award-recipient><name><surname>Bafna</surname><given-names>Vineet</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>OT2CA278635</award-id><principal-award-recipient><name><surname>Bafna</surname><given-names>Vineet</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>OT2CA278688</award-id><principal-award-recipient><name><surname>Chang</surname><given-names>Howard Y</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01GM114362</award-id><principal-award-recipient><name><surname>Lin</surname><given-names>Miin S</given-names></name><name><surname>Luebeck</surname><given-names>Jens</given-names></name><name><surname>Bafna</surname><given-names>Vineet</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>U24CA264379</award-id><principal-award-recipient><name><surname>Luebeck</surname><given-names>Jens</given-names></name><name><surname>Bafna</surname><given-names>Vineet</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100000289</institution-id><institution>Cancer Research UK</institution></institution-wrap></funding-source><award-id>CGCATF-2021/100023</award-id><principal-award-recipient><name><surname>Wu</surname><given-names>Sihan</given-names></name></principal-award-recipient></award-group><award-group id="fund8"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>OT2CA278683</award-id><principal-award-recipient><name><surname>Wu</surname><given-names>Sihan</given-names></name></principal-award-recipient></award-group><award-group id="fund9"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100004917</institution-id><institution>Cancer Prevention and Research Institute of Texas</institution></institution-wrap></funding-source><award-id>RR210034</award-id><principal-award-recipient><name><surname>Wu</surname><given-names>Sihan</given-names></name></principal-award-recipient></award-group><award-group id="fund10"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100003710</institution-id><institution>Korea Health Industry Development Institute</institution></institution-wrap></funding-source><award-id>HI19C1330</award-id><principal-award-recipient><name><surname>Jo</surname><given-names>Se-Young</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Extrachromosomal DNA containing tumors up-regulate DNA damage and repair, cell cycle control, and mitotic processes, but down-regulate global immune regulation pathways, suggesting new avenues for therapeutic intervention.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Extrachromosomal DNA (ecDNA) are large, functional, circular double-stranded DNA molecules that are enriched for oncogenes, highly amplified, and frequently observed in a wide variety of cancer types (<xref ref-type="bibr" rid="bib57">Turner et al., 2017</xref>; <xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>). ecDNAs lack centromeres and are asymmetrically segregated into daughter cells during cell division, driving intratumoral genetic heterogeneity, accelerated evolution, and rapid treatment resistance (<xref ref-type="bibr" rid="bib43">Nathanson et al., 2014</xref>; <xref ref-type="bibr" rid="bib31">Lange et al., 2022</xref>). Further, recent studies demonstrate strong positive selection for ecDNA during tumor progression (<xref ref-type="bibr" rid="bib38">Luebeck et al., 2023</xref>). ecDNAs also exhibit highly accessible chromatin and altered cis- and trans- regulation, including cooperative intramolecular interactions (<xref ref-type="bibr" rid="bib26">Hung et al., 2021</xref>), promoting elevated expression of oncogenic transcriptional programs (<xref ref-type="bibr" rid="bib62">Wu et al., 2019</xref>; <xref ref-type="bibr" rid="bib42">Morton et al., 2019</xref>; <xref ref-type="bibr" rid="bib59">van Leen et al., 2022</xref>), further contributing to poor outcome for patients (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>).</p><p>The recent development of computational tools that enable the detection of ecDNA from whole genome sequencing data, has facilitated analyses of well-curated, publicly available datasets, including The Cancer Genome Atlas (TCGA), thereby providing an important opportunity to identify transcriptional repertoires that are preferentially detected in bona fide, clinical ecDNA-containing tumors. To shed new light on the gene expression patterns that may enhance ecDNA development and progression, we examined a global transcriptional analysis of ecDNA-containing tumors.</p></sec><sec id="s2" sec-type="results"><title>Results</title><p>A recent analysis utilized the tools AmpliconArchitect and AmpliconClassifier on 1921 tumors from TCGA to suggest that ecDNA prevalence ranges from 0–59.6% across multiple tumor tissue subtypes (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>). Using AmpliconClassifier (AC), the analysis classified tumor samples into five subtypes: ecDNA(+), Breakage Fusion Bridge (BFB), complex non-cyclic, linear, and no-amplification. However, due to limitations imposed by short-read sequencing, AC may classify some ecDNA(+) structures as complex non-cyclic when breakpoints are missed. Second, BFB cycles can give rise to ecDNA formation, making discernment of the two modes of amplification difficult. To limit false-negative ecDNA classifications in the ecDNA(-) set, we treated samples with only a linear or no-amplification status as ecDNA(-), removing complex non-cyclic and BFB(+) samples from the analysis. In order to understand the transcriptional programs active in maintaining ecDNA, we selected 870 samples from 14 tumor types with at least three ecDNA(+) samples each, and compared the gene expression data of the resulting 234 ecDNA(+) and 636 ecDNA(-) samples (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1B</xref>).</p><sec id="s2-1"><title>Machine learning identifies candidate genes for ecDNA maintenance</title><p>In lieu of identifying genes that are highly differentially expressed between ecDNA(+) and ecDNA(-) samples but driven by a small subset of cases (e.g. gene A in <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), we sought to identify genes (e.g. gene B) whose expression level was predictive of ecDNA presence. We assumed that genes that were persistently over-expressed or under-expressed in ecDNA(+) samples relative to ecDNA(-) samples were more likely to be involved in ecDNA biogenesis or maintenance, or in mediating the cellular response to the presence of ecDNA.</p><p>To identify a minimal set of genes whose expression values were consistently predictive of ecDNA presence, we used Boruta, (<xref ref-type="bibr" rid="bib30">Kursa and Rudnicki, 2010</xref>) an automated feature selection algorithm (<xref ref-type="fig" rid="fig1">Figure 1A</xref> and Methods). Given the unequal representation of ecDNA(+) and ecDNA(-) samples within each of the 14 tumor types, we performed Boruta on 200 datasets, each consisting of a random selection of 80% of the 870 samples (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), and chose the criterion of a gene being labeled as a Boruta gene in at least 10 of the 200 trials to be selected for downstream analysis. The Boruta analysis identified a set of 408 genes with persistent differential expression, hereafter denoted as the Core gene set.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Genes predictive of extrachromosomal DNA (ecDNA) status.</title><p>(<bold>A</bold>) The feature selection algorithm, Boruta, was applied to 200 datasets of randomly selected subsets consisting of 80% of all samples. Genes selected by Boruta in at least 10 of the 200 trials were identified as the Core set of genes (408) that were predictive of ecDNA presence. (<bold>B</bold>) Identification of highly co-expressed and stable gene clusters using pvclust expanded the Core set by an additional 235 genes to the final list of 643 CorEx genes. (<bold>C</bold>) Out of 354 clusters, the majority (344) of clusters contained one or two Core genes. (<bold>D</bold>) Most clusters were small, with only seven clusters containing more than 10 genes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Cartoon illustration of the rationale for Boruta analysis.</title><p>A differentially expressed gene (e.g. gene A) may have a higher log-fold change (LFC) value than that of a CorEx gene (e.g. gene B), but may not be persistently over- or under-expressed in ecDNA(+) samples. Therefore, no expression value cut-off can be used to separate ecDNA(+) samples from ecDNA(-) samples using gene A expression. In contrast, gene B expression can separate the two classes perfectly.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-figsupp1-v1.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Extended co-expressed cluster characterization.</title><p>Pvclust identified 354 highly co-expressed clusters where members are selected in at least 10 out of 200 Boruta trials as a cluster. Of the 354 clusters, two did not contain any Core genes, 281 only contained Core genes, and 71 contained at least one Core and one co-expressed gene.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-figsupp2-v1.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Gene expression of Core versus co-expressed genes in clusters.</title><p>Cluster #74 contains a Core gene, <italic>RAE1</italic>, and a co-expressed gene, <italic>CSTF1</italic>. Cluster #3 contains nine Core genes and 12 co-expressed genes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-figsupp3-v1.tif"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Impact of tumor purity on CorEx gene expression.</title><p>Of the 870 TCGA samples (234 ecDNA(+), 636 ecDNA(-)), 701 samples (174 ecDNA(+), 527 ecDNA(-)) were assigned a consensus measurement of purity estimation (CPE) by <xref ref-type="bibr" rid="bib2">Aran et al., 2015</xref>. (<bold>A</bold>) ecDNA(-) samples have slightly lower purity than ecDNA(+) samples (MWU <italic>p</italic>-value 0.0036). (<bold>B</bold>) For samples with high tumor purity (CPE ≥0.8, n=287), the expression directionality of CorEx genes (i.e. based on the Mann-Whitney U test) was highly correlated with that of all samples (n=870). Among the 275 genes that had a higher expression in the ecDNA(+) samples of all samples, 243 genes also had a higher expression in the ecDNA(+) samples of high tumor purity samples (red dots). Among the 284 genes that had a lower expression in the ecDNA(+) samples of all samples, 181 genes also had a lower expression in the ecDNA(+) samples of high tumor purity samples (blue dots). Together, 424 of 559 (75.8%) CorEx genes with significant directionality were in agreement between all samples and samples with high tumor purity.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-figsupp4-v1.tif"/></fig><fig id="fig1s5" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 5.</label><caption><title>Gene Ontology (GO) biological processes enriched by up-regulated CorEx genes from eight selection criteria ranging from 5 to 200 of 200 Boruta trials.</title><p>Of the 187 GO terms (turquoise) that were enriched by 262 up-regulated CorEx genes using 10 out of 200 Boruta trials as the selection criteria, 93 terms (49.7%) were enriched for each cut-off criteria, and 155 terms (82.9%) were enriched in at least 5 of the 8 cut-off criteria (dark blue). A minimum degree of three was required when using the python UpSet plot function (upsetplot v0.8.0).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig1-figsupp5-v1.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Extending the core set with co-expressed genes</title><p>We note that the Core gene set is not a comprehensive list of discriminatory genes, using a toy example. Consider gene ‘B,’ a member of the core gene set, and another gene, ‘C,’ whose expression values across all samples are nearly identical to the expression values of core gene B. The Boruta analysis would not need to assign gene C to the core set in addition to gene B, because adding both genes incurs the same predictive power as adding one. However, either, or both genes may play an important functional role. To correct this, we ran pvclust (<xref ref-type="bibr" rid="bib54">Suzuki and Shimodaira, 2006</xref>) to cluster all gene expression values, and to identify stable clusters using multiscale bootstrap resampling (<xref ref-type="fig" rid="fig1">Figure 1B</xref>; Methods). We used an approximately unbiased (AU) confidence value of 0.95 to select the most highly co-expressed gene clusters. An AU confidence value of 0.95 represents the rejection of the null hypothesis that a group of genes fails to form a stable cluster at a significance level of 0.05. Recomputing the number of Boruta trials that members of a cluster were selected in, we selected clusters that appeared in at least 10 of the 200 Boruta trials (Methods). This resulted in the selection of 354 recurring clusters (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1C</xref>).</p><p>Notably, among the 354 clusters, only two clusters (with 14 total genes) did not contain any Core genes. As most genes do not have completely identical expression patterns, we would expect one gene to be consistently picked as a Boruta gene over another co-expressed gene. Consistent with this hypothesis, most (344/354) clusters contained only 1 or 2 Core genes (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). When selecting clusters that contained at least one Core and one co-expressed gene, 53 of 71 clusters contained 1–3 Core genes (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>), confirming that a few genes per co-expressed cluster provide sufficient predictive value, but other co-expressed genes might still play an important functional role in maintaining ecDNA presence. This is true for clusters of various sizes, including the 2-member cluster #74 and the 21-member cluster #3. In cluster #74, <italic>CSTF1</italic> had similar expression values to the Core gene <italic>RAE1</italic>, which is a mitotic checkpoint regulator implicated in tumor progression (<xref ref-type="bibr" rid="bib29">Kobayashi et al., 2021</xref>; <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1D</xref>). While not necessarily increasing the predictive value, <italic>CSTF1</italic> is also a proto-oncogene involved in aberrant alternative splicing events (<xref ref-type="bibr" rid="bib61">Wang et al., 2019</xref>). In cluster #3, 12 genes were highly co-expressed with nine Core genes (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1D</xref>), and were enriched in cell-cycle related biological processes (Methods). Importantly, the total number of genes per cluster was also small (<xref ref-type="fig" rid="fig1">Figure 1D</xref>), with only 7 of 354 clusters carrying more than 10 genes. This suggests that the Core genes have specific roles that cannot be accomplished by multiple other genes.</p><p>Summarizing, the 354 clusters contained 643 genes, which included 408 Core genes and 235 additional genes (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). Together, we define these genes as the CorEx (Core + co-expressed) genes (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1A</xref>). To address the concern that the selection of CorEx genes based on bulk RNA-seq expression data could be confounded by tumor purity, we utilized a composite tumor purity score (CPE) (<xref ref-type="bibr" rid="bib2">Aran et al., 2015</xref>), and observed that the ecDNA(-) samples had slightly (but significantly) lower purity than ecDNA(+) samples (<italic>p</italic>-value 0.0036; <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>). This is consistent with reduced detection of ecDNA in less pure samples. However, lower sensitivity of ecDNA detection would reduce the strength of the signal but not result in false positives. Indeed, when we compared the significance of CorEx gene directionality in highly pure samples (tumor purity≥0.8; n=287) versus all samples (n=870), we found a significant correlation (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>), indicating the robustness of the CorEx set. The remaining manuscript investigates the functional properties of these genes.</p></sec><sec id="s2-3"><title>CorEx genes are better predictors of ecDNA status compared to other gene sets</title><p>We validated the relevance of CorEx genes in ecDNA presence by running cross-validation experiments (<xref ref-type="fig" rid="fig2">Figure 2A</xref>; Methods) to test the predictive power of CorEx gene expression in determining the ecDNA status of the sample. For comparisons, we used three other gene lists. The first list was a randomly chosen gene subset of identical size. For the second list, we performed a differential expression analysis using DESeq2 (<xref ref-type="bibr" rid="bib37">Love et al., 2014</xref>) and picked the 643 most significantly differentially expressed genes in terms of the absolute value of their shrunken log-fold change estimate (LFC; Methods). Using the sign of the LFC value as the determinant for directionality, 240 of these genes were up-regulated, while 403 were down-regulated. Notably, only 86 of these Top-|LFC| genes overlapped with the CorEx gene set (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1F</xref>; <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). For the third list, we used a generalized linear model (GLM) to predict 3012 genes whose expression levels were significantly associated with sample ecDNA status using a logit function after controlling for tumor subtype (Methods). Together, the three additional gene lists were denoted as random, Top-|LFC|, and GLM.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Validation of CorEx genes.</title><p>(<bold>A</bold>) Cross-validation experiments validating the predictive value of CorEx genes. Precision denotes the fraction of predicted samples that were truly ecDNA(+). Recall refers to the fraction of ecDNA(+) samples that were predicted correctly. (<bold>B</bold>) For precision windows of width 0.1 and a value of at least 0.5, recall values were plotted as boxplots. The interquartile ranges for CorEx and Core genes overlap, suggesting similar predictive power. CorEx genes have higher predictive rates compared to the top 643 differentially expressed genes based on logarithmic fold changes from a DESeq2 analysis (Top-|LFC| genes), 3012 significant genes selected from a generalized linear model (GLM), and 643 randomly selected genes. (<bold>C</bold>) CorEx genes were consistently up- or down-regulated in ecDNA(+) samples across tumor types, with the exception of Sarcoma (SARC). Approximately unbiased (AU) <italic>p</italic>-values from multiscale bootstrap resampling are shown at the dendrogram branches. (<bold>D</bold>) Of the 643 Top-|LFC| genes, 240 were up-regulated while 403 were down-regulated in ecDNA(+) samples. Of the CorEx genes, 325 were up-regulated while 318 were down-regulated. The absolute log-fold change (LFC) values of the Top-|LFC| gene set was significantly greater than that of the CorEx genes (<italic>p</italic>-value 1.83e-158). (<bold>E</bold>) The normalized gene expression values of the CorEx genes were significantly higher than that of the Top-|LFC| gene set (<italic>p</italic>-value &lt;2e-308). ***<italic>p</italic>-value &lt;0.001.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Gene expression of Top-|LFC| versus CorEx genes.</title><p>86 CorEx genes are part of the 643 Top-|LFC| gene set.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Extended validation of CorEx genes.</title><p>(<bold>A</bold>) Cross-validation experiments validating the predictive value of CorEx genes. Precision denotes the fraction of predicted samples that were truly ecDNA(+). Recall refers to the fraction of ecDNA(+) samples that were predicted correctly. CorEx genes have similar predictive value to Core genes. (<bold>B</bold>) CorEx genes have higher predictive rates compared to 643 randomly selected genes and the top 643 differentially expressed genes based on logarithmic fold changes from a DESeq2 analysis (Top-|LFC| genes).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Differential expression of Core genes by tumor type.</title><p>Core genes are also consistently up- or down-regulated in ecDNA(+) samples across tumor types, with the exception of Sarcoma (SARC), similar to CorEx genes (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). AU <italic>p</italic>-values from multiscale bootstrap resampling are shown at the dendrogram branches.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig2-figsupp3-v1.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Up-regulation of CorEx genes located on amplicons.</title><p>(<bold>A</bold>) <italic>PNMT</italic> normalized gene expression in ecDNA(+) vs. ecDNA(-) samples. (<bold>B</bold>) <italic>ITLN1</italic> normalized gene expression in ecDNA(+) vs. ecDNA(-) samples. In both cases, their up-regulation in ecDNA(+) samples is explained by their presence on a region with somatic, focal copy number amplification.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig2-figsupp4-v1.tif"/></fig></fig-group><p>For cross-validation tests, we performed multiple random 80–20 splits of the samples to generate 200 training and test data-sets (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). For each training-test data-set, a Random Forest method was used to train the predictability of the five gene lists (Methods) on the training data, and the predictive performance was tested on the test data. Expectedly, none of the gene lists was a great predictor of ecDNA status of a sample (<xref ref-type="fig" rid="fig2">Figure 2B</xref>, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). Nevertheless, the average of the area under the precision-recall curve (AUPRC) was higher for CorEx and Core genes (0.48 and 0.5), relative to GLM and Top-|LFC| (mean AUPRC: 0.43 each). For precision values of at least 0.7, the CorEx genes had significantly higher recall than Top-|LFC| genes (Mann-Whitney U-test p-value 4.8e-21) or GLM genes (p-value 8.5e-20). In turn, the Top-|LFC| and GLM genes were more predictive than random (mean AUPRC: 0.36). Expectedly, the predictive performance did not change when switching between Core genes and CorEx genes, because each of the non-core genes in the CorEx list had an expression pattern similar to at least one Core gene (<xref ref-type="fig" rid="fig2">Figure 2B</xref>, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>).</p><p>To test the persistence of CorEx genes across tumor types, we re-computed Cliff’s delta values <xref ref-type="bibr" rid="bib9">Cliff, 1993</xref>; <xref ref-type="bibr" rid="bib10">Cliff, 1996</xref> for each of the 11 TCGA tumor types that had at least 10 ecDNA(+) and at least 10 ecDNA(-) samples. The directionality of gene expression patterns was significantly similar to TCGA in each tissue type, with one exception (<xref ref-type="fig" rid="fig2">Figure 2C</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1E</xref>). The sole exception was the tumor type of Sarcoma (SARC). It is notable that the TCGA-SARC samples included many liposarcomas. In addition to containing ecDNA, liposarcoma samples are known to have extensively rearranged structures indicative of chromothripsis and neo-chromosome formation (<xref ref-type="bibr" rid="bib18">Garsed et al., 2014</xref>). For other tissue types, the <italic>p</italic>-values against a null hypothesis of no match to the pan-cancer prediction ranged from 4.2e-12–3.5e-85 (Fisher’s exact test) for the significant associations (Methods). The results were similar if we tested using only Core genes (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>). Summarizing, the 643 CorEx genes are differentially expressed across a multitude of tumor types, and have consistently higher or lower expression in ecDNA(+) samples relative to ecDNA(-) samples. These results are consistent with a pan-cancer role of CorEx genes in ecDNA biogenesis and maintenance.</p><p>The Top-|LFC| genes were also different from the CorEx genes by other metrics. Not surprisingly, the log-fold change (LFC) values of the top-|LFC| genes were higher than the LFC values of the CorEx genes (<xref ref-type="fig" rid="fig2">Figure 2D</xref>, MWU <italic>p</italic>-value 1.83e-158). However, much of the LFC change was due to the very low expression of the top-|LFC| genes in either ecDNA(+), or ecDNA(-) samples. In fact, the CorEx genes had higher expression in both ecDNA(+) and ecDNA(-) samples compared to the Differentially Expressed (DE) genes (<xref ref-type="fig" rid="fig2">Figure 2E</xref>, MWU <italic>p</italic>-value &lt;2e-308). While the absolute log fold-change in expressions of CorEx genes between ecDNA(+) and ecDNA(-) samples was not that high (median: 0.30, mean: 0.41), it was persistent across all samples (variance: 0.14, standard deviation: 0.37).</p><p>For example, the genes <italic>ITLN1</italic> and <italic>PNMT</italic> had the second and eighth-highest absolute LFC values of 3.92 and 2.89 in the top-|LFC| list. However, their normalized expression values in most ecDNA(+) samples were low. <italic>ITLN1</italic> had a normalized RSEM expression value <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>≤</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> (21<sup>st</sup> percentile) in 210/234 ecDNA(+) samples. Similarly, the normalized RSEM expression value of <italic>PNMT</italic> in 223/234 ecDNA(+) samples was less than 8.5 (rank percentile: 41.1%). For <italic>PNMT</italic>, the differential expression was mediated by 11 ecDNA(+) samples having an expression value ≥11, and 5 of the 11 samples contained <italic>PNMT</italic> on an ecDNA amplicon (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). Similarly, three samples with high RSEM contained <italic>ITLN1</italic> on an amplicon (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>), partly accounting for the high |LFC| value. In contrast, the CorEx gene, <italic>RAE1</italic>, had a high normalized expression value in both ecDNA(+) and ecDNA(-) samples (average 9.72, rank percentile 74.3%), with a small but persistent LFC value of 0.33.</p><p>The results confirm our intuition that differential expression can arise due to multiple reasons, including low expression of the gene in a majority of samples, or the copy number amplification of a gene in a few samples. In contrast, the CorEx genes were selected based on persistent over- or under-expression in ecDNA(+) samples.</p></sec><sec id="s2-4"><title>CorEx genes primarily up-regulate three biological processes: Cell cycle, cell division, and DNA damage response</title><p>To identify enriched biological processes specific to either up-regulated or down-regulated genes in ecDNA(+) samples, we combined two metrics of effect size, Cliff’s delta (<xref ref-type="bibr" rid="bib9">Cliff, 1993</xref>; <xref ref-type="bibr" rid="bib10">Cliff, 1996</xref>), and log fold change <xref ref-type="bibr" rid="bib37">Love et al., 2014</xref> to determine the directionality of CorEx genes (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1G</xref>; Methods). The two effect size metrics were mostly in agreement in terms of directionality. Of the 7288 genes that passed the negligible effect size thresholds in both metrics, only 14 were not in concordance. This more stringent approach, in comparison to a simple directionality based on the sign of a single effect size value, was applied given that enrichment analysis on gene sets is dependent on not only the number of up- or down-regulated genes but also the degree of overlap with genes under a specific biological process term (Methods). Using this approach, among the 643 CorEx genes, 262 genes were found to be up-regulated in ecDNA(+), while 271 were found to be down-regulated (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1A</xref>; Methods). 110 genes did not make the effect size cut-off. The numbers were similar for the 408 Core genes, with 190 up-regulated, 196 down-regulated, and 22 genes not making the cut-off.</p><p>We performed enrichment analysis on gene sets to identify the Gene Ontology (GO) biological processes that are enriched in CorEx genes (Methods). Briefly, we applied a one-sided Fisher’s exact test using gene sets from MSigDB (<xref ref-type="bibr" rid="bib52">Subramanian et al., 2005</xref>; <xref ref-type="bibr" rid="bib34">Liberzon et al., 2011</xref>; <xref ref-type="bibr" rid="bib35">Liberzon et al., 2015</xref>), using a false discovery rate of 5% (Benjamini-Hochberg procedure). The UP-regulated genes enriched 187 Biological processes (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1H</xref>). Note that the GO-biological process (BP) terms are not independent, because of their hierarchical organization, and sharing of genes across different GO terms. Therefore, we used an approach similar to that used in DAVID <xref ref-type="bibr" rid="bib25">Huang et al., 2007</xref> to cluster the biological processes enriched by the UP-regulated genes into 11 broad categories (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1I</xref>; <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>; Methods). The 11 categories were assigned a name using manual inspection of the constituent GO terms, or called ‘Other.’ The 11 categories (including ‘Other’) are shown in a waterfall plot to explain the contribution of each gene to a category (<xref ref-type="fig" rid="fig3">Figure 3A</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Up-regulated CorEx genes.</title><p>(<bold>A</bold>) Gene Ontology (GO) biological processes enriched in up-regulated genes were clustered into 11 broad categories. The horizontal barplot represents the number of GO biological processes belonging to each of the 11 broad categories, while the vertical barplot represents the number of broad categories that a specific GO biological process belongs to. (<bold>B</bold>) Genes up- or down-regulated in processes involved in major double-strand break (DSB) damage repair pathways. Many critical genes in the classical non-homologous end-joining (c-NHEJ) pathway were down-regulated in ecDNA(+) samples relative to ecDNA(-) samples.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Biological process categories enriched in up-regulated CorEx genes.</title><p>187 enriched Gene Ontology Biological Process (GOBP) terms represented by 169 up-regulated CorEx genes. The enriched biological processes cluster into 11 categories, which in turn can be grouped into cell-cycle regulation, Mitotic cell division, double-strand break DNA Damage response, and the <italic>HOX</italic> gene cluster.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig3-figsupp1-v1.tif"/></fig></fig-group><p>The 10 categories included expected participation of biological processes involved in (a) cell-cycle regulation (Mitotic/Meiotic Cell Cycle, G1/S, G2/M) (b) cell-division (Spindle Organization, Cell Division, Chromosome Condensation, Chromosome Segregation), (c) DNA Damage response (DNA Repair), and (d) the HOX Gene cluster. Indeed, one of the largest clusters, cluster #3, containing 9 Core genes and 12 co-expressed genes, was enriched in GO-BP terms related to the cell cycle (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1J</xref>). Notably, the enriched categories also included a role for the <italic>HOX</italic> genes with 17 members of the <italic>HOX</italic> family up-regulated in ecDNA(+) cancers (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1I</xref>). Many recent reports have associated <italic>HOX</italic> genes with cancer, including an association with their phenotypic ‘hallmarks’ (<xref ref-type="bibr" rid="bib14">Feng et al., 2021</xref>). Genes involved in angiogenesis (<italic>HOXA2</italic>, <italic>HOXC5</italic>), genome instability (<italic>HOXC5</italic>, <italic>HOXC11</italic>), deregulating cellular energetics (<italic>HOXA4</italic>, <italic>HOXC5</italic>), and metastasis (<italic>HOXA2</italic>) were all up-regulated in ecDNA(+) cancers.</p><p>While the Meiotic cell-cycle was also enriched, only seven genes were allocated specifically to the group of enriched terms: <italic>SEPP1</italic>, <italic>SNRPA1</italic>, <italic>TMEM203</italic>, <italic>RNF114</italic>, <italic>TAF4</italic>, <italic>TNFAIP6</italic>, and <italic>PTX3</italic> (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1K</xref>). Though these genes have roles in the meiotic cell cycle, they have also been implicated in cancer and other inflammatory diseases. The spliceosomal protein SNRPA1 is a pro-metastatic splicing enhancer (<xref ref-type="bibr" rid="bib16">Fish et al., 2021</xref>). Over-expression of TNFAIP6 has been associated with metastasis and poor prognosis for patients (<xref ref-type="bibr" rid="bib65">Zhang et al., 2021</xref>). TMEM203 is a STING-centered signaling regulator implicated in inflammatory diseases (<xref ref-type="bibr" rid="bib33">Li et al., 2019</xref>). RNF114 is a zinc-binding protein whose over-expression is an indicator of epithelial inflammation and is implicated in various tumors (<xref ref-type="bibr" rid="bib15">Feng et al., 2022</xref>). TAF4, a transcription initiation factor, when overexpressed, is implicated in ovarian cancer by playing a role in dedifferentiation that promotes metastasis and chemoresistance (<xref ref-type="bibr" rid="bib46">Ribeiro et al., 2014</xref>). Finally, in looking at the genes in the ‘Other’ category, <italic>SHCBP1</italic> was the only gene unique to it. A member of the neural precursor cell proliferation process, SHCBP1 is reported to promote tumor cell signaling and proliferation (<xref ref-type="bibr" rid="bib64">Xu et al., 2020</xref>).</p><p>Taken together, the 11 biological process categories explain 165 of the 262 up-regulated CorEx genes, and suggest that mitotic cell-division, cell cycle regulation, and DNA damage response are the three broad categories of biological processes up-regulated to ensure ecDNA presence, along with an up-regulation of genes in the <italic>HOX</italic> cluster.</p></sec><sec id="s2-5"><title>CorEx genes up-regulate specific double-strand break repair pathways</title><p>The 16 enriched DNA damage response GO terms contained the terms ‘double-strand break repair’ and ‘recombination,’ but none contained the terms ‘single-strand,’ ‘nucleotide-excision,’ or ‘mismatch-repair’ (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1I</xref>), suggesting that the CorEx genes are largely composed of genes involved in multiple double-strand breaks (DSB) repair pathways, which include classical non-homologous end-joining (c-NHEJ), Alternative end-joining (Alt-EJ), single-strand annealing (SSA), or homology-directed repair (HDR) (<xref ref-type="bibr" rid="bib5">Chang et al., 2017</xref>).</p><p>The choice of these varied DSB repair mechanisms for ecDNA presence is not well understood. We compiled and hand-curated a list of 129 genes involved in DSB repair and marked them for their role in one or more of these four pathways (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1L</xref>). Of these genes, a high number (67) were up-regulated in ecDNA(+), while a smaller number (15) were down-regulated, relative to ecDNA(-) samples. This breakdown of 129 DDR genes contrasts with an analysis using all genes where a nearly identical number of genes (5256, and 5251) were up- and down-regulated in ecDNA(+) samples, confirming that DDR genes are significantly up-regulated relative to all differentially expressed genes (<italic>p</italic>-value &lt;0.0001; Fisher’s exact test). When broken down to the roles of genes in individual DSB repair pathways, we found that Alt-EJ with 11 up-regulated and one down-regulated genes (<italic>p</italic>-value 0.0063), SSA (11 up, one down (<italic>p</italic>-value 0.0063)), and HR (46 up, eight down; <italic>p</italic>-value &lt;0.00001) were all up-regulated. However, classical NHEJ (14 up, seven down; <italic>p</italic>-value: 0.19) was not significantly up-regulated in ecDNA(+) samples relative to ecDNA(-) samples (Methods, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1M</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1N</xref>).</p><p>The expression of key genes in these pathways raises the possibility of an increased role of non-classical-NHEJ processes in ecDNA development or progression, relative to c-NHEJ (<xref ref-type="fig" rid="fig3">Figure 3B</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1L</xref>). A number of genes involved in c-NHEJ were down-regulated in ecDNA-containing tumors relative to non-ecDNA tumors. These included <italic>XLF</italic>/<italic>NHEJ1</italic> (MWU <italic>p</italic>-value 2.05e-03), which is a key member of the ligase complex required for c-NHEJ; <italic>LIG4</italic>, another member of the ligase complex (MWU <italic>p</italic>-value 0.03), <italic>PNKP</italic>, which generates 5ʹ-phosphate/3ʹ-hydroxyl DNA termini required for ligation (MWU <italic>p-</italic>value 3.90e-06); and also, DNA polymerases λ (<italic>POLL</italic>; MWU <italic>p</italic>-value 1.09e-21) and μ (<italic>POLM</italic>; MWU <italic>p</italic>-value 0.01), which promotes the ligation of terminally compatible overhangs requiring fill-in synthesis and promotes the ligation of incompatible 3ʹ overhangs <xref ref-type="bibr" rid="bib19">Ghosh and Raghavan, 2021</xref> in a template independent manner, respectively. This does not imply a defect in these repair processes, but rather, potentially additional or preferential utilization of alternative DSB repair pathways in ecDNA-containing tumors.</p><p>TP53BP1 is key to blocking resection and promoting the c-NHEJ pathway choice, but is displaced by BRCA1 and the MRN complex to initiate resection in the broken strands <xref ref-type="bibr" rid="bib11">Daley and Sung, 2014</xref>; <xref ref-type="bibr" rid="bib17">Fouquin et al., 2017</xref>. <italic>BRCA1</italic> was significantly up-regulated in ecDNA(+) samples (<xref ref-type="fig" rid="fig3">Figure 3B</xref>), while <italic>TP53BP-1</italic> was significantly down-regulated (MWU <italic>p</italic>-value 0.016), although with negligible effect size (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1L</xref>). Supporting the role of alternative pathway choice for DDR, key genes in the Alt-EJ pathway, including <italic>PARP-1</italic>, DNA polymerase θ (<italic>POLQ</italic>), <italic>LIG1</italic>, <italic>LIG3</italic>, and <italic>FEN1</italic> were all significantly up-regulated in ecDNA(+) samples. Homology directed repair is the preferred pathway when a sister chromatid is available to act as a template. HDR is initiated by additional and extensive resection. The genes <italic>BLM</italic>, <italic>EXO1</italic>, <italic>RPA1</italic>, and <italic>RPA3</italic> which promote additional resection, as well as <italic>BRCA1</italic>, <italic>BRCA2</italic>, <italic>RAD51</italic>, and others that support HDR were all found to be significantly up-regulated. We can conclude that the specific pathway choice for DSB repair in ecDNA(+) may have enhanced dependence on alt-EJ and homology-directed repair pathways.</p></sec><sec id="s2-6"><title>CorEx genes primarily down-regulate immune system processes</title><p>Using a methodology similar to the analysis of the up-regulated genes, the down-regulated genes enriched 73 GO terms (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1O</xref>), and could be clustered into seven broad categories, including ‘Other’ (<xref ref-type="fig" rid="fig4">Figure 4A</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1P</xref>; <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). Surprisingly, all categories were immunomodulatory. The most enriched broad category contained 75 CorEx genes relating to the Lymphocyte activation pathway. It included genes enriching ‘T-cell activation’ (28 CorEx genes; <italic>p</italic>-value 2.58e-05), and ‘Positive regulation of cell-cell adhesion’ (16 CorEx genes; <italic>p</italic>-value 4.49e-03). Other down-regulated pathways included Cytokine activation, especially for genes in the IL-12 pathway (six CorEx genes, <italic>p</italic>-value 7.17e-03), TNF super-family (12 CorEx genes, <italic>p</italic>-value 7.24e-03), and Inflammation, including, for example, down-regulation of Toll-like receptor 2 signaling (four CorEx genes, <italic>p</italic>-value 4.49e-03). Finally, the broad category of Leukocyte chemotaxis was also enriched among the down-regulated genes. The chemotaxis genes include many chemokines and their receptors involved in the trafficking of T cells to the site of the tumor. The remaining down-regulated genes included four fucosyltransferases, and the category marked ‘Other.’ <italic>FUT2</italic> silencing is associated with reduced adhesion and increased metastatic potential (<xref ref-type="bibr" rid="bib23">He et al., 2023</xref>). Notably, the category marked ‘Other’ was dominated by genes in NF-κB pathway regulation (14 CorEx genes, <italic>p</italic>-value 2.28e-02).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Down-regulated CorEx genes.</title><p>(<bold>A</bold>) Gene Ontology (GO) biological processes enriched in down-regulated genes were clustered into seven broad categories. The horizontal barplot represents the number of GO biological processes belonging to each of the seven broad categories, while the vertical barplot represents the number of broad categories that a specific GO biological process belongs to. (<bold>B</bold>) Four of these categories map to steps in the cancer-immunity cycle. CorEx genes in three of the four categories were significantly down-regulated compared to all genes (Fisher’s exact test).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Biological process categories enriched in down-regulated CorEx genes.</title><p>73 enriched GOBP terms represented by 119 down-regulated CorEx genes. The enriched biological processes can be clustered into 7 categories, 6 of which are related to immune response, suggesting a broad-based deregulation.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig4-figsupp1-v1.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Tumor microenvironment (TME) subtypes in extrachromosomal DNA (ecDNA)-containing tumors.</title><p>TME sub-typing of The Cancer Genome Atlas (TCGA) samples classified as ecDNA(+) or ecDNA(-) (Amplicon Classifier ver. 0.4.9) based on the immune subtypes provided by <xref ref-type="bibr" rid="bib56">Thorsson et al., 2018</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig4-figsupp2-v1.tif"/></fig></fig-group><p>NF-κB signaling represents a prototypical, proinflammatory pathway (<xref ref-type="bibr" rid="bib32">Lawrence, 2009</xref>) with multiple roles, including apoptosis. Specifically, 8 of the 14 down-regulated genes involved caspase activation (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1Q</xref>), representing the pro-apoptotic arm of NF-κB signaling. A parallel pathway for sensing endogenous ligands secreted in cell death and cancer is mediated by Toll-like receptor (TLR) proteins (<xref ref-type="bibr" rid="bib58">Urban-Wojciuk et al., 2019</xref>). Remarkably, all ten TLRs were significantly down-regulated in ecDNA(+) tumors. They included TLRs expressed on the cell membrane that bind lipids and proteins as well as TLRs expressed on endosomal membranes that bind DNA. The CorEx down-regulated genes also included many involved in TLR signaling, such as <italic>TLR3</italic>, <italic>CYBA</italic>, <italic>LYN</italic>, and <italic>TIRAP</italic>.</p><p>Four of the seven broad categories mapped to facets of the cancer immune cycle (<xref ref-type="bibr" rid="bib6">Chen and Mellman, 2013</xref>; <xref ref-type="fig" rid="fig4">Figure 4B</xref>). We tested if CorEx genes in these categories were more likely to be down-regulated rather than up-regulated, when compared to the non-CorEx differentially expressed genes. The Inflammation category, which mapped to the ‘Cancer antigen presentation’ facet, showed 18 up-regulated and 76 down-regulated CorEx genes (<italic>p</italic>-value 0.005, Fisher exact test, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1P</xref>). Similarly, the CorEx genes related to the ‘Trafficking of T cells’ facet (‘Leukocyte migration and chemotaxis’ category, 13 up, 49 down-regulated; <italic>p</italic>-value 0.03) and ‘Infiltration and recognition of tumor cells by cytotoxic T cells’ facet (‘Lymphocyte activation’ category, 23 up, 75 down-regulated; <italic>p</italic>-value 0.0075) were also significantly down-regulated. However, down-regulation in the ‘Priming and activation’ facet (‘Cytokine production’ category, 6 up, 28 down-regulated) was not significant at the 5% level.</p><p>As the RNA data were bulk-sequenced, transcripts were sampled from tumor cells and cells from the tumor microenvironment. <xref ref-type="bibr" rid="bib56">Thorsson et al., 2018</xref> mined immune cell expression signatures to identify six immune subtypes: wound healing (C1), IFN-γ dominant (C2), inflammatory (C3), lymphocyte depleted (C4), immunologically quiet (C5), and TGF-β dominant (C6). A recent study analyzing the tumor microenvironment (TME) of ecDNA(+) vs. ecDNA(-) samples in seven tumor subtypes revealed an association of ecDNA presence with immune evasion (<xref ref-type="bibr" rid="bib63">Wu et al., 2022</xref>). Our results (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>), which used an updated version of the classification method for these ecDNA(+) samples, were broadly consistent with those from the <xref ref-type="bibr" rid="bib63">Wu et al., 2022</xref>. study. Our results suggested an increase in C1 and C2 subtypes and a depletion of C3 and C6 between ecDNA(+) and ecDNA(-) categories (<italic>p</italic>-value 3.96e-03, Chi-squared test). Notably, the C3 (inflammatory) subtype is associated with lower levels of somatic copy number alterations, and C6 with high lymphocyte infiltration, while C1 is associated with elevated levels of angiogenic genes. These are consistent with our findings of increased somatic copy numbers, increased expression of angiogenic genes on ecDNA(+) samples, and reduced lymphocyte infiltration.</p></sec><sec id="s2-7"><title>ecDNA(+) samples carry a higher mutational burden relative to ecDNA(-) samples</title><p>In order to understand if the change in the transcriptional program was driven by mutations to the genes, we checked if ecDNA(+) samples have differential levels of mutation relative to ecDNA(-). Intriguingly, we found that the total mutation burden was significantly higher in ecDNA(+) samples relative to ecDNA(-) samples (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). The result was significant also when mutations were limited to deleterious substitutions as measured by SIFT or PolyPhen2, and high-impact insertions and deletions (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). However, when controlling for cancer type, only glioblastoma (GBM; lower mutations in ecDNA(+)), low-grade gliomas (LGG; higher mutations in ecDNA(+)), and uterine corpus endometrial carcinoma (UCEC; lower mutations in ecDNA(+)) continued to show differential total mutational burden (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). Of note, we did not examine the impact of a hypermutator phenotype, which could lead to high tumor mutation burden in tumors with mismatch repair deficiencies. Next, we tested if specific genes were differentially mutated between the two classes (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). For deleterious/high-impact mutations, <italic>TP53</italic> was the only gene whose mutational patterns were significantly higher in ecDNA(+) compared to ecDNA(-) (OR 2.67, Bonferroni adjusted <italic>p</italic>-value 4.22e-07). BRAF mutations, however, were more common in ecDNA(-) samples and were significant to an adjusted p-value &lt;0.1 (OR 0.27). The excess of <italic>TP53</italic> mutations in ecDNA(+) samples provides additional support to the hypothesis that mutations in DNA damage response or cell cycle checkpoints are important for ecDNA presence. Other genes that are differentially mutated with nominal significance (unadjusted p-value &lt;0.005) are shown in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1R</xref>.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Mutational characteristics of extrachromosomal DNA (ecDNA)-containing tumors.</title><p>(<bold>A</bold>) Total mutation burden of ecDNA(+) and ecDNA(-) samples. ecDNA(+) samples have significantly higher mutation burden than the ecDNA(-) samples (<italic>p</italic>-value &lt;0.0001, Mann Whitney test). (<bold>B</bold>) Odds ratios of differentially mutated genes in ecDNA(+) and ecDNA(-) (<italic>p</italic>-value &lt;0.005). The size of the dot indicates whether the corresponding gene belongs to the Cancer Gene Census (CGC) or not (Non-CGC). Only <italic>TP53</italic> and <italic>BRAF</italic> showed significance at the level of FDR &lt;0.1 (Benjamini-Hochberg).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Mutation burden based on damaging mutations.</title><p>Mutation burden of ecDNA(+) and ecDNA(-) calculated only with damaging mutations (deleterious SNVs predicted with SIFT or PolyPhen2, frameshift INDELs).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-figsupp1-v1.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Mutation burden by tumor type.</title><p>Only mutation burdens of glioblastoma (GBM), low-grade gliomas (LGG), and uterine corpus endometrial carcinoma (UCEC) were significantly different between ecDNA(+) and ecDNA(-).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-figsupp2-v1.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>Performance of XGBoost model.</title><p>Performance evaluation of the best XGBoost model selected by HyperOpt.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-figsupp3-v1.tif"/></fig><fig id="fig5s4" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 4.</label><caption><title>Unsupervised principal component analysis on gene mutations.</title><p>PCA result performed with a binary mutational matrix of all mutated genes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-figsupp4-v1.tif"/></fig><fig id="fig5s5" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 5.</label><caption><title>Single base substitution (SBS) signatures in The Cancer Genome Atlas (TCGA) samples by extrachromosomal DNA (ecDNA) status.</title><p>The activities of SBS signatures 1–60 in ecDNA(+) (orange) and ecDNA(-) (blue) samples (<bold>A</bold>) measured across 1440 TCGA samples and (<bold>B</bold>) separated by tumor types with at least one ecDNA(+) sample each. Activity refers to the estimated relative contribution of a specific mutational signature, and is calculated as follows: (number of variants that are attributed to a specific signature in a group)/(total number of variants in a group).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-fig5-figsupp5-v1.tif"/></fig></fig-group><p>We also tested if a collection of gene mutations could predict ecDNA status using XGBoost (<xref ref-type="bibr" rid="bib7">Chen and Guestrin, 2016</xref>), which uses an adaptive boosting of ‘weak classifiers’ to predict class. Here, each mutated gene was treated as a weak classifier of ecDNA status. However, the two classes could not be separated with high accuracy (<xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>). An unsupervised principal component analysis did not separate the two classes either. Only the first principal component explained a significant proportion (14%) of the total variance (<xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4</xref>) and did not separate the bulk of the samples. Finally, we recapitulated earlier findings that ecDNA(+) samples enrich for APOBEC activity through the presence of the mutation signatures SBS2 and SBS13 (<xref ref-type="bibr" rid="bib4">Bergstrom et al., 2022</xref>; <xref ref-type="bibr" rid="bib22">Hadi et al., 2020</xref>; <xref ref-type="fig" rid="fig5s5">Figure 5—figure supplement 5</xref>). The enrichment in <italic>TP53</italic> mutations was also consistent with previous findings (<xref ref-type="bibr" rid="bib38">Luebeck et al., 2023</xref>). On balance, however, collections of gene mutations did not distinguish ecDNA(+) samples from ecDNA(-) samples, at least at a pan-cancer level, in contrast to the gene expression data.</p></sec><sec id="s2-8"><title>Persistently occurring genes in ecDNA(+) samples may represent potential vulnerabilities</title><p>Any CorEx gene is either a Core gene that was selected as a feature in at least 5% of 200 Boruta trials, or be highly co-expressed with a Core gene. Because the selection criterion of 5% is arbitrary, we also tested robustness with eight other cut-offs ranging from 5-of-200–200-of-200 Boruta trials. The number of CorEx genes expectedly decreases with more stringent cut-offs. However, of the 187 GO terms that were enriched by 262 CorEx UP-genes using 10 of 200 Boruta trials as the selection criteria, 93 terms (49.7%) were enriched for each cut-off (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>), and 155 terms (82.9%) were enriched in at least 5 of the 8 cut-off criteria. Given that our subsequent analyses utilized the hierarchy of GO terms and identified four GO-categories enriched by UP-regulated genes, the conclusions would hold regardless of the specific cut-off.</p><p>To rank CorEx genes by importance, we computed harmonic mean rank values based on three categories: (a) the average GINI importance statistic from the trained random forest models; (b) the number of Boruta trials that a gene was selected in; and (c) the number of Boruta trials (out of 200) that a gene was selected in when counting by cluster (Methods). 65 genes that were up-regulated (47 genes) or down-regulated (18 genes) had a harmonic rank lower than 3 (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1A</xref>). The next highest-ranked gene had a harmonic rank exceeding 17. These 65 genes represent the most persistent differentially expressed CorEx genes, and appeared as Core (or clustered gene) in all 200 Boruta trials. Notably, of the 24 genes most frequently expressed on ecDNA, (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>) only EGFR, and CDK4 were included in the list of 65 genes, suggesting that the most persistent CorEx genes do not themselves appear frequently on ecDNA.</p><p>Expectedly, the high-ranked up-regulated genes impacted cell division (16 genes), cell cycle regulation (10 genes), and DNA damage response (16 genes). Only 12 of the 47 genes were not included in the gene sets of any enriched GO term. Many of these genes were from small CorEx clusters with less than three members, but we also found six genes from the HOX gene cluster (cluster #17), and another cluster of 21 genes (cluster #3). Members of cluster #3 appeared in all 200 Boruta trials; however, there were three genes all involved in cell-division (<italic>TPX2</italic>, <italic>KIF2C</italic>, and <italic>AURKA</italic>), each of which appeared in at least 180 Boruta trials. High expression among these three genes is associated with poor prognosis (<xref ref-type="bibr" rid="bib12">De Luca et al., 2006</xref>), and due to the highly persistent nature of their differential expression across ecDNA(+) samples, they represent a possible widespread vulnerability for ecDNA(+) samples.</p><p>Intriguingly, 14 of the 18 down-regulated genes with low harmonic rank came from a single cluster (#2; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1C</xref>), and 13 of the 18 genes did not specifically enrich any specific BP ontology. Six of the down-regulated genes appeared in 180 or more Boruta trials (<italic>CHMP7</italic>, <italic>XPO7</italic>, <italic>INTS9</italic>, <italic>TACR1</italic>, <italic>KIAA1967</italic>, and <italic>PCM1</italic>). Some of these genes (<italic>CHMP7</italic>, <italic>XPO7</italic>, <italic>KIAA1967</italic>) are reported to be tumor suppressor genes (<xref ref-type="bibr" rid="bib20">Guo et al., 2021</xref>; <xref ref-type="bibr" rid="bib27">Innes et al., 2021</xref>; <xref ref-type="bibr" rid="bib45">Qin et al., 2015</xref>). However, the exact functional role of down-regulating these genes in ecDNA(+) samples remains to be elucidated.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>ecDNA is increasingly recognized as a major cause of oncogene amplification, intratumoral genetic heterogeneity, accelerated evolution, and treatment resistance, but many of the underlying processes involved in its formation, function, and progression are not fully understood. The ability to conduct multi-omic studies of well-curated, bona fide clinical tumor samples, such as the TCGA, presents an opportunity to learn about differentially regulated gene expression programs that may be involved in ecDNA biogenesis or maintenance, and in worse outcomes for patients (<xref ref-type="bibr" rid="bib62">Wu et al., 2019</xref>; <xref ref-type="bibr" rid="bib42">Morton et al., 2019</xref>; <xref ref-type="bibr" rid="bib59">van Leen et al., 2022</xref>). Using a relatively intuitive set of principles, we have developed a machine learning approach that identifies differentially expressed, co-regulated genes in ecDNA-containing tumors, highlighting four main biological processes: non-c-NHEJ DSB repair, cell cycle, proliferation control, and immune regulation.</p><p>The GO analysis revealed three core biological processes that were up-regulated and only the immune system processes as being down-regulated. These observations strengthen the case for targeting proteins involved in mitotic cell-division (<xref ref-type="bibr" rid="bib60">Von Hoff et al., 1992</xref>), cell-cycle regulation, and DNA damage response in ecDNA(+) cancers, but also reveal roles for the <italic>HOX</italic> cluster of genes. Also, in this paper, we did not extensively study the role of ncRNA in the prediction of ecDNA status. We do note that <italic>HOTAIR</italic>, encoded in the <italic>HOXC</italic> locus, is independently associated with metastasis and poor outcomes (<xref ref-type="bibr" rid="bib21">Gupta et al., 2010</xref>). Further experiments are needed to provide a mechanistic basis for the role of <italic>HOX</italic> cluster genes in maintaining ecDNA presence, as also for the involvement of ncRNA.</p><p>The DNA damage genes are broadly up-regulated in ecDNA(+) samples, especially in double-strand break repair. Within this broad category of mechanisms, our analysis suggests that alternative DSB repair pathways such as Alt-EJ are preferred relative to classical NHEJ. This is consistent with previous observations of small microhomologies at breakpoint junctions (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>; <xref ref-type="bibr" rid="bib50">Sanborn et al., 2013</xref>), and has important implications in therapeutic selection that will need to be validated in future experimental studies. We note, however, that the microhomology analyses typically study breakpoint junctions, and might ignore double-strand breaks in non-junctional sequences which could be observed, for example at replication-transcription junctions.</p><p>The down-regulated genes were primarily immunomodulatory in nature, in addition to a few persistently down-regulated tumor suppressor genes. Lowered expression of immunomodulatory genes in ecDNA(+) samples has been previously reported (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>; <xref ref-type="bibr" rid="bib63">Wu et al., 2022</xref>), but not mechanistically explained. Remarkably, the down-regulated immunomodulatory genes encompassed most aspects of the cancer immune cycle, suggesting impaired recognition of tumor DNA and proteins as foreign in ecDNA(+) tumors. Sensing of foreign DNA, including tumor DNA, is often mediated by the cGAS/STING pathway (<xref ref-type="bibr" rid="bib49">Samson and Ablasser, 2022</xref>; <xref ref-type="bibr" rid="bib53">Sun et al., 2013</xref>). Intriguingly, cGAS was significantly up-regulated in ecDNA(+) samples, while STING was significantly down-regulated, suggesting a role for STING agonists in intervention. Finally, in addition to the down-regulation of genes in the toll-like receptor family, we observed a down-regulation of genes involved in regulating TLR signaling pathways that were part of the CorEx list. Understanding the mechanisms of broad down-regulation of TLRs could provide insight into vulnerabilities of ecDNA(+) tumors.</p><p>Mutation data alone does not provide as clear a picture of the genes involved in ecDNA status prediction. We did observe that the total mutation burden (TMB) was higher in ecDNA(+) samples. However, that relationship is much less clear after controlling for cancer type. High TMB has been positively correlated with sensitivity to immunotherapy (<xref ref-type="bibr" rid="bib47">Rizvi et al., 2015</xref>), and better patient outcomes; however, the gene expression patterns suggest that immunomodulatory genes are down-regulated in ecDNA(+) samples, and patients with ecDNA(+) tumors have worse outcomes (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>). Notably, other results have suggested that the correlation between TMB and response to immunotherapy is not uniform, and it can vary across different tumor subtypes (<xref ref-type="bibr" rid="bib41">McGrail et al., 2021</xref>). Specifically, our data is consistent with previous results which showed that Gliomas with high TMB have worse response to immunotherapy relative to gliomas with low TMB (<xref ref-type="bibr" rid="bib41">McGrail et al., 2021</xref>). In general, no collection of gene mutations was predictive of ecDNA status, although mutations in <italic>TP53</italic> were more likely in ecDNA(+) samples, and perhaps are an important driver for ecDNA formation (<xref ref-type="bibr" rid="bib38">Luebeck et al., 2023</xref>).</p><p>These results suggest that cancer cells that contain ecDNA have profound alterations in their global transcriptional patterns. Importantly, these transcriptional differences do not arise solely from genes on the ecDNAs themselves, but rather suggest that fundamental global processes involved in DSB repair, cell cycle control, and immune regulation contribute to ecDNA formation and pathogenesis.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Reagent type (species) or resource</th><th align="left" valign="bottom">Designation</th><th align="left" valign="bottom">Source or reference</th><th align="left" valign="bottom">Identifiers</th><th align="left" valign="bottom">Additional information</th></tr></thead><tbody><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Amplicon Classifier</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/jluebeck/AmpliconClassifier">https://github.com/jluebeck/AmpliconClassifier</ext-link>; <xref ref-type="bibr" rid="bib39">Luebeck et al., 2024</xref></td><td align="left" valign="bottom">v0.4.9</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Boruta</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/scikit-learn-contrib/boruta_py">https://github.com/scikit-learn-contrib/boruta_py</ext-link>; <xref ref-type="bibr" rid="bib24">Homola et al., 2024</xref></td><td align="char" char="." valign="bottom">6.21.2021</td><td align="left" valign="bottom">Modified to allow for early termination based on stagnant tentative counts.</td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Cliff’s delta</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/neilernst/cliffsDelta">https://github.com/neilernst/cliffsDelta</ext-link>: copy archived at <xref ref-type="bibr" rid="bib13">Ernst, 2021</xref></td><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">DESeq2</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://www.bioconductor.org/packages/release/bioc/html/DESeq2.html">https://www.bioconductor.org/packages/release/bioc/html/DESeq2.html</ext-link></td><td align="left" valign="bottom">v1.36.0</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">generalized linear model</td><td align="left" valign="bottom">glm() function in R stats package</td><td align="left" valign="bottom">v4.2.0</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">pvclust</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/shimo-lab/pvclust">https://github.com/shimo-lab/pvclust</ext-link>; <xref ref-type="bibr" rid="bib55">Suzuki et al., 2019</xref></td><td align="left" valign="bottom">v2.2–0</td><td align="left" valign="bottom"><xref ref-type="bibr" rid="bib54">Suzuki and Shimodaira, 2006</xref></td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">CorEx</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/miinslin/ecDNA_Gene_Expression">https://github.com/miinslin/ecDNA_Gene_Expression</ext-link>, copy archived at <xref ref-type="bibr" rid="bib36">Lin, 2024</xref></td><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr></tbody></table></table-wrap><sec id="s4-1"><title>TCGA sample ecDNA status classification</title><p>Amplicon Classifier (version 0.4.9, <ext-link ext-link-type="uri" xlink:href="https://github.com/jluebeck/AmpliconClassifier">https://github.com/jluebeck/AmpliconClassifier</ext-link>) classified amplicons detected in 1,921 TCGA samples into five sub-types: ecDNA, BFB, complex non-cyclic, linear, and no-amplification. When classifying a sample with multiple amplicons, the order of preference is as follows: ecDNA, BFB, complex non-cyclic, linear, and no-amplification. Given the challenges of detecting ecDNA from short read data, and to avoid possible false-negative ecDNA classifications, samples with a BFB or complex non-cyclic status which were not called ecDNA(+), were removed from the analysis. We treated samples with the linear amplification and no-amplification classifications as ecDNA(-). Of the 1921 samples, 1535 samples classified as ecDNA(+) and ecDNA(-) had RNA-seq data, including 1406 primary solid tumor samples, 95 tumor metastasis samples, and 34 primary blood-derived cancer – peripheral blood samples. Removing metastases results in a total of 1440 samples, including 243 ecDNA(+) and 1197 ecDNA(-) samples. While the set of 1440 samples represented 24 tumor types, ten of these tumor types had insufficient numbers of ecDNA(+) samples, including four tumor types with no ecDNA(+) samples. To prevent the 561 ecDNA(-) samples representing these tumors from skewing the analysis, we removed 570 samples representing tumor types with less than three ecDNA(+) samples. This resulted in a total of 870 samples representing 14 tumor types, of which 234 were classified as ecDNA(+) and 636 were classified as ecDNA(-).</p></sec><sec id="s4-2"><title>Gene expression datasets</title><p>Gene expression data for 32 studies part of the TCGA Pan-cancer Atlas was downloaded from cBioPortal (01.05.2021) (<ext-link ext-link-type="uri" xlink:href="https://www.cbioportal.org/">https://www.cbioportal.org/</ext-link>). The cBioPortal ‘data_RNA_Seq_v2_expression_median.txt’ data is sourced from the file ‘EB ++AdjustPANCAN_IlluminaHiSeq_RNASeqV2.geneExp.tsv’ (synapse id: syn4976363). Briefly, the matrices contain batch-corrected values of the upper-quartile (UQ) normalized RSEM estimated counts data from Broad firehose (tumor.uncv2.mRNAseq_RSEM_all.txt). Missing values due to the batch effect correction process were imputed using <italic>K</italic>-nearest neighbors (KNN). For each tumor type, values were imputed based on gene vectors under the assumption that genes are similarly expressed between samples of the same tumor type. For genes with less than 60% of samples with missing values, values were imputed using the logarithmic (base 2) of the gene expression value plus one, and subsequently back-transformed when writing the imputed matrices to file. The resulting gene expression matrix used for the Boruta analysis described below consisted of 870 TCGA samples and 16,309 protein-coding genes (based on ‘hgnc_complete_set.txt’ downloaded from HGNC on 7.24.2018). To generate a RSEM raw counts matrix for the DESeq2 analysis described below, mRNAseq_Preprocess.Level_3 data was downloaded from Broad Firehose (tumor.uncv2.mRNAseq_raw_counts.txt).</p></sec><sec id="s4-3"><title>Boruta analysis</title><p>To identify a minimal set of genes whose expression values were predictive of the sample being ecDNA(+), we used Boruta (<xref ref-type="bibr" rid="bib30">Kursa and Rudnicki, 2010</xref>), an automated feature selection algorithm that utilizes multiple iterations of the random forest classifier to determine the statistical significance of selected features. The algorithm is terminated when all features are categorized as ‘confirmed’ or ‘rejected,’ or until the user-defined number of iterations is reached. In our modified version of the BorutaPy python package (6.21.2021; <ext-link ext-link-type="uri" xlink:href="https://github.com/scikit-learn-contrib/boruta_py">https://github.com/scikit-learn-contrib/boruta_py</ext-link>), we set the maximum number of iterations to 400, a stagnant count maximum of 5, and a tentative count minimum of 50. This translates to the termination of Boruta if 400 iterations are reached, or if the tentative count (features that have yet to be ‘confirmed’ or ‘rejected’) falls to or below 50 and these tentative features remain tentative for five iterations.</p><p>While we use a standard implementation of Boruta, the method is briefly described here for expository purposes. In each iteration, <italic>i</italic>, within a single Boruta trial, the input is a gene expression matrix, <italic>M</italic>, of dimension <italic>r</italic> x c, where <italic>r</italic> is the number of samples and <italic>c</italic> is the number of tentative or confirmed features (i.e. genes). Boruta generates <italic>c</italic> shadow feature vectors by random shuffling of feature vectors in matrix <italic>M</italic>, generating a new matrix <italic>M’</italic> of dimension <italic>r</italic> × 2c. A random forest classifier (class_weight = balanced_subsample, max_depth = 7) is then used to quantify the importance of each feature in separating ecDNA(+) from ecDNA(-) samples. Specifically, there are two possible outcomes for a feature: (1) if the feature scores higher than the best-scoring shadow feature, the feature is considered a ‘hit,’ and (2) if the feature scores lower than the best-scoring shadow feature, the feature is considered a ‘non-hit.’ Features are rejected after <italic>i</italic> iterations, if the number of hits is not significantly higher than expected by chance, using a Bonferroni corrected <italic>p</italic>-value.</p><p>In order to evaluate the ability of selected features in predicting the ecDNA status of a tumor sample, we left out 20% of ecDNA(+) and 20% of ecDNA(-) samples for the hold-out testing dataset in the evaluation procedure described below, and performed Boruta on the gene expression matrix consisting of the remaining 80% of samples. However, due to the unequal representation of ecDNA(+) and ecDNA(-) samples within each of the tumor subtypes, we opted to generate 200 training (80%) and testing (20%) datasets to decrease the bias that may be introduced during random sampling. For each of the 200 datasets, a Boruta analysis was performed on the 80% training data. Features categorized as ‘confirmed’ were considered as Boruta genes for that specific trial. Of the 941 Boruta genes combined across the 200 trials, 408 genes were present in at least 10 of the 200 Boruta trials, and subsequently defined as the Core set of genes in downstream analyses.</p></sec><sec id="s4-4"><title>Highly co-expressed genes</title><p>To identify genes co-expressed with the core set of Boruta genes, hierarchical clustering of the 16,309 genes was performed using the R package pvclust (<xref ref-type="bibr" rid="bib54">Suzuki and Shimodaira, 2006</xref>) (ver. 2.2–0; dist.method=correlation, method = ward.D2, nboot = 1000). A total of 843 significant clusters (AU &gt;0.95) with at least one Boruta gene were selected, consisting of 1375 genes. To obtain the final list of CorEx genes, we apply a minimal count of 10 trials for the gene or 10 trials for the cluster of genes seen in 200 Boruta trials. A cluster is determined to be seen in a Boruta trial if at least one of its members is selected in the trial. This results in 354 clusters, with a total number of 643 genes, of which 408 are Core genes.</p></sec><sec id="s4-5"><title>Evaluation of CorEx genes</title><p>To evaluate a set of genes, <italic>G</italic>, as predictive of ecDNA presence in tumor samples, we performed cross-validation and hyper-parameter tuning on each of the 80% training datasets, and evaluated the final model on the corresponding hold-out 20% testing dataset using the scikit-learn package. Specifically, the gene expression matrix for cross-validation and hyper-parameter tuning consisted of <italic>m</italic> samples from the training dataset and <italic>n</italic> genes from set <italic>G</italic>. RandomizedSearchCV (n_iter = 50, cv = 5, scoring = f1) was first used to narrow down a wide range of hyper-parameters for the random forest classifier (RandomForestClassifier, class_weight=’balanced_subsample’), and GridSearchCV (cv = StratifiedKFold(n_splits = 5, shuffle = True), scoring = f1) was then used to test every combination of a smaller range of hyper-parameters given the best parameters from RandomizedSearchCV. The hyper-parameters tuned (initial values) include the number of trees in the forest (n_estimators: np.linspace(100, 2000, num = 10)), the maximum depth of the tree (max_depth: None, np.linspace(10, 100, num = 10)), the minimum number of samples required to split an internal node (min_samples_split: (2, 5, 10)), and the minimum number of samples required to be at a leaf node (min_samples_leaf: (1, 2, 4)). The best estimator from the GridSearchCV hyper-parameter tuning was then evaluated on the 20% testing dataset, where the gene expression matrix consisted of <italic>m</italic> samples from the testing dataset and <italic>n</italic> genes from set <italic>G</italic>. Performing this procedure on each of the 200 training/testing datasets resulted in 200 data points for each of the three metrics computed using sklearn.metrics: precision_score, recall_score, and average_precision_score (AUPR).</p><p>We performed this procedure on the following sets of genes, <italic>G</italic>: the 408 Core genes, 408 randomly selected genes, the 643 CorEx genes, 643 randomly selected genes, a set of 643 most differentially expressed genes based on the absolute log-fold change estimates from a conventional DE analysis using DESeq2 (<xref ref-type="bibr" rid="bib37">Love et al., 2014</xref>) as described below, and the set of 3012 significant genes from the GLM analysis described below.</p></sec><sec id="s4-6"><title>Generalized linear model (GLM) analysis</title><p>We tested each of the 16,309 genes independently in a separate logistic regression model using the glm() function in the R stats package (v4.2.0), and retained genes that were significant (<italic>p</italic>-value 0.01). Specifically, the model was defined as glm(<inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>y</mml:mi><mml:mspace width="thinmathspace"/><mml:mo>∼</mml:mo><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, data = <inline-formula><mml:math id="inf3"><mml:mi>M</mml:mi></mml:math></inline-formula>, family = binomial(link = 'logit')), where <italic>y</italic> is the response vector where  <inline-formula><mml:math id="inf4"><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> =1 if sample <inline-formula><mml:math id="inf5"><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mn>870</mml:mn><mml:mo>}</mml:mo></mml:math></inline-formula> is ecDNA(+) and <italic>y</italic><sub><italic>i</italic></sub> =0 otherwise, <inline-formula><mml:math id="inf6"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the vector of expression values for gene <inline-formula><mml:math id="inf7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mn>16309</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> in samples <inline-formula><mml:math id="inf8"><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mn>870</mml:mn><mml:mo>}</mml:mo></mml:math></inline-formula>, <italic>t</italic> is the covariate vector representing the tumor subtypes of samples <inline-formula><mml:math id="inf9"><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mn>870</mml:mn><mml:mo>}</mml:mo></mml:math></inline-formula>, and <inline-formula><mml:math id="inf10"><mml:mi>M</mml:mi></mml:math></inline-formula> is the data matrix containing values of gene expression, tumor subtype, and ecDNA status for all samples. The equation for the binomial logistic regression described above is formulated as <inline-formula><mml:math id="inf11"><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>β</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>β</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>β</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> , where <italic>p</italic> is the probability that the dependent variable <italic>y</italic> is 1, <inline-formula><mml:math id="inf12"><mml:mi>X</mml:mi></mml:math></inline-formula> are the independent variables, and <inline-formula><mml:math id="inf13"><mml:mi>β</mml:mi></mml:math></inline-formula> are the coefficients of the model. In this case, <italic>k</italic>=1 represents the independent variable gene <italic>j</italic> and <italic>k</italic>=2 represents the tumor subtype covariate <italic>t</italic>. Of the 16,309 genes tested independently, 3012 genes were significant at <italic>p</italic>-value &lt;0.01.</p></sec><sec id="s4-7"><title>Default DE analysis</title><p>We performed a default DESeq2 (<xref ref-type="bibr" rid="bib37">Love et al., 2014</xref>) (R package, ver. 1.36.0) analysis to obtain shrunken maximum <italic>a posteriori</italic> (MAP) log-fold change estimates for effect size (i.e. LFC). Specifically, to obtain (i) LFC effect size values per gene for integration with its Cliff’s delta effect size value when determining if a gene is up- or down-regulated in ecDNA(+) samples, and (ii) a list of <italic>n</italic> top-ranked genes by absolute value of the LFC (with application of an adjusted <italic>p</italic>-value &lt;0.05 cutoff and LFC threshold of <inline-formula><mml:math id="inf14"><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mfenced separators="|"><mml:mrow><mml:mn>1.1</mml:mn></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mn>0.13</mml:mn></mml:math></inline-formula>) for use in comparison against genes selected as important in the prediction of ecDNA in samples. For comparisons against Core genes, <italic>n</italic> is set to 408, and for comparisons against CorEx genes, <italic>n</italic>, is set to 643.</p><p>To obtain the LFC effect size metric between ecDNA(+) vs. ecDNA(-) samples for each gene’s expression, we fed as input to DESeq2 a matrix of raw RSEM estimated counts. To take into account batch effects, we included the center and platform information of samples, downloaded from synapse id syn4976363 (EB ++GeneExpAnnotation.tsv), in the design of the DESeq object:</p><p><code xml:space="preserve">DESeq_object &lt;- DESeqDataSetFromMatrix(countData =
AC_0_4_9_TCGA_matrix, colData = coldata, design =
~batch + condition)</code></p><p>To compute results, the lfcThreshold was set to <inline-formula><mml:math id="inf15"><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>(</mml:mo><mml:mn>1.1</mml:mn><mml:mo>)</mml:mo></mml:math></inline-formula> for an accurate computation of <italic>p</italic>-values and the contrast set to c(‘condition,’ ‘ecDNA(+),’ ‘ecDNA(-)’) to obtain the logarithmic fold change of the form <inline-formula><mml:math id="inf16"><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>D</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mo>+</mml:mo></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>D</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced></mml:math></inline-formula> . By setting the lfcThreshold, the null hypothesis tested is that <inline-formula><mml:math id="inf17"><mml:mfenced open="|" close="|" separators="|"><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:mfenced><mml:mo>≤</mml:mo><mml:mi>θ</mml:mi></mml:math></inline-formula>, where <inline-formula><mml:math id="inf18"><mml:mi>θ</mml:mi><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mn>1.1</mml:mn><mml:mo>)</mml:mo></mml:math></inline-formula> , and the alternative hypothesis is that <inline-formula><mml:math id="inf19"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&gt;</mml:mo><mml:mi>θ</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. A <inline-formula><mml:math id="inf20"><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mfenced separators="|"><mml:mrow><mml:mn>1.1</mml:mn></mml:mrow></mml:mfenced></mml:math></inline-formula> value is chosen as the minimal value/negligible effect size threshold as it represents a 10% fold-change, and anything below this fold-change would likely not be of biological interest (<xref ref-type="bibr" rid="bib40">McCarthy and Smyth, 2009</xref>). The specific commands run are as follows:</p><p><code xml:space="preserve">DESeq_object$condition &lt;- relevel(DESeq_object$condition, ref = 
&quot;Non_ecDNA&quot;)
DESeq_object &lt;- DESeq(DESeq_object) 
results &lt;- results(DESeq_object, lfcThreshold = log2(1.1), 
contrast = c(&quot;condition&quot;,&quot;ecDNA&quot;,&quot;Non_ecDNA&quot;))</code></p><p>To obtain the shrunken MAP log-fold change estimates, we used the lfcShrink function provided in DESeq2, using the default apeglm method for the empirical Bayes shrinkage procedure (<xref ref-type="bibr" rid="bib66">Zhu et al., 2019</xref>):</p><p><code xml:space="preserve">lfcShrink(DESeq_object, lfcThreshold = log2(1.1), 
coef=&quot;condition_ecDNA_vs_Non_ecDNA&quot;, type=&quot;apeglm&quot;)</code></p></sec><sec id="s4-8"><title>Up- or down-regulated genes in ecDNA(+) samples</title><p>To categorize genes as ‘up-’ or ‘down-’ regulated in ecDNA(+) samples, we integrated two effect size metrics, Cliff’s delta (<italic>d</italic>) (<xref ref-type="bibr" rid="bib9">Cliff, 1993</xref>; <xref ref-type="bibr" rid="bib10">Cliff, 1996</xref>) and the DESeq2 shrunken MAPlog-fold change estimate (LFC <xref ref-type="bibr" rid="bib37">Love et al., 2014</xref>). Effect size is a measure of the magnitude of deviation from the null hypothesis, and unlike <italic>p</italic>-values, has the advantage of not being impacted by sample size (<xref ref-type="bibr" rid="bib48">Romano et al., 2006</xref>). This property is especially useful when comparing effect size values of a gene between tests where sample sizes differ. Comparing <italic>p</italic>-values between such tests would be invalid.</p><p>Cliff’s delta, <italic>d</italic>, is a non-parametric measure of the separation between two distributions and ranges from –1 to 1. Given two distributions, <inline-formula><mml:math id="inf21"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>X</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf22"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, comparisons are made between each of <italic>m</italic> values in <inline-formula><mml:math id="inf23"><mml:mi>X</mml:mi></mml:math></inline-formula> and <italic>n</italic> values in <inline-formula><mml:math id="inf24"><mml:mi>Y</mml:mi></mml:math></inline-formula>. Cliff’s delta is computed as <inline-formula><mml:math id="inf25"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&gt;</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&lt;</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> , where <inline-formula><mml:math id="inf26"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&gt;</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is the number of times a member of <italic>X</italic> is greater than a member of <italic>Y</italic>, and <inline-formula><mml:math id="inf27"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&lt;</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is the number of times a member of <italic>X</italic> is less than a member of <italic>Y (</italic><xref ref-type="bibr" rid="bib9">Cliff, 1993</xref>). A negative <italic>d</italic> indicates that values in <italic>Y</italic> tend to be higher than <italic>X</italic>, while a positive <italic>d</italic> indicates that values in <italic>X</italic> tend to be higher than <italic>Y</italic>. The magnitude of the effect size of Cliff’s delta can be separated into four levels: <inline-formula><mml:math id="inf28"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>d</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mn>0.147</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> for negligible effects, <inline-formula><mml:math id="inf29"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.147</mml:mn><mml:mo>≤</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mi>d</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mn>0.33</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> for small effects, <inline-formula><mml:math id="inf30"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>0.33</mml:mn><mml:mo>≤</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mi>d</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mn>0.474</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> for medium effects, and <inline-formula><mml:math id="inf31"><mml:mfenced open="|" close="|" separators="|"><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:mfenced><mml:mo>≥</mml:mo><mml:mn>0.474</mml:mn></mml:math></inline-formula> for large effects (<xref ref-type="bibr" rid="bib48">Romano et al., 2006</xref>). The python package used to compute Cliff’s delta values can be accessed at <ext-link ext-link-type="uri" xlink:href="https://github.com/neilernst/cliffsDelta">https://github.com/neilernst/cliffsDelta</ext-link> (copy archived at <xref ref-type="bibr" rid="bib13">Ernst, 2021</xref>). The input values used to compute Cliff’s delta are log-transformed normalized gene expression values plus one as described in the ‘Gene expression datasets’ section of methods.</p><p>The DESeq2 LFC is computed as described above. To allow integration of the Cliff’s delta effect size with the DESeq2 LFC, we also separated the LFC values into four levels: <inline-formula><mml:math id="inf32"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>1.1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> for negligible effects, <inline-formula><mml:math id="inf33"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>1.1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>≤</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>1.5</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> for small effects, <inline-formula><mml:math id="inf34"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>1.5</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>≤</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mn>2</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> or medium effects, and <inline-formula><mml:math id="inf35"><mml:mfenced open="|" close="|" separators="|"><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:mfenced><mml:mo>≥</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>(</mml:mo><mml:mn>2</mml:mn><mml:mo>)</mml:mo></mml:math></inline-formula> for large effects.</p><p>For each gene, <italic>g</italic>, whether the gene is up-regulated or down-regulated in ecDNA(+) samples is determined by the signage and magnitude of its effect sizes <inline-formula><mml:math id="inf36"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf37"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> . The initial criteria for <inline-formula><mml:math id="inf38"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf39"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to be used as a determinant in the direction of a gene is for it to have a magnitude larger than that of a negligible effect. If the signage of <inline-formula><mml:math id="inf40"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf41"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are both positive, gene <italic>g</italic> is considered up-regulated in ecDNA(+). If only a single value has a magnitude larger than the negligible effect threshold (e.g, <inline-formula><mml:math id="inf42"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf43"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>0.1</mml:mn></mml:math></inline-formula>), gene <italic>g</italic> is considered up-regulated in ecDNA(+). In the case of conflicting signages between the two values, the effect size with a larger magnitude takes precedence. For example, if <inline-formula><mml:math id="inf44"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>0.2</mml:mn></mml:math></inline-formula> and <inline-formula><mml:math id="inf45"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.847</mml:mn></mml:math></inline-formula>, given that <inline-formula><mml:math id="inf46"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> has a small negative effect and <inline-formula><mml:math id="inf47"><mml:mi>L</mml:mi><mml:mi>F</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> has a large positive effect, gene <italic>g</italic> is considered up-regulated in ecDNA(+) samples.</p></sec><sec id="s4-9"><title>Tumor heatmap</title><p>A Cliff’s delta effect size matrix representing 643 CorEx genes was generated to compare TCGA with tumor expression patterns. For each of the 11 tumor types with at least 10 ecDNA(+) and at least 10 ecDNA(-) samples, we re-computed Cliff’s delta. Using a Fisher’s exact test (fisher_exact function from the scipy.stats python package; alternative hypothesis: two-sided), we tested the null-hypothesis of whether the up- and down- directionality of CorEx genes in TCGA vs. each tumor were independent of each other. The contingency table is as below. The directionality of a gene (up or down) was based solely on the signage of the gene’s Cliff’s delta effect size value.</p><table-wrap id="inlinetable1" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom" rowspan="2"/><th align="left" valign="bottom" rowspan="2"/><th align="left" valign="bottom" colspan="2">Tumor</th></tr><tr><th align="left" valign="bottom">UP</th><th align="left" valign="bottom">DOWN</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="2">TCGA</td><td align="left" valign="bottom">UP</td><td align="left" valign="bottom">a</td><td align="left" valign="bottom">b</td></tr><tr><td align="left" valign="bottom">DOWN</td><td align="left" valign="bottom">c</td><td align="left" valign="bottom">d</td></tr></tbody></table></table-wrap></sec><sec id="s4-10"><title>Gene ontology (GO) enrichment analysis</title><p>To identify Gene Ontology Biological Process (GOBP) terms that were enriched in either the set of down-regulated or up-regulated CorEx genes, we applied one-sided Fisher’s exact tests (alternative=‘greater;’ scipy.stats python package) on 2x2 contingency tables for each GOBP term. Specifically, in the contingency table below, N is the total number of genes in the universe (i.e. 16 k for the number of genes measured in the RNAseq data), n is the number of DE genes (either up- or down-regulated in ecDNA(+) samples), m is the number of genes belonging to the GOBP term as defined by gene sets from MSigDB (c5.go.bp.v7.5.1.entrez), and k is the number of DE genes that belong to the GOBP term. The false discovery rate was controlled at 5% and adjusted <italic>p</italic>-values were computed using the Benjamini-Hochberg procedure (fdr correction from python statsmodels package). A final set of GOBP terms with adjusted <italic>p</italic>-value &lt;0.05 was used for downstream analysis.</p><table-wrap id="inlinetable2" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom">DE</th><th align="left" valign="bottom">Non-DE</th></tr></thead><tbody><tr><td align="left" valign="bottom">Inside GOBP term</td><td align="left" valign="bottom">k</td><td align="left" valign="bottom">m-k</td></tr><tr><td align="left" valign="bottom">Outside GOBP term</td><td align="left" valign="bottom">n-k</td><td align="left" valign="bottom">N+k-n-m</td></tr></tbody></table></table-wrap></sec><sec id="s4-11"><title>Clustering gene sets</title><p>To cluster enriched gene sets into categories for visualization purposes, Cohen’s kappa coefficient (python sklearn cohen_kappa_score) was used to determine term-term ‘connectivity’ (agreement of term-term pairs) – an approach described in the DAVID paper (<xref ref-type="bibr" rid="bib25">Huang et al., 2007</xref>). Given a <inline-formula><mml:math id="inf48"><mml:mi>r</mml:mi><mml:mo>×</mml:mo><mml:mi>c</mml:mi></mml:math></inline-formula> binary matrix, where enriched GOBP terms are rows, CorEx genes are columns, and values are 1 if a CorEx gene is part of the GOBP term or 0 otherwise, kappa scores were computed between each pair of terms, where term <inline-formula><mml:math id="inf49"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>∈</mml:mo><mml:mi>t</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf50"><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mo>{</mml:mo><mml:mn>1,2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>.</mml:mo><mml:mi>n</mml:mi><mml:mo>}</mml:mo></mml:math></inline-formula>: <inline-formula><mml:math id="inf51"><mml:mi>K</mml:mi><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mo>_</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:math></inline-formula>, where <inline-formula><mml:math id="inf52"><mml:mi>x</mml:mi><mml:mo>∈</mml:mo><mml:mi>i</mml:mi></mml:math></inline-formula> and y <inline-formula><mml:math id="inf53"><mml:mo>∈</mml:mo><mml:mi>i</mml:mi></mml:math></inline-formula>.</p><p>Each term, <inline-formula><mml:math id="inf54"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> , formed an initial seeding group, <inline-formula><mml:math id="inf55"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> , where a term <inline-formula><mml:math id="inf56"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is part of <inline-formula><mml:math id="inf57"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> if <inline-formula><mml:math id="inf58"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>K</mml:mi><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>≥</mml:mo><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. If at least 50% of term-term pairs in <inline-formula><mml:math id="inf59"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> have a <inline-formula><mml:math id="inf60"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>K</mml:mi><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>≥</mml:mo><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, the initial seeding group <inline-formula><mml:math id="inf61"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is retained for the next step. The second criteria ensures that terms within the same seeding group have strong interconnectivity. An iterative merging of seeding groups then follows: groups sharing p% or more members are merged. The representative term for each group was determined as the member with the highest interconnectivity score with other members of the group. After the automatic grouping process, a manual inspection leads to the merging of outliers or smaller groups into representative groups. We used a kappa score threshold of 0.5, and a condition of ≥25% of shared members when merging for the down-regulated genes, and a kappa score threshold of 0.6 and a condition of p≥50% of shared members when merging for the up-regulated genes.</p></sec><sec id="s4-12"><title>DDR pathway genes</title><p>We hand-curated 88 genes for double-stranded break DNA damage repair pathways (a-EJ, HR, c-NHEJ, SSA) via an extensive literature search, and added an additional 51 genes from the following MSigDB (c5.go.bp.v2022.1.Hs.entrez.gmt) GO biological process terms: GO:0097680 (double-strand break repair via classical nonhomologous end joining), GO:0097681 (double-strand break repair via alternative nonhomologous end joining), GO:1905168 (positive regulation of double-strand break repair via homologous recombination), and GO:0045002 (double strand break repair via single-strand annealing). This resulted in a final list of 129 genes. The directionality of the genes, as either up- or down-regulated in ecDNA(+) samples, is based on the full set of 1440 samples, consisting of 243 ecDNA(+) and 1197 ecDNA(-) samples representing 24 tumor types (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1M</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1N</xref>).</p><p>To test whether the number of genes passing our effect size thresholds for all genes 5256 (UP) and 5251 (DOWN) was significantly different from the up-/down-regulated genes implicated in each of the four pathways, we performed a Fisher’s exact test (fisher_exact function from the scipy.stats python package) on the contingency table below, where <italic>c</italic> and <italic>d</italic> are the up- and down-regulated genes for each of the pathways tested.</p><p>Contingency table:</p><table-wrap id="inlinetable3" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom">UP</th><th align="left" valign="bottom">DOWN</th></tr></thead><tbody><tr><td align="left" valign="bottom">All genes</td><td align="char" char="." valign="bottom">5256</td><td align="char" char="." valign="bottom">5251</td></tr><tr><td align="left" valign="bottom">Pathway</td><td align="left" valign="bottom"><italic>c</italic></td><td align="left" valign="bottom"><italic>d</italic></td></tr></tbody></table></table-wrap><p>Pathway values:</p><table-wrap id="inlinetable4" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom"><italic><bold>c</bold></italic></th><th align="left" valign="bottom"><italic><bold>d</bold></italic></th></tr></thead><tbody><tr><td align="left" valign="bottom">c-NHEJ</td><td align="char" char="." valign="bottom">14</td><td align="char" char="." valign="bottom">7</td></tr><tr><td align="left" valign="bottom">Alt-EJ</td><td align="char" char="." valign="bottom">11</td><td align="char" char="." valign="bottom">1</td></tr><tr><td align="left" valign="bottom">HR</td><td align="char" char="." valign="bottom">46</td><td align="char" char="." valign="bottom">8</td></tr><tr><td align="left" valign="bottom">SSA</td><td align="char" char="." valign="bottom">11</td><td align="char" char="." valign="bottom">1</td></tr></tbody></table></table-wrap></sec><sec id="s4-13"><title>Physical presence on amplicons</title><p>To determine the physical presence of a gene on an amplicon, gene coordinates listed in the gene annotations file GRCh37/human_hg19_september_2011/Genes_July_2010_hg19.gff, downloaded from the AA repo on 3/21/2022, were mapped to amplicon genomic intervals (bed files). A gene is determined to be physically present on an amplicon if its genomic coordinates are fully encompassed within the amplicon genomic intervals.</p></sec><sec id="s4-14"><title>Mutational analysis</title><p>We pulled the list of mutations from the open-access version of the MC3 dataset (<ext-link ext-link-type="uri" xlink:href="https://ellrottlab.org/project/mc3/">https://ellrottlab.org/project/mc3/</ext-link>), and then investigated differences between ecDNA(+) and ecDNA(-) samples. Synonymous mutations were excluded when calculating the mutation burden. For the differentially mutated gene analysis, only damaging mutations were selected by using snpEff (<xref ref-type="bibr" rid="bib8">Cingolani et al., 2012</xref>) annotation which was originally included in the MC3 dataset. First, mutations annotated as HIGH in the IMPACT column were selected to obtain frameshift INDELs and stop gain SNVs. Next, mutations predicted to be damaging by SIFT (<xref ref-type="bibr" rid="bib44">Ng and Henikoff, 2003</xref>) and PolyPhen2 (<xref ref-type="bibr" rid="bib1">Adzhubei et al., 2013</xref>), were selected to obtain damaging missense mutations. Finally, we generated 2-by-2 contingency tables for each gene with cells <italic>a</italic>, <italic>b</italic>, <italic>c</italic>, and <italic>d</italic> representing the number of individuals with and without damaging mutations in ecDNA(+) and (-) tumors. The odds ratios were computed as <inline-formula><mml:math id="inf62"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>O</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>a</mml:mi><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> , where <italic>a</italic> is the number of individuals in ecDNA(+) with the mutation, <italic>b</italic> is the number of individuals in ecDNA(+) without the mutation, <italic>c</italic> is the number of individuals in ecDNA(-) with the mutation, and <italic>d</italic> is the number of individuals in ecDNA(-) without the mutation. To determine if a gene contained mutations that were implicated in cancer, we checked genes against the Cancer Gene Census (CGC) database (v97) (<xref ref-type="bibr" rid="bib51">Sondka et al., 2018</xref>), marking genes in the database as CGC and those that were not as non-CGC.</p></sec><sec id="s4-15"><title>Classification with mutational status</title><p>First, we created a binary matrix representing whether a gene is damaged or not, from the MC3 damaging mutation set described as above. Then we divided the whole matrix into 80% of the training set and 20% of the test set. Hyperopt was applied to the training set to select the best parameters for the XGBoost (<xref ref-type="bibr" rid="bib7">Chen and Guestrin, 2016</xref>) model. The optimal parameters estimated by Hyperopt (eta = 0.1, max_depth = 6, min_child_weight = 3.0, scale_pos_weight = 4.9) were input to the XGBoost model, and the ecDNA status of each sample were also input as an answer set. The performance of the model was checked by inputting the 20% test set to the model and comparing the output result with the answer. Principal component analysis was also performed with the same mutational matrix as above, using the scikit.learn package.</p></sec><sec id="s4-16"><title>Ranking of CorEx genes</title><p>To rank CorEx genes by importance, we computed harmonic mean rank values based on three categories: (a) the average Gini importance statistic or mean decrease impurity (MDI) MDI (feature importance) values extracted from the trained random forest models on 200 training sets during the evaluation method described above, (b) the number of Boruta trials (out of 200) that a gene is selected in, rounded to 2-digits, and (c) the number of Boruta trials (out of 200) that a gene is selected in when counting by cluster, rounded to 2-digits. The ranks for each category are adjusted separately so that genes with the same value share the same rank value. For example, if using the Boruta trial count as the rank value:</p><table-wrap id="inlinetable5" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top"/><th align="left" valign="top">Trial count</th><th align="left" valign="top">Round (Trial count /10.0)</th><th align="left" valign="top">Rank</th><th align="left" valign="top">Adjusted rank</th></tr></thead><tbody><tr><td align="left" valign="top">Gene A</td><td align="char" char="." valign="top">200</td><td align="char" char="." valign="top">20</td><td align="char" char="." valign="top">1</td><td align="char" char="." valign="top">1</td></tr><tr><td align="left" valign="top">Gene B</td><td align="char" char="." valign="top">200</td><td align="char" char="." valign="top">20</td><td align="char" char="." valign="top">2</td><td align="char" char="." valign="top">1</td></tr><tr><td align="left" valign="top">Gene C</td><td align="char" char="." valign="top">200</td><td align="char" char="." valign="top">20</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">1</td></tr><tr><td align="left" valign="top">Gene D</td><td align="char" char="." valign="top">187</td><td align="char" char="." valign="top">19</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">4</td></tr></tbody></table></table-wrap><p>The harmonic mean rank is defined as:<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mi>h</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mspace width="thinmathspace"/><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>3</mml:mn><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>a</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>b</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>c</mml:mi></mml:mfrac></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p></sec><sec id="s4-17"><title>Tumor immune subtype</title><p>We classified ecDNA(+) and ecDNA(-) samples into the six immune subtype categories (<xref ref-type="bibr" rid="bib56">Thorsson et al., 2018</xref>) provided in Table S6 from <xref ref-type="bibr" rid="bib3">Bagaev et al., 2021</xref>.</p></sec><sec id="s4-18"><title>Impact of tumor purity on CorEx gene expression</title><p>To investigate the effects of the presence of non-cancer tissue (impurity) in bulk RNA-seq samples on the analyses performed in this study, we utilized the consensus measurement of purity estimations (CPE) for TCGA samples from a publication by <xref ref-type="bibr" rid="bib2">Aran et al., 2015</xref>. Of the 870 TCGA samples (234 ecDNA(+), 636 ecDNA(-)) with gene expression (RNA-seq) data, 701 samples (174 ecDNA(+), 527 ecDNA(-)) were assigned a CPE value by Aran et al.,. To determine if the presence of undetected ecDNA in ecDNA(-) samples would confound the results by reducing the discriminating power of genes, we measured the expression directionality of CorEx genes in all samples (n=870) versus samples which had a high tumor purity (CPE ≥0.8, n=287). Specifically, <italic>p</italic>-values were obtained by performing Mann-Whitney U rank tests (scipy.stats python package) on gene expression values of ecDNA(+) and ecDNA(-) samples for both the 870 TCGA samples and 287 TCGA samples with high tumor purity. Genes with a significantly (p-value ≤0.05) higher expression in ecDNA(+) samples (alternative=‘greater’) were labeled as ‘UP,’ while genes with a significantly lower expression in ecDNA(+) samples (alternative=‘less’) were labeled as ‘DOWN.’ To generate a plot that compared the gene directionality of all samples vs. high-purity samples using <italic>p</italic>-values, a function <italic>F</italic> was applied to <italic>p</italic>-values. Specifically, <inline-formula><mml:math id="inf63"><mml:mi>F</mml:mi><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>d</mml:mi><mml:mo>⋅</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:math></inline-formula>, where <italic>p</italic> is the <italic>p</italic>-value and <inline-formula><mml:math id="inf64"><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> if directionality is ‘DOWN’ and <inline-formula><mml:math id="inf65"><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> if directionality is ‘UP.’</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>J.L. receives compensation as a consultant for Boundless Bio</p></fn><fn fn-type="COI-statement" id="conf3"><p>Reviewing editor, eLife</p></fn><fn fn-type="COI-statement" id="conf4"><p>S. Wu is a member of the scientific advisory board of Dimension Genomics Inc</p></fn><fn fn-type="COI-statement" id="conf5"><p>P.S.M. is a co-founder and advisor of Boundless Bio. J.L. receives compensation as a consultant for Boundless Bio</p></fn><fn fn-type="COI-statement" id="conf6"><p>V.B. is a co-founder, paid consultant, SAB member and has equity interest in Boundless Bio, Inc and Abterra Biosciences, Inc</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Methodology, Writing – original draft, Writing – review and editing, All analyses were performed by M.S.L., apart from the following ones. Mutational analysis was performed by S.J. The amplicon classifications were performed by J.L</p></fn><fn fn-type="con" id="con2"><p>Data curation, Formal analysis, Writing – original draft</p></fn><fn fn-type="con" id="con3"><p>Data curation, Writing – original draft</p></fn><fn fn-type="con" id="con4"><p>Supervision, Writing – review and editing</p></fn><fn fn-type="con" id="con5"><p>Supervision, Writing – review and editing</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Supervision, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con7"><p>Conceptualization, Supervision, Funding acquisition, Methodology, Writing – original draft, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Supplementary Tables.</title><p><bold>(A</bold>) CorEx genes. (<bold>B</bold>) Extrachromosomal DNA (ecDNA) status of 870 The Cancer Genome Atlas (TCGA) samples across 14 tumor types. (<bold>C</bold>) Highly co-expressed gene clusters identified using multiscale bootstrap resampling. (<bold>D</bold>) Cluster #3 and #74 members. (<bold>E</bold>) Cliff’s delta values of 643 CorEx genes in TCGA samples, and in 11 tumor types with at least 10 ecDNA(+) and 10 ecDNA(-) samples each. (<bold>F</bold>) Top-|LFC| genes: 643 most significantly differentially expressed genes based on logarithmic fold changes from a DESeq2 analysis. (<bold>G</bold>) Up-/down-regulated genes in 870 TCGA samples. (<bold>H</bold>) Gene Ontology (GO) biological processes enriched in up-regulated CorEx genes. (<bold>I</bold>) Clustering of GO biological processes enriched in up-regulated CorEx genes into 11 broad categories. (<bold>J</bold>) GO biological processes enriched in Cluster #3 genes. (<bold>K</bold>) Up-regulated CorEx genes unique to biological process categories. (<bold>L</bold>) Hand-curated list of 129 genes involved in DSB repair pathways. (<bold>M</bold>) ecDNA status of 1440 TCGA samples across 24 tumor types. (<bold>N</bold>) Up-/down-regulated genes in 1,440 TCGA samples. (<bold>O</bold>) GO biological processes enriched in down-regulated CorEx genes. (<bold>P</bold>) Clustering of GO biological processes enriched in down-regulated CorEx genes into seven broad categories. (<bold>Q</bold>) Down-regulated CorEx genes in NF-κB signaling (GO:0007249) involved in pro-apoptotic caspase activation. (<bold>R</bold>) Differentially mutated genes in ecDNA(+) and ecDNA(-) samples and their odds ratios.</p></caption><media xlink:href="elife-88895-supp1-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-88895-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Public data from cBioPortal and Broad Firehose were used for this study. Scripts to generate CorEx genes are located on GitHub at <ext-link ext-link-type="uri" xlink:href="https://github.com/miinslin/ecDNA_Gene_Expression">https://github.com/miinslin/ecDNA_Gene_Expression</ext-link> (copy archived at <xref ref-type="bibr" rid="bib36">Lin, 2024</xref>).</p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset1"><person-group person-group-type="author"><name><surname>Akbani</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><data-title>EB++AdjustPANCAN_IlluminaHiSeq_RNASeqV2.geneExp.tsv</data-title><source>synapse</source><pub-id pub-id-type="accession" xlink:href="https://www.synapse.org/#!Synapse:syn4976363">syn4976363</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><collab>TCGA</collab></person-group><year iso-8601-date="2018">2018</year><data-title>mRNAseq_Preprocess.Level_3</data-title><source>Broad Firehose</source><pub-id pub-id-type="accession" xlink:href="https://gdac.broadinstitute.org/">*.uncv2.mRNAseq_raw_counts.txt</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adzhubei</surname><given-names>I</given-names></name><name><surname>Jordan</surname><given-names>DM</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Predicting functional effect of human missense mutations using PolyPhen-2</article-title><source>Current Protocols in Human Genetics</source><volume>Chapter 7</volume><elocation-id>Unit7</elocation-id><pub-id pub-id-type="doi">10.1002/0471142905.hg0720s76</pub-id><pub-id pub-id-type="pmid">23315928</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aran</surname><given-names>D</given-names></name><name><surname>Sirota</surname><given-names>M</given-names></name><name><surname>Butte</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Systematic pan-cancer analysis of tumour purity</article-title><source>Nature Communications</source><volume>6</volume><elocation-id>8971</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms9971</pub-id><pub-id pub-id-type="pmid">26634437</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bagaev</surname><given-names>A</given-names></name><name><surname>Kotlov</surname><given-names>N</given-names></name><name><surname>Nomie</surname><given-names>K</given-names></name><name><surname>Svekolkin</surname><given-names>V</given-names></name><name><surname>Gafurov</surname><given-names>A</given-names></name><name><surname>Isaeva</surname><given-names>O</given-names></name><name><surname>Osokin</surname><given-names>N</given-names></name><name><surname>Kozlov</surname><given-names>I</given-names></name><name><surname>Frenkel</surname><given-names>F</given-names></name><name><surname>Gancharova</surname><given-names>O</given-names></name><name><surname>Almog</surname><given-names>N</given-names></name><name><surname>Tsiper</surname><given-names>M</given-names></name><name><surname>Ataullakhanov</surname><given-names>R</given-names></name><name><surname>Fowler</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Conserved pan-cancer microenvironment subtypes predict response to immunotherapy</article-title><source>Cancer Cell</source><volume>39</volume><fpage>845</fpage><lpage>865</lpage><pub-id pub-id-type="doi">10.1016/j.ccell.2021.04.014</pub-id><pub-id pub-id-type="pmid">34019806</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bergstrom</surname><given-names>EN</given-names></name><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Petljak</surname><given-names>M</given-names></name><name><surname>Khandekar</surname><given-names>A</given-names></name><name><surname>Barnes</surname><given-names>M</given-names></name><name><surname>Zhang</surname><given-names>T</given-names></name><name><surname>Steele</surname><given-names>CD</given-names></name><name><surname>Pillay</surname><given-names>N</given-names></name><name><surname>Landi</surname><given-names>MT</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name><name><surname>Harris</surname><given-names>RS</given-names></name><name><surname>Alexandrov</surname><given-names>LB</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Mapping clustered mutations in cancer reveals APOBEC3 mutagenesis of ecDNA</article-title><source>Nature</source><volume>602</volume><fpage>510</fpage><lpage>517</lpage><pub-id pub-id-type="doi">10.1038/s41586-022-04398-6</pub-id><pub-id pub-id-type="pmid">35140399</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>HHY</given-names></name><name><surname>Pannunzio</surname><given-names>NR</given-names></name><name><surname>Adachi</surname><given-names>N</given-names></name><name><surname>Lieber</surname><given-names>MR</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Non-homologous DNA end joining and alternative pathways to double-strand break repair</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>18</volume><fpage>495</fpage><lpage>506</lpage><pub-id pub-id-type="doi">10.1038/nrm.2017.48</pub-id><pub-id pub-id-type="pmid">28512351</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>DS</given-names></name><name><surname>Mellman</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Oncology meets immunology: the cancer-immunity cycle</article-title><source>Immunity</source><volume>39</volume><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.1016/j.immuni.2013.07.012</pub-id><pub-id pub-id-type="pmid">23890059</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>T</given-names></name><name><surname>Guestrin</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>XGBoost: A Scalable Tree Boosting System</article-title><conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name><fpage>785</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cingolani</surname><given-names>P</given-names></name><name><surname>Platts</surname><given-names>A</given-names></name><name><surname>Wang</surname><given-names>LL</given-names></name><name><surname>Coon</surname><given-names>M</given-names></name><name><surname>Nguyen</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>L</given-names></name><name><surname>Land</surname><given-names>SJ</given-names></name><name><surname>Lu</surname><given-names>X</given-names></name><name><surname>Ruden</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A program for annotating and predicting the effects of single nucleotide polymorphisms, SnpEff: SNPs in the genome of <italic>Drosophila melanogaster</italic> strain w1118; iso-2; iso-3</article-title><source>Fly</source><volume>6</volume><fpage>80</fpage><lpage>92</lpage><pub-id pub-id-type="doi">10.4161/fly.19695</pub-id><pub-id pub-id-type="pmid">22728672</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cliff</surname><given-names>N</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>Dominance statistics: Ordinal analyses to answer ordinal questions</article-title><source>Psychological Bulletin</source><volume>114</volume><fpage>494</fpage><lpage>509</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.114.3.494</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cliff</surname><given-names>N</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>Answering ordinal questions with ordinal data using ordinal statistics</article-title><source>Multivariate Behavioral Research</source><volume>31</volume><fpage>331</fpage><lpage>350</lpage><pub-id pub-id-type="doi">10.1207/s15327906mbr3103_4</pub-id><pub-id pub-id-type="pmid">26741071</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daley</surname><given-names>JM</given-names></name><name><surname>Sung</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>53BP1, BRCA1, and the choice between recombination and end joining at DNA double-strand breaks</article-title><source>Molecular and Cellular Biology</source><volume>34</volume><fpage>1380</fpage><lpage>1388</lpage><pub-id pub-id-type="doi">10.1128/MCB.01639-13</pub-id><pub-id pub-id-type="pmid">24469398</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>De Luca</surname><given-names>M</given-names></name><name><surname>Lavia</surname><given-names>P</given-names></name><name><surname>Guarguaglini</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>A functional interplay between Aurora-A, Plk1 and TPX2 at spindle poles: Plk1 controls centrosomal localization of Aurora-A and TPX2 spindle association</article-title><source>Cell Cycle</source><volume>5</volume><fpage>296</fpage><lpage>303</lpage><pub-id pub-id-type="doi">10.4161/cc.5.3.2392</pub-id><pub-id pub-id-type="pmid">16418575</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Ernst</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>cliffsDelta</data-title><version designator="swh:1:rev:8f652b4d0b2c31814a1b93b8f31cc42746359c08">swh:1:rev:8f652b4d0b2c31814a1b93b8f31cc42746359c08</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:2b860d6e741e9c87d05420b71dd0b392a22e9af8;origin=https://github.com/neilernst/cliffsDelta;visit=swh:1:snp:f99ab9063c5021cd27ae3bdf4589768124f01a84;anchor=swh:1:rev:8f652b4d0b2c31814a1b93b8f31cc42746359c08">https://archive.softwareheritage.org/swh:1:dir:2b860d6e741e9c87d05420b71dd0b392a22e9af8;origin=https://github.com/neilernst/cliffsDelta;visit=swh:1:snp:f99ab9063c5021cd27ae3bdf4589768124f01a84;anchor=swh:1:rev:8f652b4d0b2c31814a1b93b8f31cc42746359c08</ext-link></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Xie</surname><given-names>M</given-names></name><name><surname>Ji</surname><given-names>X</given-names></name><name><surname>Luo</surname><given-names>X</given-names></name><name><surname>Huang</surname><given-names>W</given-names></name><name><surname>Xia</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Homeobox genes in cancers: From carcinogenesis to recent therapeutic intervention</article-title><source>Frontiers in Oncology</source><volume>11</volume><elocation-id>770428</elocation-id><pub-id pub-id-type="doi">10.3389/fonc.2021.770428</pub-id><pub-id pub-id-type="pmid">34722321</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Zeng</surname><given-names>Q</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Tu</surname><given-names>Y</given-names></name><name><surname>Chen</surname><given-names>W</given-names></name><name><surname>Shu</surname><given-names>X</given-names></name><name><surname>Wu</surname><given-names>A</given-names></name><name><surname>Xiong</surname><given-names>J</given-names></name><name><surname>Cao</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title><italic>RNF114</italic> silencing inhibits the proliferation and metastasis of gastric cancer</article-title><source>Journal of Cancer</source><volume>13</volume><fpage>565</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.7150/jca.62033</pub-id><pub-id pub-id-type="pmid">35069903</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fish</surname><given-names>L</given-names></name><name><surname>Khoroshkin</surname><given-names>M</given-names></name><name><surname>Navickas</surname><given-names>A</given-names></name><name><surname>Garcia</surname><given-names>K</given-names></name><name><surname>Culbertson</surname><given-names>B</given-names></name><name><surname>Hänisch</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>S</given-names></name><name><surname>Nguyen</surname><given-names>HCB</given-names></name><name><surname>Soto</surname><given-names>LM</given-names></name><name><surname>Dermit</surname><given-names>M</given-names></name><name><surname>Mardakheh</surname><given-names>FK</given-names></name><name><surname>Molina</surname><given-names>H</given-names></name><name><surname>Alarcón</surname><given-names>C</given-names></name><name><surname>Najafabadi</surname><given-names>HS</given-names></name><name><surname>Goodarzi</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>A prometastatic splicing program regulated by SNRPA1 interactions with structured RNA elements</article-title><source>Science</source><volume>372</volume><elocation-id>eabc7531</elocation-id><pub-id pub-id-type="doi">10.1126/science.abc7531</pub-id><pub-id pub-id-type="pmid">33986153</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fouquin</surname><given-names>A</given-names></name><name><surname>Guirouilh-Barbat</surname><given-names>J</given-names></name><name><surname>Lopez</surname><given-names>B</given-names></name><name><surname>Hall</surname><given-names>J</given-names></name><name><surname>Amor-Guéret</surname><given-names>M</given-names></name><name><surname>Pennaneach</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>PARP2 controls double-strand break repair pathway choice by limiting 53BP1 accumulation at DNA damage sites and promoting end-resection</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>12325</fpage><lpage>12339</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx881</pub-id><pub-id pub-id-type="pmid">29036662</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garsed</surname><given-names>DW</given-names></name><name><surname>Marshall</surname><given-names>OJ</given-names></name><name><surname>Corbin</surname><given-names>VDA</given-names></name><name><surname>Hsu</surname><given-names>A</given-names></name><name><surname>Di Stefano</surname><given-names>L</given-names></name><name><surname>Schröder</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Feng</surname><given-names>Z-P</given-names></name><name><surname>Kim</surname><given-names>BW</given-names></name><name><surname>Kowarsky</surname><given-names>M</given-names></name><name><surname>Lansdell</surname><given-names>B</given-names></name><name><surname>Brookwell</surname><given-names>R</given-names></name><name><surname>Myklebost</surname><given-names>O</given-names></name><name><surname>Meza-Zepeda</surname><given-names>L</given-names></name><name><surname>Holloway</surname><given-names>AJ</given-names></name><name><surname>Pedeutour</surname><given-names>F</given-names></name><name><surname>Choo</surname><given-names>KHA</given-names></name><name><surname>Damore</surname><given-names>MA</given-names></name><name><surname>Deans</surname><given-names>AJ</given-names></name><name><surname>Papenfuss</surname><given-names>AT</given-names></name><name><surname>Thomas</surname><given-names>DM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The architecture and evolution of cancer neochromosomes</article-title><source>Cancer Cell</source><volume>26</volume><fpage>653</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1016/j.ccell.2014.09.010</pub-id><pub-id pub-id-type="pmid">25517748</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname><given-names>D</given-names></name><name><surname>Raghavan</surname><given-names>SC</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>20 years of DNA Polymerase μ, the polymerase that still surprises</article-title><source>The FEBS Journal</source><volume>288</volume><fpage>7230</fpage><lpage>7242</lpage><pub-id pub-id-type="doi">10.1111/febs.15852</pub-id><pub-id pub-id-type="pmid">33786971</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname><given-names>Y</given-names></name><name><surname>Shi</surname><given-names>J</given-names></name><name><surname>Zhao</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Multidimensional analysis of the role of charged multivesicular body protein 7 in pan-cancer</article-title><source>International Journal of General Medicine</source><volume>14</volume><fpage>7907</fpage><lpage>7923</lpage><pub-id pub-id-type="doi">10.2147/IJGM.S337876</pub-id><pub-id pub-id-type="pmid">34785938</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname><given-names>RA</given-names></name><name><surname>Shah</surname><given-names>N</given-names></name><name><surname>Wang</surname><given-names>KC</given-names></name><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Horlings</surname><given-names>HM</given-names></name><name><surname>Wong</surname><given-names>DJ</given-names></name><name><surname>Tsai</surname><given-names>M-C</given-names></name><name><surname>Hung</surname><given-names>T</given-names></name><name><surname>Argani</surname><given-names>P</given-names></name><name><surname>Rinn</surname><given-names>JL</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Brzoska</surname><given-names>P</given-names></name><name><surname>Kong</surname><given-names>B</given-names></name><name><surname>Li</surname><given-names>R</given-names></name><name><surname>West</surname><given-names>RB</given-names></name><name><surname>van de Vijver</surname><given-names>MJ</given-names></name><name><surname>Sukumar</surname><given-names>S</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Long non-coding RNA HOTAIR reprograms chromatin state to promote cancer metastasis</article-title><source>Nature</source><volume>464</volume><fpage>1071</fpage><lpage>1076</lpage><pub-id pub-id-type="doi">10.1038/nature08975</pub-id><pub-id pub-id-type="pmid">20393566</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hadi</surname><given-names>K</given-names></name><name><surname>Yao</surname><given-names>X</given-names></name><name><surname>Behr</surname><given-names>JM</given-names></name><name><surname>Deshpande</surname><given-names>A</given-names></name><name><surname>Xanthopoulakis</surname><given-names>C</given-names></name><name><surname>Tian</surname><given-names>H</given-names></name><name><surname>Kudman</surname><given-names>S</given-names></name><name><surname>Rosiene</surname><given-names>J</given-names></name><name><surname>Darmofal</surname><given-names>M</given-names></name><name><surname>DeRose</surname><given-names>J</given-names></name><name><surname>Mortensen</surname><given-names>R</given-names></name><name><surname>Adney</surname><given-names>EM</given-names></name><name><surname>Shaiber</surname><given-names>A</given-names></name><name><surname>Gajic</surname><given-names>Z</given-names></name><name><surname>Sigouros</surname><given-names>M</given-names></name><name><surname>Eng</surname><given-names>K</given-names></name><name><surname>Wala</surname><given-names>JA</given-names></name><name><surname>Wrzeszczyński</surname><given-names>KO</given-names></name><name><surname>Arora</surname><given-names>K</given-names></name><name><surname>Shah</surname><given-names>M</given-names></name><name><surname>Emde</surname><given-names>AK</given-names></name><name><surname>Felice</surname><given-names>V</given-names></name><name><surname>Frank</surname><given-names>MO</given-names></name><name><surname>Darnell</surname><given-names>RB</given-names></name><name><surname>Ghandi</surname><given-names>M</given-names></name><name><surname>Huang</surname><given-names>F</given-names></name><name><surname>Dewhurst</surname><given-names>S</given-names></name><name><surname>Maciejowski</surname><given-names>J</given-names></name><name><surname>de Lange</surname><given-names>T</given-names></name><name><surname>Setton</surname><given-names>J</given-names></name><name><surname>Riaz</surname><given-names>N</given-names></name><name><surname>Reis-Filho</surname><given-names>JS</given-names></name><name><surname>Powell</surname><given-names>S</given-names></name><name><surname>Knowles</surname><given-names>DA</given-names></name><name><surname>Reznik</surname><given-names>E</given-names></name><name><surname>Mishra</surname><given-names>B</given-names></name><name><surname>Beroukhim</surname><given-names>R</given-names></name><name><surname>Zody</surname><given-names>MC</given-names></name><name><surname>Robine</surname><given-names>N</given-names></name><name><surname>Oman</surname><given-names>KM</given-names></name><name><surname>Sanchez</surname><given-names>CA</given-names></name><name><surname>Kuhner</surname><given-names>MK</given-names></name><name><surname>Smith</surname><given-names>LP</given-names></name><name><surname>Galipeau</surname><given-names>PC</given-names></name><name><surname>Paulson</surname><given-names>TG</given-names></name><name><surname>Reid</surname><given-names>BJ</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Wilkes</surname><given-names>D</given-names></name><name><surname>Sboner</surname><given-names>A</given-names></name><name><surname>Mosquera</surname><given-names>JM</given-names></name><name><surname>Elemento</surname><given-names>O</given-names></name><name><surname>Imielinski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Distinct classes of complex structural variation uncovered across thousands of cancer genome graphs</article-title><source>Cell</source><volume>183</volume><fpage>197</fpage><lpage>210</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2020.08.006</pub-id><pub-id pub-id-type="pmid">33007263</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>He</surname><given-names>L</given-names></name><name><surname>Guo</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Tian</surname><given-names>S</given-names></name><name><surname>Lin</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>FUT2 inhibits the EMT and metastasis of colorectal cancer by increasing LRP1 fucosylation</article-title><source>Cell Communication and Signaling</source><volume>21</volume><elocation-id>63</elocation-id><pub-id pub-id-type="doi">10.1186/s12964-023-01060-0</pub-id><pub-id pub-id-type="pmid">36973740</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Homola</surname><given-names>D</given-names></name><name><surname>Bernico</surname><given-names>M</given-names></name><name><surname>Tallent</surname><given-names>E</given-names></name><collab>Ingvar-Y</collab><name><surname>Peng</surname><given-names>M</given-names></name><name><surname>Christ</surname><given-names>M</given-names></name><name><surname>Massaron</surname><given-names>L</given-names></name><name><surname>Miner</surname><given-names>L</given-names></name><name><surname>Vandeputte</surname><given-names>E</given-names></name><collab>diegogm</collab><name><surname>Pfannschmidt</surname><given-names>L</given-names></name><name><surname>Mottl</surname><given-names>D</given-names></name><name><surname>Biesinger</surname><given-names>D</given-names></name><name><surname>Glover</surname><given-names>A</given-names></name><name><surname>Bittremieux</surname><given-names>W</given-names></name><collab>arsenkhy</collab><name><surname>Baum</surname><given-names>A</given-names></name><name><surname>Stein</surname><given-names>D</given-names></name><name><surname>Wu</surname><given-names>L</given-names></name><collab>Mao771</collab><collab>zoj613</collab></person-group><year iso-8601-date="2024">2024</year><data-title>Boruta_Py</data-title><version designator="92e4b4e">92e4b4e</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/scikit-learn-contrib/boruta_py">https://github.com/scikit-learn-contrib/boruta_py</ext-link></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>DW</given-names></name><name><surname>Sherman</surname><given-names>BT</given-names></name><name><surname>Tan</surname><given-names>Q</given-names></name><name><surname>Collins</surname><given-names>JR</given-names></name><name><surname>Alvord</surname><given-names>WG</given-names></name><name><surname>Roayaei</surname><given-names>J</given-names></name><name><surname>Stephens</surname><given-names>R</given-names></name><name><surname>Baseler</surname><given-names>MW</given-names></name><name><surname>Lane</surname><given-names>HC</given-names></name><name><surname>Lempicki</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The DAVID Gene Functional Classification Tool: a novel biological module-centric algorithm to functionally analyze large gene lists</article-title><source>Genome Biology</source><volume>8</volume><elocation-id>R183</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2007-8-9-r183</pub-id><pub-id pub-id-type="pmid">17784955</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hung</surname><given-names>KL</given-names></name><name><surname>Yost</surname><given-names>KE</given-names></name><name><surname>Xie</surname><given-names>L</given-names></name><name><surname>Shi</surname><given-names>Q</given-names></name><name><surname>Helmsauer</surname><given-names>K</given-names></name><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Schöpflin</surname><given-names>R</given-names></name><name><surname>Lange</surname><given-names>JT</given-names></name><name><surname>Chamorro González</surname><given-names>R</given-names></name><name><surname>Weiser</surname><given-names>NE</given-names></name><name><surname>Chen</surname><given-names>C</given-names></name><name><surname>Valieva</surname><given-names>ME</given-names></name><name><surname>Wong</surname><given-names>IT-L</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Dehkordi</surname><given-names>SR</given-names></name><name><surname>Duffy</surname><given-names>CV</given-names></name><name><surname>Kraft</surname><given-names>K</given-names></name><name><surname>Tang</surname><given-names>J</given-names></name><name><surname>Belk</surname><given-names>JA</given-names></name><name><surname>Rose</surname><given-names>JC</given-names></name><name><surname>Corces</surname><given-names>MR</given-names></name><name><surname>Granja</surname><given-names>JM</given-names></name><name><surname>Li</surname><given-names>R</given-names></name><name><surname>Rajkumar</surname><given-names>U</given-names></name><name><surname>Friedlein</surname><given-names>J</given-names></name><name><surname>Bagchi</surname><given-names>A</given-names></name><name><surname>Satpathy</surname><given-names>AT</given-names></name><name><surname>Tjian</surname><given-names>R</given-names></name><name><surname>Mundlos</surname><given-names>S</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Henssen</surname><given-names>AG</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>ecDNA hubs drive cooperative intermolecular oncogene expression</article-title><source>Nature</source><volume>600</volume><fpage>731</fpage><lpage>736</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-04116-8</pub-id><pub-id pub-id-type="pmid">34819668</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Innes</surname><given-names>AJ</given-names></name><name><surname>Sun</surname><given-names>B</given-names></name><name><surname>Wagner</surname><given-names>V</given-names></name><name><surname>Brookes</surname><given-names>S</given-names></name><name><surname>McHugh</surname><given-names>D</given-names></name><name><surname>Pombo</surname><given-names>J</given-names></name><name><surname>Porreca</surname><given-names>RM</given-names></name><name><surname>Dharmalingam</surname><given-names>G</given-names></name><name><surname>Vernia</surname><given-names>S</given-names></name><name><surname>Zuber</surname><given-names>J</given-names></name><name><surname>Vannier</surname><given-names>J-B</given-names></name><name><surname>García-Escudero</surname><given-names>R</given-names></name><name><surname>Gil</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>XPO7 is a tumor suppressor regulating p21<sup>CIP1</sup>-dependent senescence</article-title><source>Genes &amp; Development</source><volume>35</volume><fpage>379</fpage><lpage>391</lpage><pub-id pub-id-type="doi">10.1101/gad.343269.120</pub-id><pub-id pub-id-type="pmid">33602872</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>H</given-names></name><name><surname>Nguyen</surname><given-names>N-P</given-names></name><name><surname>Turner</surname><given-names>K</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Gujar</surname><given-names>AD</given-names></name><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Deshpande</surname><given-names>V</given-names></name><name><surname>Rajkumar</surname><given-names>U</given-names></name><name><surname>Namburi</surname><given-names>S</given-names></name><name><surname>Amin</surname><given-names>SB</given-names></name><name><surname>Yi</surname><given-names>E</given-names></name><name><surname>Menghi</surname><given-names>F</given-names></name><name><surname>Schulte</surname><given-names>JH</given-names></name><name><surname>Henssen</surname><given-names>AG</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name><name><surname>Beck</surname><given-names>CR</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Verhaak</surname><given-names>RGW</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Extrachromosomal DNA is associated with oncogene amplification and poor outcome across multiple cancers</article-title><source>Nature Genetics</source><volume>52</volume><fpage>891</fpage><lpage>897</lpage><pub-id pub-id-type="doi">10.1038/s41588-020-0678-2</pub-id><pub-id pub-id-type="pmid">32807987</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kobayashi</surname><given-names>Y</given-names></name><name><surname>Masuda</surname><given-names>T</given-names></name><name><surname>Fujii</surname><given-names>A</given-names></name><name><surname>Shimizu</surname><given-names>D</given-names></name><name><surname>Sato</surname><given-names>K</given-names></name><name><surname>Kitagawa</surname><given-names>A</given-names></name><name><surname>Tobo</surname><given-names>T</given-names></name><name><surname>Ozato</surname><given-names>Y</given-names></name><name><surname>Saito</surname><given-names>H</given-names></name><name><surname>Kuramitsu</surname><given-names>S</given-names></name><name><surname>Noda</surname><given-names>M</given-names></name><name><surname>Otsu</surname><given-names>H</given-names></name><name><surname>Mizushima</surname><given-names>T</given-names></name><name><surname>Doki</surname><given-names>Y</given-names></name><name><surname>Eguchi</surname><given-names>H</given-names></name><name><surname>Mori</surname><given-names>M</given-names></name><name><surname>Mimori</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Mitotic checkpoint regulator RAE1 promotes tumor growth in colorectal cancer</article-title><source>Cancer Science</source><volume>112</volume><fpage>3173</fpage><lpage>3189</lpage><pub-id pub-id-type="doi">10.1111/cas.14969</pub-id><pub-id pub-id-type="pmid">34008277</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kursa</surname><given-names>MB</given-names></name><name><surname>Rudnicki</surname><given-names>WR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Feature selection with the boruta package</article-title><source>Journal of Statistical Software</source><volume>36</volume><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.18637/jss.v036.i11</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lange</surname><given-names>JT</given-names></name><name><surname>Rose</surname><given-names>JC</given-names></name><name><surname>Chen</surname><given-names>CY</given-names></name><name><surname>Pichugin</surname><given-names>Y</given-names></name><name><surname>Xie</surname><given-names>L</given-names></name><name><surname>Tang</surname><given-names>J</given-names></name><name><surname>Hung</surname><given-names>KL</given-names></name><name><surname>Yost</surname><given-names>KE</given-names></name><name><surname>Shi</surname><given-names>Q</given-names></name><name><surname>Erb</surname><given-names>ML</given-names></name><name><surname>Rajkumar</surname><given-names>U</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Taschner-Mandl</surname><given-names>S</given-names></name><name><surname>Bernkopf</surname><given-names>M</given-names></name><name><surname>Swanton</surname><given-names>C</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Huang</surname><given-names>W</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Henssen</surname><given-names>AG</given-names></name><name><surname>Werner</surname><given-names>B</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The evolutionary dynamics of extrachromosomal DNA in human cancers</article-title><source>Nature Genetics</source><volume>54</volume><fpage>1527</fpage><lpage>1533</lpage><pub-id pub-id-type="doi">10.1038/s41588-022-01177-x</pub-id><pub-id pub-id-type="pmid">36123406</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lawrence</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The nuclear factor NF-kappaB pathway in inflammation</article-title><source>Cold Spring Harbor Perspectives in Biology</source><volume>1</volume><elocation-id>a001651</elocation-id><pub-id pub-id-type="doi">10.1101/cshperspect.a001651</pub-id><pub-id pub-id-type="pmid">20457564</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>James</surname><given-names>SJ</given-names></name><name><surname>Wyllie</surname><given-names>DH</given-names></name><name><surname>Wynne</surname><given-names>C</given-names></name><name><surname>Czibula</surname><given-names>A</given-names></name><name><surname>Bukhari</surname><given-names>A</given-names></name><name><surname>Pye</surname><given-names>K</given-names></name><name><surname>Bte Mustafah</surname><given-names>SM</given-names></name><name><surname>Fajka-Boja</surname><given-names>R</given-names></name><name><surname>Szabo</surname><given-names>E</given-names></name><name><surname>Angyal</surname><given-names>A</given-names></name><name><surname>Hegedus</surname><given-names>Z</given-names></name><name><surname>Kovacs</surname><given-names>L</given-names></name><name><surname>Hill</surname><given-names>AVS</given-names></name><name><surname>Jefferies</surname><given-names>CA</given-names></name><name><surname>Wilson</surname><given-names>HL</given-names></name><name><surname>Yongliang</surname><given-names>Z</given-names></name><name><surname>Kiss-Toth</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>TMEM203 is a binding partner and regulator of STING-mediated inflammatory signaling in macrophages</article-title><source>PNAS</source><volume>116</volume><fpage>16479</fpage><lpage>16488</lpage><pub-id pub-id-type="doi">10.1073/pnas.1901090116</pub-id><pub-id pub-id-type="pmid">31346090</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liberzon</surname><given-names>A</given-names></name><name><surname>Subramanian</surname><given-names>A</given-names></name><name><surname>Pinchback</surname><given-names>R</given-names></name><name><surname>Thorvaldsdóttir</surname><given-names>H</given-names></name><name><surname>Tamayo</surname><given-names>P</given-names></name><name><surname>Mesirov</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Molecular signatures database (MSigDB) 3.0</article-title><source>Bioinformatics</source><volume>27</volume><fpage>1739</fpage><lpage>1740</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr260</pub-id><pub-id pub-id-type="pmid">21546393</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liberzon</surname><given-names>A</given-names></name><name><surname>Birger</surname><given-names>C</given-names></name><name><surname>Thorvaldsdóttir</surname><given-names>H</given-names></name><name><surname>Ghandi</surname><given-names>M</given-names></name><name><surname>Mesirov</surname><given-names>JP</given-names></name><name><surname>Tamayo</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The Molecular Signatures Database (MSigDB) hallmark gene set collection</article-title><source>Cell Systems</source><volume>1</volume><fpage>417</fpage><lpage>425</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2015.12.004</pub-id><pub-id pub-id-type="pmid">26771021</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>MS</given-names></name></person-group><year iso-8601-date="2024">2024</year><data-title>Ecdna_Gene_Expression</data-title><version designator="swh:1:rev:ba8994f067d8f6e50201979137bd0dcb8e9e86ac">swh:1:rev:ba8994f067d8f6e50201979137bd0dcb8e9e86ac</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:82ef3d6063fe5e5e582f0bd9be9c6d62dbde9ab6;origin=https://github.com/miinslin/ecDNA_Gene_Expression;visit=swh:1:snp:ea6fe839e466985120e3f4ea5ec4533722043aaa;anchor=swh:1:rev:ba8994f067d8f6e50201979137bd0dcb8e9e86ac">https://archive.softwareheritage.org/swh:1:dir:82ef3d6063fe5e5e582f0bd9be9c6d62dbde9ab6;origin=https://github.com/miinslin/ecDNA_Gene_Expression;visit=swh:1:snp:ea6fe839e466985120e3f4ea5ec4533722043aaa;anchor=swh:1:rev:ba8994f067d8f6e50201979137bd0dcb8e9e86ac</ext-link></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Love</surname><given-names>MI</given-names></name><name><surname>Huber</surname><given-names>W</given-names></name><name><surname>Anders</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title><source>Genome Biology</source><volume>15</volume><elocation-id>550</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id><pub-id pub-id-type="pmid">25516281</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Ng</surname><given-names>AWT</given-names></name><name><surname>Galipeau</surname><given-names>PC</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Sanchez</surname><given-names>CA</given-names></name><name><surname>Katz-Summercorn</surname><given-names>AC</given-names></name><name><surname>Kim</surname><given-names>H</given-names></name><name><surname>Jammula</surname><given-names>S</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Lippman</surname><given-names>SM</given-names></name><name><surname>Verhaak</surname><given-names>RGW</given-names></name><name><surname>Maley</surname><given-names>CC</given-names></name><name><surname>Alexandrov</surname><given-names>LB</given-names></name><name><surname>Reid</surname><given-names>BJ</given-names></name><name><surname>Fitzgerald</surname><given-names>RC</given-names></name><name><surname>Paulson</surname><given-names>TG</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Extrachromosomal DNA in the cancerous transformation of Barrett’s oesophagus</article-title><source>Nature</source><volume>616</volume><fpage>798</fpage><lpage>805</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-05937-5</pub-id><pub-id pub-id-type="pmid">37046089</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Dameracharla</surname><given-names>B</given-names></name><name><surname>Khan</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2024">2024</year><data-title>Ampliconclassifier</data-title><version designator="b3fe4cc">b3fe4cc</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/AmpliconSuite/AmpliconClassifier">https://github.com/AmpliconSuite/AmpliconClassifier</ext-link></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McCarthy</surname><given-names>DJ</given-names></name><name><surname>Smyth</surname><given-names>GK</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Testing significance relative to a fold-change threshold is a TREAT</article-title><source>Bioinformatics</source><volume>25</volume><fpage>765</fpage><lpage>771</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp053</pub-id><pub-id pub-id-type="pmid">19176553</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McGrail</surname><given-names>DJ</given-names></name><name><surname>Pilié</surname><given-names>PG</given-names></name><name><surname>Rashid</surname><given-names>NU</given-names></name><name><surname>Voorwerk</surname><given-names>L</given-names></name><name><surname>Slagter</surname><given-names>M</given-names></name><name><surname>Kok</surname><given-names>M</given-names></name><name><surname>Jonasch</surname><given-names>E</given-names></name><name><surname>Khasraw</surname><given-names>M</given-names></name><name><surname>Heimberger</surname><given-names>AB</given-names></name><name><surname>Lim</surname><given-names>B</given-names></name><name><surname>Ueno</surname><given-names>NT</given-names></name><name><surname>Litton</surname><given-names>JK</given-names></name><name><surname>Ferrarotto</surname><given-names>R</given-names></name><name><surname>Chang</surname><given-names>JT</given-names></name><name><surname>Moulder</surname><given-names>SL</given-names></name><name><surname>Lin</surname><given-names>S-Y</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>High tumor mutation burden fails to predict immune checkpoint blockade response across all cancer types</article-title><source>Annals of Oncology</source><volume>32</volume><fpage>661</fpage><lpage>672</lpage><pub-id pub-id-type="doi">10.1016/j.annonc.2021.02.006</pub-id><pub-id pub-id-type="pmid">33736924</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morton</surname><given-names>AR</given-names></name><name><surname>Dogan-Artun</surname><given-names>N</given-names></name><name><surname>Faber</surname><given-names>ZJ</given-names></name><name><surname>MacLeod</surname><given-names>G</given-names></name><name><surname>Bartels</surname><given-names>CF</given-names></name><name><surname>Piazza</surname><given-names>MS</given-names></name><name><surname>Allan</surname><given-names>KC</given-names></name><name><surname>Mack</surname><given-names>SC</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Gimple</surname><given-names>RC</given-names></name><name><surname>Wu</surname><given-names>Q</given-names></name><name><surname>Rubin</surname><given-names>BP</given-names></name><name><surname>Shetty</surname><given-names>S</given-names></name><name><surname>Angers</surname><given-names>S</given-names></name><name><surname>Dirks</surname><given-names>PB</given-names></name><name><surname>Sallari</surname><given-names>RC</given-names></name><name><surname>Lupien</surname><given-names>M</given-names></name><name><surname>Rich</surname><given-names>JN</given-names></name><name><surname>Scacheri</surname><given-names>PC</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Functional enhancers shape extrachromosomal oncogene amplifications</article-title><source>Cell</source><volume>179</volume><fpage>1330</fpage><lpage>1341</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2019.10.039</pub-id><pub-id pub-id-type="pmid">31761532</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nathanson</surname><given-names>DA</given-names></name><name><surname>Gini</surname><given-names>B</given-names></name><name><surname>Mottahedeh</surname><given-names>J</given-names></name><name><surname>Visnyei</surname><given-names>K</given-names></name><name><surname>Koga</surname><given-names>T</given-names></name><name><surname>Gomez</surname><given-names>G</given-names></name><name><surname>Eskin</surname><given-names>A</given-names></name><name><surname>Hwang</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Masui</surname><given-names>K</given-names></name><name><surname>Paucar</surname><given-names>A</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Ohashi</surname><given-names>M</given-names></name><name><surname>Zhu</surname><given-names>S</given-names></name><name><surname>Wykosky</surname><given-names>J</given-names></name><name><surname>Reed</surname><given-names>R</given-names></name><name><surname>Nelson</surname><given-names>SF</given-names></name><name><surname>Cloughesy</surname><given-names>TF</given-names></name><name><surname>James</surname><given-names>CD</given-names></name><name><surname>Rao</surname><given-names>PN</given-names></name><name><surname>Kornblum</surname><given-names>HI</given-names></name><name><surname>Heath</surname><given-names>JR</given-names></name><name><surname>Cavenee</surname><given-names>WK</given-names></name><name><surname>Furnari</surname><given-names>FB</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Targeted therapy resistance mediated by dynamic regulation of extrachromosomal mutant EGFR DNA</article-title><source>Science</source><volume>343</volume><fpage>72</fpage><lpage>76</lpage><pub-id pub-id-type="doi">10.1126/science.1241328</pub-id><pub-id pub-id-type="pmid">24310612</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ng</surname><given-names>PC</given-names></name><name><surname>Henikoff</surname><given-names>SS</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>SIFT: Predicting amino acid changes that affect protein function</article-title><source>Nucleic Acids Research</source><volume>31</volume><fpage>3812</fpage><lpage>3814</lpage><pub-id pub-id-type="doi">10.1093/nar/gkg509</pub-id><pub-id pub-id-type="pmid">12824425</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>B</given-names></name><name><surname>Minter-Dykhouse</surname><given-names>K</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>T</given-names></name><name><surname>Zhang</surname><given-names>H</given-names></name><name><surname>Lee</surname><given-names>S</given-names></name><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>L</given-names></name><name><surname>Lou</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>DBC1 functions as a tumor suppressor by regulating p53 stability</article-title><source>Cell Reports</source><volume>10</volume><fpage>1324</fpage><lpage>1334</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2015.01.066</pub-id><pub-id pub-id-type="pmid">25732823</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ribeiro</surname><given-names>JR</given-names></name><name><surname>Lovasco</surname><given-names>LA</given-names></name><name><surname>Vanderhyden</surname><given-names>BC</given-names></name><name><surname>Freiman</surname><given-names>RN</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Targeting TBP-Associated Factors in Ovarian Cancer</article-title><source>Frontiers in Oncology</source><volume>4</volume><elocation-id>45</elocation-id><pub-id pub-id-type="doi">10.3389/fonc.2014.00045</pub-id><pub-id pub-id-type="pmid">24653979</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rizvi</surname><given-names>NA</given-names></name><name><surname>Hellmann</surname><given-names>MD</given-names></name><name><surname>Snyder</surname><given-names>A</given-names></name><name><surname>Kvistborg</surname><given-names>P</given-names></name><name><surname>Makarov</surname><given-names>V</given-names></name><name><surname>Havel</surname><given-names>JJ</given-names></name><name><surname>Lee</surname><given-names>W</given-names></name><name><surname>Yuan</surname><given-names>J</given-names></name><name><surname>Wong</surname><given-names>P</given-names></name><name><surname>Ho</surname><given-names>TS</given-names></name><name><surname>Miller</surname><given-names>ML</given-names></name><name><surname>Rekhtman</surname><given-names>N</given-names></name><name><surname>Moreira</surname><given-names>AL</given-names></name><name><surname>Ibrahim</surname><given-names>F</given-names></name><name><surname>Bruggeman</surname><given-names>C</given-names></name><name><surname>Gasmi</surname><given-names>B</given-names></name><name><surname>Zappasodi</surname><given-names>R</given-names></name><name><surname>Maeda</surname><given-names>Y</given-names></name><name><surname>Sander</surname><given-names>C</given-names></name><name><surname>Garon</surname><given-names>EB</given-names></name><name><surname>Merghoub</surname><given-names>T</given-names></name><name><surname>Wolchok</surname><given-names>JD</given-names></name><name><surname>Schumacher</surname><given-names>TN</given-names></name><name><surname>Chan</surname><given-names>TA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Cancer immunology. Mutational landscape determines sensitivity to PD-1 blockade in non-small cell lung cancer</article-title><source>Science</source><volume>348</volume><fpage>124</fpage><lpage>128</lpage><pub-id pub-id-type="doi">10.1126/science.aaa1348</pub-id><pub-id pub-id-type="pmid">25765070</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Romano</surname><given-names>J</given-names></name><name><surname>Kromrey</surname><given-names>JD</given-names></name><name><surname>Coraggio</surname><given-names>J</given-names></name><name><surname>Skowronek</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Appropriate statistics for ordinal level data: Should we really be using t-test and Cohen’sd for evaluating group differences on the NSSE and other surveys</article-title><source>Annu. Meet. Fla. Assoc. Institutional Res</source><volume>177</volume><elocation-id>34</elocation-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Samson</surname><given-names>N</given-names></name><name><surname>Ablasser</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The cGAS-STING pathway and cancer</article-title><source>Nature Cancer</source><volume>3</volume><fpage>1452</fpage><lpage>1463</lpage><pub-id pub-id-type="doi">10.1038/s43018-022-00468-w</pub-id><pub-id pub-id-type="pmid">36510011</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sanborn</surname><given-names>JZ</given-names></name><name><surname>Salama</surname><given-names>SR</given-names></name><name><surname>Grifford</surname><given-names>M</given-names></name><name><surname>Brennan</surname><given-names>CW</given-names></name><name><surname>Mikkelsen</surname><given-names>T</given-names></name><name><surname>Jhanwar</surname><given-names>S</given-names></name><name><surname>Katzman</surname><given-names>S</given-names></name><name><surname>Chin</surname><given-names>L</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Double minute chromosomes in glioblastoma multiforme are revealed by precise reconstruction of oncogenic amplicons</article-title><source>Cancer Research</source><volume>73</volume><fpage>6036</fpage><lpage>6045</lpage><pub-id pub-id-type="doi">10.1158/0008-5472.CAN-13-0186</pub-id><pub-id pub-id-type="pmid">23940299</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sondka</surname><given-names>Z</given-names></name><name><surname>Bamford</surname><given-names>S</given-names></name><name><surname>Cole</surname><given-names>CG</given-names></name><name><surname>Ward</surname><given-names>SA</given-names></name><name><surname>Dunham</surname><given-names>I</given-names></name><name><surname>Forbes</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The COSMIC Cancer Gene Census: describing genetic dysfunction across all human cancers</article-title><source>Nature Reviews. Cancer</source><volume>18</volume><fpage>696</fpage><lpage>705</lpage><pub-id pub-id-type="doi">10.1038/s41568-018-0060-1</pub-id><pub-id pub-id-type="pmid">30293088</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Subramanian</surname><given-names>A</given-names></name><name><surname>Tamayo</surname><given-names>P</given-names></name><name><surname>Mootha</surname><given-names>VK</given-names></name><name><surname>Mukherjee</surname><given-names>S</given-names></name><name><surname>Ebert</surname><given-names>BL</given-names></name><name><surname>Gillette</surname><given-names>MA</given-names></name><name><surname>Paulovich</surname><given-names>A</given-names></name><name><surname>Pomeroy</surname><given-names>SL</given-names></name><name><surname>Golub</surname><given-names>TR</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Mesirov</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Gene set enrichment analysis: A knowledge-based approach for interpreting genome-wide expression profiles</article-title><source>PNAS</source><volume>102</volume><fpage>15545</fpage><lpage>15550</lpage><pub-id pub-id-type="doi">10.1073/pnas.0506580102</pub-id><pub-id pub-id-type="pmid">16199517</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname><given-names>L</given-names></name><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Du</surname><given-names>F</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Chen</surname><given-names>ZJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Cyclic GMP-AMP synthase is a cytosolic DNA sensor that activates the type I interferon pathway</article-title><source>Science</source><volume>339</volume><fpage>786</fpage><lpage>791</lpage><pub-id pub-id-type="doi">10.1126/science.1232458</pub-id><pub-id pub-id-type="pmid">23258413</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Suzuki</surname><given-names>R</given-names></name><name><surname>Shimodaira</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Pvclust: an R package for assessing the uncertainty in hierarchical clustering</article-title><source>Bioinformatics</source><volume>22</volume><fpage>1540</fpage><lpage>1542</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btl117</pub-id><pub-id pub-id-type="pmid">16595560</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Suzuki</surname><given-names>R</given-names></name><name><surname>Terada</surname><given-names>Y</given-names></name><name><surname>Shimodaira</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Pvclust</data-title><version designator="5d7626b">5d7626b</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/shimo-lab/pvclust">https://github.com/shimo-lab/pvclust</ext-link></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thorsson</surname><given-names>V</given-names></name><name><surname>Gibbs</surname><given-names>DL</given-names></name><name><surname>Brown</surname><given-names>SD</given-names></name><name><surname>Wolf</surname><given-names>D</given-names></name><name><surname>Bortone</surname><given-names>DS</given-names></name><name><surname>Ou Yang</surname><given-names>TH</given-names></name><name><surname>Porta-Pardo</surname><given-names>E</given-names></name><name><surname>Gao</surname><given-names>GF</given-names></name><name><surname>Plaisier</surname><given-names>CL</given-names></name><name><surname>Eddy</surname><given-names>JA</given-names></name><name><surname>Ziv</surname><given-names>E</given-names></name><name><surname>Culhane</surname><given-names>AC</given-names></name><name><surname>Paull</surname><given-names>EO</given-names></name><name><surname>Sivakumar</surname><given-names>IKA</given-names></name><name><surname>Gentles</surname><given-names>AJ</given-names></name><name><surname>Malhotra</surname><given-names>R</given-names></name><name><surname>Farshidfar</surname><given-names>F</given-names></name><name><surname>Colaprico</surname><given-names>A</given-names></name><name><surname>Parker</surname><given-names>JS</given-names></name><name><surname>Mose</surname><given-names>LE</given-names></name><name><surname>Vo</surname><given-names>NS</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Rader</surname><given-names>J</given-names></name><name><surname>Dhankani</surname><given-names>V</given-names></name><name><surname>Reynolds</surname><given-names>SM</given-names></name><name><surname>Bowlby</surname><given-names>R</given-names></name><name><surname>Califano</surname><given-names>A</given-names></name><name><surname>Cherniack</surname><given-names>AD</given-names></name><name><surname>Anastassiou</surname><given-names>D</given-names></name><name><surname>Bedognetti</surname><given-names>D</given-names></name><name><surname>Mokrab</surname><given-names>Y</given-names></name><name><surname>Newman</surname><given-names>AM</given-names></name><name><surname>Rao</surname><given-names>A</given-names></name><name><surname>Chen</surname><given-names>K</given-names></name><name><surname>Krasnitz</surname><given-names>A</given-names></name><name><surname>Hu</surname><given-names>H</given-names></name><name><surname>Malta</surname><given-names>TM</given-names></name><name><surname>Noushmehr</surname><given-names>H</given-names></name><name><surname>Pedamallu</surname><given-names>CS</given-names></name><name><surname>Bullman</surname><given-names>S</given-names></name><name><surname>Ojesina</surname><given-names>AI</given-names></name><name><surname>Lamb</surname><given-names>A</given-names></name><name><surname>Zhou</surname><given-names>W</given-names></name><name><surname>Shen</surname><given-names>H</given-names></name><name><surname>Choueiri</surname><given-names>TK</given-names></name><name><surname>Weinstein</surname><given-names>JN</given-names></name><name><surname>Guinney</surname><given-names>J</given-names></name><name><surname>Saltz</surname><given-names>J</given-names></name><name><surname>Holt</surname><given-names>RA</given-names></name><name><surname>Rabkin</surname><given-names>CS</given-names></name><collab>Cancer Genome Atlas Research Network</collab><name><surname>Lazar</surname><given-names>AJ</given-names></name><name><surname>Serody</surname><given-names>JS</given-names></name><name><surname>Demicco</surname><given-names>EG</given-names></name><name><surname>Disis</surname><given-names>ML</given-names></name><name><surname>Vincent</surname><given-names>BG</given-names></name><name><surname>Shmulevich</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The immune landscape of cancer</article-title><source>Immunity</source><volume>48</volume><fpage>812</fpage><lpage>830</lpage><pub-id pub-id-type="doi">10.1016/j.immuni.2018.03.023</pub-id><pub-id pub-id-type="pmid">29628290</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Turner</surname><given-names>KM</given-names></name><name><surname>Deshpande</surname><given-names>V</given-names></name><name><surname>Beyter</surname><given-names>D</given-names></name><name><surname>Koga</surname><given-names>T</given-names></name><name><surname>Rusert</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>B</given-names></name><name><surname>Arden</surname><given-names>K</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Nathanson</surname><given-names>DA</given-names></name><name><surname>Kornblum</surname><given-names>HI</given-names></name><name><surname>Taylor</surname><given-names>MD</given-names></name><name><surname>Kaushal</surname><given-names>S</given-names></name><name><surname>Cavenee</surname><given-names>WK</given-names></name><name><surname>Wechsler-Reya</surname><given-names>R</given-names></name><name><surname>Furnari</surname><given-names>FB</given-names></name><name><surname>Vandenberg</surname><given-names>SR</given-names></name><name><surname>Rao</surname><given-names>PN</given-names></name><name><surname>Wahl</surname><given-names>GM</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Extrachromosomal oncogene amplification drives tumour evolution and genetic heterogeneity</article-title><source>Nature</source><volume>543</volume><fpage>122</fpage><lpage>125</lpage><pub-id pub-id-type="doi">10.1038/nature21356</pub-id><pub-id pub-id-type="pmid">28178237</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Urban-Wojciuk</surname><given-names>Z</given-names></name><name><surname>Khan</surname><given-names>MM</given-names></name><name><surname>Oyler</surname><given-names>BL</given-names></name><name><surname>Fåhraeus</surname><given-names>R</given-names></name><name><surname>Marek-Trzonkowska</surname><given-names>N</given-names></name><name><surname>Nita-Lazar</surname><given-names>A</given-names></name><name><surname>Hupp</surname><given-names>TR</given-names></name><name><surname>Goodlett</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The Role of TLRs in Anti-cancer Immunity and Tumor Rejection</article-title><source>Frontiers in Immunology</source><volume>10</volume><elocation-id>2388</elocation-id><pub-id pub-id-type="doi">10.3389/fimmu.2019.02388</pub-id><pub-id pub-id-type="pmid">31695691</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Leen</surname><given-names>E</given-names></name><name><surname>Brückner</surname><given-names>L</given-names></name><name><surname>Henssen</surname><given-names>AG</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The genomic and spatial mobility of extrachromosomal DNA and its implications for cancer therapy</article-title><source>Nature Genetics</source><volume>54</volume><fpage>107</fpage><lpage>114</lpage><pub-id pub-id-type="doi">10.1038/s41588-021-01000-z</pub-id><pub-id pub-id-type="pmid">35145302</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Von Hoff</surname><given-names>DD</given-names></name><name><surname>McGill</surname><given-names>JR</given-names></name><name><surname>Forseth</surname><given-names>BJ</given-names></name><name><surname>Davidson</surname><given-names>KK</given-names></name><name><surname>Bradley</surname><given-names>TP</given-names></name><name><surname>Van Devanter</surname><given-names>DR</given-names></name><name><surname>Wahl</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Elimination of extrachromosomally amplified MYC genes from human tumor cells reduces their tumorigenicity</article-title><source>PNAS</source><volume>89</volume><fpage>8165</fpage><lpage>8169</lpage><pub-id pub-id-type="doi">10.1073/pnas.89.17.8165</pub-id><pub-id pub-id-type="pmid">1518843</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>Q</given-names></name><name><surname>Xu</surname><given-names>T</given-names></name><name><surname>Tong</surname><given-names>Y</given-names></name><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Zhu</surname><given-names>W</given-names></name><name><surname>Lu</surname><given-names>Z</given-names></name><name><surname>Ying</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Prognostic potential of alternative splicing markers in endometrial cancer</article-title><source>Molecular Therapy. Nucleic Acids</source><volume>18</volume><fpage>1039</fpage><lpage>1048</lpage><pub-id pub-id-type="doi">10.1016/j.omtn.2019.10.027</pub-id><pub-id pub-id-type="pmid">31785579</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Turner</surname><given-names>KM</given-names></name><name><surname>Nguyen</surname><given-names>N</given-names></name><name><surname>Raviram</surname><given-names>R</given-names></name><name><surname>Erb</surname><given-names>M</given-names></name><name><surname>Santini</surname><given-names>J</given-names></name><name><surname>Luebeck</surname><given-names>J</given-names></name><name><surname>Rajkumar</surname><given-names>U</given-names></name><name><surname>Diao</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Jameson</surname><given-names>N</given-names></name><name><surname>Corces</surname><given-names>MR</given-names></name><name><surname>Granja</surname><given-names>JM</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Coruh</surname><given-names>C</given-names></name><name><surname>Abnousi</surname><given-names>A</given-names></name><name><surname>Houston</surname><given-names>J</given-names></name><name><surname>Ye</surname><given-names>Z</given-names></name><name><surname>Hu</surname><given-names>R</given-names></name><name><surname>Yu</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>H</given-names></name><name><surname>Law</surname><given-names>JA</given-names></name><name><surname>Verhaak</surname><given-names>RGW</given-names></name><name><surname>Hu</surname><given-names>M</given-names></name><name><surname>Furnari</surname><given-names>FB</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Bafna</surname><given-names>V</given-names></name><name><surname>Mischel</surname><given-names>PS</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Circular ecDNA promotes accessible chromatin and high oncogene expression</article-title><source>Nature</source><volume>575</volume><fpage>699</fpage><lpage>703</lpage><pub-id pub-id-type="doi">10.1038/s41586-019-1763-5</pub-id><pub-id pub-id-type="pmid">31748743</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>T</given-names></name><name><surname>Wu</surname><given-names>C</given-names></name><name><surname>Zhao</surname><given-names>X</given-names></name><name><surname>Wang</surname><given-names>G</given-names></name><name><surname>Ning</surname><given-names>W</given-names></name><name><surname>Tao</surname><given-names>Z</given-names></name><name><surname>Chen</surname><given-names>F</given-names></name><name><surname>Liu</surname><given-names>XS</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Extrachromosomal DNA formation enables tumor immune escape potentially through regulating antigen presentation gene expression</article-title><source>Scientific Reports</source><volume>12</volume><elocation-id>3590</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-022-07530-8</pub-id><pub-id pub-id-type="pmid">35246593</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>N</given-names></name><name><surname>Wu</surname><given-names>Y-P</given-names></name><name><surname>Yin</surname><given-names>H-B</given-names></name><name><surname>Chen</surname><given-names>S-H</given-names></name><name><surname>Li</surname><given-names>X-D</given-names></name><name><surname>Xue</surname><given-names>X-Y</given-names></name><name><surname>Gou</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SHCBP1 promotes tumor cell proliferation, migration, and invasion, and is associated with poor prostate cancer prognosis</article-title><source>Journal of Cancer Research and Clinical Oncology</source><volume>146</volume><fpage>1953</fpage><lpage>1969</lpage><pub-id pub-id-type="doi">10.1007/s00432-020-03247-1</pub-id><pub-id pub-id-type="pmid">32447485</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Xue</surname><given-names>J</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Zhou</surname><given-names>T</given-names></name><name><surname>Zu</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>TNFAIP6 promotes invasion and metastasis of gastric cancer and indicates poor prognosis of patients</article-title><source>Tissue &amp; Cell</source><volume>68</volume><elocation-id>101455</elocation-id><pub-id pub-id-type="doi">10.1016/j.tice.2020.101455</pub-id><pub-id pub-id-type="pmid">33221562</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname><given-names>A</given-names></name><name><surname>Ibrahim</surname><given-names>JG</given-names></name><name><surname>Love</surname><given-names>MI</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Heavy-tailed prior distributions for sequence count data: removing the noise and preserving large differences</article-title><source>Bioinformatics</source><volume>35</volume><fpage>2084</fpage><lpage>2092</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty895</pub-id><pub-id pub-id-type="pmid">30395178</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88895.3.sa0</article-id><title-group><article-title>eLife assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Gingeras</surname><given-names>Thomas R</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Cold Spring Harbor Laboratory</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Convincing</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Important</kwd></kwd-group></front-stub><body><p>This study of extrachromosomal DNA (ecDNA) identifies genes that distinguish ecDNA+ and ecDNA- tumors. The findings in the manuscript are <bold>important</bold> and the genomic analyses <bold>convincing</bold>. However, some of the data remain observational and the inferences would therefore be more robust with experimental validation. This manuscript could well be of relevance to biologists interested in cancer biology and gene regulation.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88895.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Recently discovered extrachromosomal DNA (ecDNA) provides an alternative non-chromosomal means for oncogene amplification and a potent substrate for selective evolution of tumors. The current work aims to identify key genes whose expression distinguishes ecDNA+ and ecDNA- tumors and the associated processes to shed light on the biological mechanisms underlying ecDNA genesis and their oncogenic effects. This is clearly an important question and through detailed analysis this work points to specific GO processes associated (up and down) with ecDNA+ tumors, namely, specific DNA damage repair processes and specific oncogenic processes.</p><p>In the initial submission I had commented on lack of clarity of method, potential biases, and in some cases inappropriate interpretation. In the revised version, the authors have addressed all my comments satisfactorily and I think this is an important work furthering our understanding of mechanisms underlying ecDNA+ tumors.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88895.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>In their manuscript Lin et al. describe an important study on the transcriptional programs associated with the presence of extrachromosomal DNA in a cohort of 870 cancers of different origins. The authors find that compared to cancers lacking such amplifications, ecDNA+ cancers express higher levels of DNA damage repair-associated genes, but lower levels of immune-related gene programs.</p><p>This work is very timely and its findings have the potential to be very impactful, as the transcriptional context differences between ecDNA+ and ecDNA- cancers are currently largely unknown. The observation that immune programs are downregulated in ecDNA+ cancers may initiate new preclinical and translational studies that impact the way ecDNA+ cancers are treated in the future. Thus, this study has important theoretical implications that have the potential to substantially advance our understanding of ecDNA+ cancers.</p><p>Strengths:</p><p>The authors provide compelling evidence for their conclusions based on large patient datasets. The methods they used and analyses are rigorous.</p><p>Weaknesses:</p><p>The biological interpretation of the data remains observational. The direct implication of these genes in ecDNA(+) tumors is not tested experimentally.</p></body></sub-article><sub-article article-type="referee-report" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88895.3.sa3</article-id><title-group><article-title>Reviewer #3 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>Using a combination of approaches, including automated feature selection and hierarchical clustering, the author identified a set of genes persistently associated with extrachromosomal DNA (ecDNA) presence across cancer types. The authors further validated the gene set identified using gene ontology enrichment analysis and identified that upregulated genes in extrachromosomal DNA-containing tumors are enriched in biological processes like DNA damage and cell proliferation, whereas downregulated genes are enriched in immune response processes.</p><p>Comments for the previous version:</p><p>Major comments:</p><p>(1) The authors presented a solid comparative analysis of ecDNA-containing and ecDNA-free tumors. An established automated feature selection approach, Boruta, was used to select differentially expressed genes (DEG) in ecDNA(+) and ecDNA(-) TCGA tumor samples, and the iterative selection process and two-tier multiple hypothesis testing ensured the selection of reliable DEGs. The author showed that the DEG selected using Boruta has stronger predictive power than genes with top log-fold changes.</p><p>(2) The author performed a thorough interpretation of the findings with GO enrichment analysis of biological processes enriched in the identified DEG set and presented interesting findings, including the enrichment in DNA damage process among the genes upregulated in ecDNA(+) tumors.</p><p>(3) Overall, the authors achieved their aims with solid data mining and analysis approaches applied to public data tumor data sets.</p><p>(4) While it may not be the scope of this study, it will be interesting to at least have some justification for choosing Boruta over other feature selection methods, such as Recursive Feature Elimination (RFE) and backward stepwise selection.</p><p>(5) The authors showed that DESEQ-selected DEGs with top log-fold changes have less strong predictive power and speculated that this may be due to the fact that genes with top log-fold changes (LFC) are confined only to a small subset of samples. It will be interesting to select DEGs with top log-fold changes after first partitioning the tumor samples. For example, randomly partition the tumor samples, identify the DEGs with top LFC, combine the DEGs identified from each partition, then evaluate the predictive power of these DEGs against the Boruta-selected DEGs.</p><p>(6) While the authors showed that the presence of mutations was not able to classify ecDNA(+) and (-) tumor samples, it will be interesting to see if variant allele frequencies of the genes containing these mutations have predictive power.</p><p>Comments for the revised version:</p><p>The authors addressed the comments and recommendations with solid analysis and explanations in the revision. The added analysis using GLM is especially appreciated and provides convincing evidence for the predicting power of the Boruta-selected genes. The only comment is at this point is that it is recommended that the author provide some justification for choosing Boruta over other feature selection methods. It is not necessary to provide benchmarking results - justification based on the review of previous literature is sufficient, as it is not well explained in the paper why Boruta was chosen in the first place. Is it state-of-the-art? Has it demonstrated better performance in other settings? A few sentences answering these questions should suffice.</p></body></sub-article><sub-article article-type="author-comment" id="sa4"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88895.3.sa4</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Lin</surname><given-names>Miin S</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, San Diego</institution><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Jo</surname><given-names>Se-Young</given-names></name><role specific-use="author">Author</role><aff><institution>Yonsei University College of Medicine</institution><addr-line><named-content content-type="city">Seoul</named-content></addr-line><country>Republic of Korea</country></aff></contrib><contrib contrib-type="author"><name><surname>Luebeck</surname><given-names>Jens</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, San Diego</institution><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Chang</surname><given-names>Howard Y</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Wu</surname><given-names>Sihan</given-names></name><role specific-use="author">Author</role><aff><institution>The University of Texas Southwestern Medical Center</institution><addr-line><named-content content-type="city">Dallas</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Mischel</surname><given-names>Paul S</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Bafna</surname><given-names>Vineet</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, San Diego</institution><addr-line><named-content content-type="city">La Jolla</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p><bold>eLife assessment</bold></p><p>This study of extrachromosomal DNA (ecDNA) aims to identify genes that distinguish ecDNA+ and ecDNA- tumors. This timely study is important in addressing the genes responding to the amplification of the ecDNA. The data presented are for the most part solid, there were concerns regarding the clarity in the description of the analysis methods and whether the evidence for specific genes required to maintain the ecDNA+ state was entirely conclusive.</p><p><bold>Public Reviews:</bold></p><p><bold>Reviewer #1 (Public Review):</bold></p><p>Recently discovered extrachromosomal DNA (ecDNA) provides an alternative non-chromosomal means for oncogene amplification and a potent substrate for selective evolution of tumors. The current work aims to identify key genes whose expression distinguishes ecDNA+ and ecDNA- tumors and the associated processes to shed light on the biological mechanisms underlying ecDNA genesis and their oncogenic effects. While this is clearly an important question, the analysis and the evidence supporting the claims are weak. The specific machine learning approach seems unnecessarily convoluted, insufficiently justified and explained, and the language used by the authors conflates correlation with causality. This work points to specific GO processes associated (up and down) with ecDNA+ tumors, many of which are expected but some seem intriguing, such as association with DSB pathways. My specific comments are listed below.</p></disp-quote><p><bold>Response.</bold> As some of the specific questions below address similar concerns, we have answered them briefly here. As a high level point, the reviewer is correct in that other statistical or ML approaches could potentially have been used, and that some are simpler. However, the test used here directly addresses the question: <italic>Find a collection of genes whose expression value is predictive of ecDNA status in the sample.</italic> Because the underlying method in the Boruta analysis uses random forests, it can test predictive power without relying on a linearity assumption implicit in other methods. In this revision, we also compare against a Generalized Linear Model and show that it is less suited to the specific task above. We also address the reviewer concerns about specific parameter choices by showing robustness to the specific parameter.</p><disp-quote content-type="editor-comment"><p>(A) The claim of identifying genes required to 'maintain' ecDNA+ status is not justified - predictive features are not necessarily causal.</p></disp-quote><p><bold>Response.</bold> We agree with the reviewer that predictive features are correlative and not causal. In the manuscript, we identify genes whose expression (when used as a feature) is predictive of ecDNA presence or absence. Such predictive genes are consistently over-expressed or consistently under-expressed in ecDNA(+) samples relative to ecDNA(-) samples even though they are not required to be on ecDNA. To our knowledge, we did not claim that these genes are causal for ecDNA formation or maintenance, only that such genes and the underlying biological processes are worth investigating. In the beginning of the manuscript, we had written the following paragraph, but we have removed the last line:</p><p>“In lieu of identifying genes that are highly differentially expressed between ecDNA(+) and ecDNA(-) samples but driven by a small subset of cases (e.g. gene A in <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), we sought to identify genes (e.g. gene B) whose expression level was predictive of ecDNA presence. We assumed that genes that were persistently over-expressed or under-expressed in ecDNA(+) samples relative to ecDNA(-) samples were more likely to be involved in ecDNA biogenesis or maintenance, or in mediating the cellular response to the presence of ecDNA.”</p><p>We revised the manuscript to make sure that there are no claims that refer to causality. We revisited all phrases where the words like “maintain” were used and added appropriate disclaimers, or replaced them by the phrase, “ecDNA presence.” The remaining statements say, for example, “These results are consistent with a pan-cancer role of CorEx genes in ecDNA biogenesis and maintenance,” and do not claim causality.</p><disp-quote content-type="editor-comment"><p>(B) The methods and procedures to identify the key genes is hyper-parameterized and convoluted and casts doubt on the robustness of the findings given the size and heterogeneity of the data.</p><p>(a) In the first two paragraphs of Boruta Analysis Methods section, authors describe an iterative procedure where in each iteration, a binomial p-value is computed for each gene based on number of iterations thus far in which the gene was selected (higher GINI index than max of shadow features). But then in the third paragraph they simply perform Random Forest in 200 random 80% of samples and pick a gene if it is selected in at least 10/200. It is ultimately not clear what was done. Why 10/200? Also &quot;the probability that a gene is a &quot;hit&quot; or &quot;non-hit&quot; in each iteration is 0.5&quot; is unclear. That probability is of a gene achieving GINI index higher than the max of shadow features. How can it be 0.5?</p></disp-quote><p><bold>Response.</bold> We believe that there is some misunderstanding about the algorithm, and we agree that the description should have been more clear. We have greatly simplified the description in the manuscript. However, we want to provide some higher-level explanation here. Boruta is a standard feature extraction algorithm (Kursa, <italic>Journal of Statistical Software</italic> September 2010, Volume 36, Issue 11), and we used a Python implementation of the method. Given a gene expression data-set with class labels on samples, Boruta extracts features (genes) that best predict the class labels using a Random Forest Classifier, as long as the features are more predictive than permuted features added in each iteration. As we are using an implementation of a published method, we have removed non-essential details, referring directly to the publication. Nevertheless, to address the reviewer’s specific critique, the number of false-features added changes in each iteration (it equals the number of accepted+uncommitted features). Therefore, the choice of 0.5 by Boruta (it is fixed in the published method and not a user-specified parameter) is a conservative approach. If a gene was no better than a randomly chosen feature, its predictive performance would exceed the <italic>most predictive</italic> randomly chosen feature by at most 0.5 (but could be lower, making the choice of 0.5 conservative).</p><p>While Boruta iteratively picks genes that are significantly better than random features, the list of genes predicted might be specific to the data-set, and might change with different data-sets. Therefore, we employed a bootstrapping strategy: we performed 200 trials each time picking 80% of the ecDNA(+) samples and 80% of the ecDNA(-) samples at random, thus generating many data-sets while maintaining class imbalance. For each of the 200 trials, we performed a Boruta analysis. Finally, we picked a gene if it was selected as a Boruta feature in at least 10 of 200 trials.</p><p>The reviewer has a reasonable critique about why 10 (of 200) specifically, and why not fewer or more. Most genes are weak predictors by themselves. For example, RAE1, which is the top ranked gene, picked in all 200 Boruta trials, can only predict ecDNA status with poor recall for any meaningful precision.</p><fig id="sa4fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig1-v1.tif"/></fig><p>Given the weakness of an individual gene as a classifier, its repeated selection in multiple Boruta trials is already a significant event. By requiring a gene to be picked in 5% of the trials (10/200), we were selecting a small, but more robust list of genes. However, to further explore the reviewer’s concerns, we also applied 8 other selection criteria ranging from 5 (of 200 Boruta trials) to 200 of 200 Boruta trials. See Figure below. The number of CorEx genes expectedly decreases. However, of the 187 GO terms that were enriched by 262 UP-genes using 10 of 200 Boruta trials as the selection criteria, 93 terms (49.7%) were enriched for each cut-off (see Author response image 2), and 155 terms (82.9%) were enriched in at least 5 of the 8 cut-off criteria. Given that the remaining analysis works on the hierarchy of GO terms and finds 4 GO-categories (Mitotic Cell Cycle, G1/S, G2/M; cell-division; DSB DNA Damage response; and the HOX Gene cluster) enriched by UP-regulated genes, those conclusions would hold regardless of the specific cut-off.</p><fig id="sa4fig2" position="float"><label>Author response image 2.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig2-v1.tif"/></fig><p>The number of GO terms that were enriched by DOWN-regulated genes is smaller, only 73, and falls rapidly for higher cut-offs, with 25 at a cut-off of 15. Therefore we see fewer terms enriched for more stringent cut-offs. However, they all support immune processes. These results do suggest that there are fewer genes that are consistently down-regulated in ecDNA(+) cancers, and expression change in a small number of genes may be sufficient to promote conditions for ecDNA.</p><p>Finally, we note that in the final section we discuss the 65 most highly ranked genes with a harmonic mean rank &lt;= 3. These 65 CorEx genes (or a member of their cluster) appear in each of 200 Boruta trials. Thus, their choice is also not dependent on the cut-off of 10 in 200. In summary, the conclusions of the paper do not depend upon the specific cut-off of 10 in 200 trials.</p><p>We have added the figure as a supplemental figure and have added the following text to the manuscript:</p><p>“Any CorEx gene is either a Core gene that was selected as a feature in at least 5% of 200 Boruta trials, or be highly co-expressed with a Core gene. Because the selection criterion of 5% is arbitrary, we also tested robustness with 8 other cut-offs ranging from 5-of-200 to 200-of-200 Boruta trials. The number of CorEx genes expectedly decreases with more stringent cut-offs. However, of the 187 GO terms that were enriched by 262 CorEx UP-genes using 10 of 200 Boruta trials as the selection criteria, 93 terms (49.7%) were enriched for each cut-off (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>), and 155 terms (82.9%) were enriched in at least 5 of the 8 cut-offs. Given that our subsequent analyses utilized the hierarchy of GO terms and identified 4 GO-categories enriched by UP-regulated genes, the conclusions would hold regardless of the specific cut-off.”</p><disp-quote content-type="editor-comment"><p>(b) The approach of combining genes with clusters is arbitrary. Why not start with clusters and evaluate each cluster (using some gene set summary score) for their ability to discriminate? Ultimately, one needs additional information to disambiguate correlated genes (i.e. in a coexpression cluster) in terms of causality.</p></disp-quote><p><bold>Response.</bold> In general, the approach proposed by the reviewer is reasonable. However, we did consider that possibility and found that our approach was easier to implement. For example, if we clustered first, we would have the challenge of choosing the correct set of clusters. Also, the Boruta analysis would become very difficult while dealing with clusters (e.g., how to define falsefeatures?). We tested other methods of picking genes that were suggested by other reviewers such as generalized linear models. They turned out not to be as predictive of ecDNA status, as described later in the response. Finally, we performed many experiments to ensure the validity of the clustering. Specifically, we had the following text in the paper:</p><p>“Notably, among the 354 clusters, only 2 clusters (with 14 total genes) did not contain any Core genes. As most genes do not have completely identical expression patterns, we would expect one gene to be consistently picked as a Boruta gene over another co-expressed gene. Consistent with this hypothesis, most (344/354) clusters contained only 1 or 2 Core genes (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). When selecting clusters that contained at least 1 Core and 1 co-expressed gene, 53 of 71 clusters contained 1 to 3 Core genes (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>), confirming that a few genes per co-expressed cluster provide sufficient predictive value, but other co-expressed genes might still play an important functional role in maintaining ecDNA(+) status.”</p><p>These experiments suggest that the genes found by extending the Core genes through clustering do not radically change the Core genes, but only enhance the set.</p><disp-quote content-type="editor-comment"><p>(c) The cross-validation procedure is not clear at all. There is a mention of 80-20 split but exactly how/if the evaluation is done on the 20% is muddled. The way precision-recall procedure is also a bit convoluted - why not simply use the area under the PR curve?</p></disp-quote><p><bold>Response.</bold> We apologize if the method was unclear. We have rewritten the methods part to make things clearer. As a high level point, there are <italic>two places</italic> where we use the same 80-20 split, and that resulted in some confusion. We start by randomly picking 80% of the ecDNA(+) and 80% of ecDNA(-) samples to create an 80-20 split of all samples. This procedure is repeated to generate 200 80-20 split data-sets. These data-sets are hereafter called 200 training and test samples.</p><p>In the first usage, we use only the ‘training’ part of the 200 samples. We apply Boruta to each training set, and this helps us select the Core genes, which are then expanded to form the CorEx set. At this point, the CorEx genes are frozen for analysis in the rest of the paper. One question that we subsequently answer is <italic>what is the predictive power of the CorEx genes in determining if the sample is ecDNA(+) or ecDNA(-)?</italic> We also compare the predictive performance of CorEx genes relative to (a) Core genes, (b) LFC genes, and (c) random genes. In the revised manuscript, we have added another list of 3,012 genes selected using a single gene generalized linear model (GLM) for feature prediction. To make these comparisons, we utilized the same 200 training and test data-sets as before. In each test, we trained a random forest classifier on the training set and predicted on the ‘test’ set, for each of the 5 gene lists. This provided a uniform and fair method for testing which of the 5 gene lists was the better predictor of ecDNA status.</p><p>The precision recall values are plotted in <xref ref-type="fig" rid="fig2">Figure 2B</xref> (also included below). We note that <italic>none of the</italic> gene lists was a great predictor of ecDNA status of a sample. However, the CorEx and Core genes were significantly more predictive than GLM, LFC, and random genes. The predictive power of GLM genes was very similar to LFC, and better than random.</p><p>For each of these 200 tests, we obtained a separate area under the precision-recall curve number for each of the gene-sets. To address the reviewer’s comments regarding a single number, we reported the average of the AUPRC for each of the gene-sets in the revision. The mean AUPRC values were added to the manuscript and are described here as well: Core_408_genes (0.495), CorEx_643_genes (0.48), Random_643_genes (0.36), top_lfc_643_genes (0.429), and GLM_R_3012_genes (0.426).</p><p>We also changed <xref ref-type="fig" rid="fig2">Figure 2B</xref> to show box-plots showing distribution of recall values for specific precision windows instead of maximum recall. For ease of checking, the figure is reproduced below.</p><fig id="sa4fig3" position="float"><label>Author response image 3.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig3-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(d) The claim is that Boruta genes are different from differentially expressed genes but the differential expression seems to be estimated without regards to cancer type, which would certainly be highly biased and misleading. Why not do a simple regression of gene expression by ecDNA status, cancer type and select the genes that show significant coefficient for ecDNA status?</p></disp-quote><p><bold>Response.</bold> As requested by the reviewer, and in the more detailed questions below, we added an alternative model with a generalized linear model (GLM) analysis that controlled for tumor subtype. The method itself is described in the Methods section and pasted below. The GLM genes were tested along with the LFC, CorEx, Core genes as described in response to the previous question, and those results are now presented in <xref ref-type="fig" rid="fig2">Figure 2B</xref> and the revised manuscript.</p><p>“We tested each of 16,309 genes independently in a separate logistic regression model using the glm() function in the R stats package (v4.2.0), and retained genes that were significant(<italic>p</italic>-value 0.01). Specifically, the model was defined as glm(𝑦 ~ 𝑔<sub>𝑗</sub> + 𝑡, data = 𝑀, family = binomial(link = 'logit')), where 𝑦 is the response vector where 𝑦<sub>𝑖</sub>=1 if sample 𝑖 ∈ {1, . . . ,870} is ecDNA(+) and 𝑦<sub>𝑖</sub> =0 otherwise, 𝑔<sub>𝑗</sub> is the vector of expression values for gene <italic>j</italic> ∈ {1, . . . ,16309} in samples 𝑖 ∈ {1,. . . ,870}, <italic>t</italic> is the covariate vector representing the tumor subtypes of samples 𝑖 ∈ {1, . . . ,870}, and 𝑀 is the data matrix containing values of gene expression, tumor subtype, and ecDNA status for all samples. The equation for the binomial logistic regression described above is formulated as <inline-formula><mml:math id="sa4m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mi>p</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mo>…</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> where <italic>p</italic> is the probability that the dependent variable 𝑦 is 1, 𝑋 are the independent variables, and 𝛽 are the coefficients of the model. In this case, <italic>k</italic>=1 represents independent variable gene <italic>j</italic> and <italic>k</italic>=2 represents the tumor subtype covariate <italic>t</italic>. Of the 16,309 genes tested independently, 3,012 genes were significant at <italic>p-</italic>value&lt;0.01.”</p><disp-quote content-type="editor-comment"><p>(C) After identifying key features (which the authors inappropriate imply to be causal) they perform a series of enrichment/correlative analysis.</p></disp-quote><p>Response. We have reviewed the document to ensure that we did not use the word ‘causal.’ If the reviewer can point to specific text, we are happy to change the phrasing.</p><disp-quote content-type="editor-comment"><p>(a) It is known that ecDNA status associates with poor survival, and so are cell cycle related signal. Then the association between Boruta genes and those processes is entirely expected. Is it not? The same goes for downregulation of immune processes.</p></disp-quote><p><bold>Response.</bold> We agree with the reviewer that cell cycle related signals and immune related signals are associated with low survival, and so does ecDNA. However, many cellular processes could be associated with low survival (including for example, metabolic processes, protein and DNA biosynthesis, etc.). The unexpected part is that there appear to be only 4 major processes that are upregulated in ecDNA(+) cancers relative to ecDNA(-) cancers, and only one (immune response) that is downregulated.</p><disp-quote content-type="editor-comment"><p>(b) The association with DSB specifically is interesting. Further analysis or discussion of why this should be would strengthen the work.</p></disp-quote><p><bold>Response.</bold> We thank the reviewer for their comment, and agree with their perspective. Note that we devoted a fair amount of text to analysis of DSB pathways. Specifically, we parsed the 4 main pathways in <xref ref-type="fig" rid="fig3">Figure 3B</xref>, and found our data to suggest that many genes in the classical nonhomologous end joining repair pathway are down-regulated in ecDNA(+) samples relative to ecDNA(-) samples. In contrast, Alternative end-joining and homology directed repair pathways are upregulated. This is a surprising result because c-NHEJ is considered to be an important mechanism of DSB repair. We have some lines in the discussion that address this:</p><p>“The DNA damage genes are broadly up-regulated in ecDNA(+) samples, especially in double-strand break repair. Within this broad category of mechanisms, our analysis suggests that alternative DSB repair pathways such as Alt-EJ are preferred relative to classical NHEJ. This is consistent with previous observations of small microhomologies at breakpoint junctions, and has important implications in therapeutic selection that will need to be validated in future experimental studies. We note, however, the microhomology analyses typically study breakpoint junctions, and might ignore double-strand breaks in non-junctional sequences which could be observed, for example at replication-transcription junctions.”</p><p>We note that additional experimental work to corroborate these findings is significant effort and will be part of ongoing research in our collaborators’ laboratories.</p><disp-quote content-type="editor-comment"><p>(c) On page 15, second paragraph, when providing the up versus down CorEx genes, please also provide up versus down for non-CorEx genes as well to get a sense of magnitude.</p></disp-quote><p><bold>Response.</bold> We thank the reviewer for the comment. We note that <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1P</xref> has the complete contingency tables as well as the Fisher Exact Test statistic for all categories. For the specific categories mentioned in the paper, the chi-square tables are reproduced below. As we are citing <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1P</xref> (containing all numbers and the statistic <italic>p</italic>-value) in the main text, we thought it was better to leave the text as it was.</p><p>Category: <bold>Inflammation</bold> (<italic>p</italic>-value: 0.005)</p><p>CorEx: 18 (UP), 76 (DOWN)</p><p>Non-CorEx: 325 (UP), 657 (DOWN)</p><p>Category: <bold>Leukocyte migration and chemotaxis</bold> (<italic>p</italic>-value: 0.03)</p><p>CorEx: 13 (UP), 49 (DOWN)</p><p>Non-CorEx: 213 (UP), 410 (DOWN)</p><p>Category: <bold>Lymphocyte activation</bold> (<italic>p</italic>-value: 0.0075)</p><p>CorEx: 23 (UP), 75 (DOWN)</p><p>Non-CorEx: 334 (UP), 560 (DOWN)</p><p>Category: <bold>Cytokine production</bold> (<italic>p</italic>-value: 0.117)</p><p>CorEx: 6 (UP), 28 (DOWN)</p><p>Non-CorEx: 93 (UP), 208 (DOWN)</p><disp-quote content-type="editor-comment"><p>(d) The finding that Boruta genes are associated with high mutation burden is intriguing because in general mutation burden is associated with better survival and immunotherapy response. This counter-intuitive result should be scrutinized more to strengthen the work.</p></disp-quote><p><bold>Response.</bold> We agree with the reviewer that it is an intriguing observation. However, we are cautious in our interpretation. This is for the following reasons (all mentioned in the text):</p><p>(1) The total mutation burden was significantly higher in ecDNA(+) samples relative to ecDNA(-) samples (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). However, when controlling for cancer type, only glioblastoma, low-grade gliomas, and uterine corpus endometrial carcinoma continued to show differential total mutational burden (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>).</p><p>(2) We tested if specific genes were differentially mutated between the two classes (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). For deleterious/high-impact mutations, TP53 was the only gene whose mutational patterns were significantly higher in ecDNA(+) compared to ecDNA(-) (OR 2.67, Bonferroni adjusted <italic>p</italic>-value 4.22e-07). BRAF mutations, however, were more common in ecDNA(-) samples and were significant to an adjusted <italic>p</italic>-value &lt; 0.1 (OR 0.27).</p><p>(3) In response to another reviewer’s comment, we also tested correlation with variant allele frequencies, and did not find any significant correlation except for TP53. We decided not to include that result in the paper.</p><p>These tissue specific cases might be confounding the main observation, but we have placed all of them together so that the reader can gain a better understanding. It is worth noting that the correlation between high TMB and immunotherapy response is also now controversial, and perhaps not true for all cancer types. See for example (<ext-link ext-link-type="uri" xlink:href="https://www.annalsofoncology.org/article/S0923-7534(21)00123-X/fulltext">https://www.annalsofoncology.org/article/S0923-7534(21)00123-X/fulltext</ext-link>), which suggests that this relationship is not true for Glioma, and in Glioma (which is ecDNA enriched), higher TMB is associated with worse immunotherapy response. Our results are consistent with that finding. We have modified the discussion paragraph to better reflect this.</p><p>“Mutation data alone does not provide as clear a picture of the genes involved in ecDNA maintenance. We did observe that the total mutation burden (TMB) was higher in ecDNA(+) samples. However, that relationship is much less clear after controlling for cancer type. High TMB has been positively correlated with sensitivity to immunotherapy (<xref ref-type="bibr" rid="bib47">Rizvi et al., 2015</xref>), and better patient outcomes; however, the gene expression patterns suggest that immunomodulatory genes are downregulated in ecDNA(+) samples, and patients with ecDNA(+) tumors have worse outcomes (<xref ref-type="bibr" rid="bib28">Kim et al., 2020</xref>). Notably, other results have suggested that the correlation between TMB and response to immunotherapy is not uniform, and it can vary across different tumor subtypes (<xref ref-type="bibr" rid="bib41">McGrail et al., 2021</xref>). <bold>Specifically, our data is consistent with previous results which showed that Gliomas with high TMB have worse response to immunotherapy relative to gliomas with low TMB</bold> (<xref ref-type="bibr" rid="bib41">McGrail et al., 2021</xref>). In general, no collection of gene mutations was predictive of ecDNA status, although mutations in TP53 were more likely in ecDNA(+) samples, and perhaps are an important driver for ecDNA formation (<xref ref-type="bibr" rid="bib38">Luebeck et al., 2023</xref>).”</p><disp-quote content-type="editor-comment"><p>(e) On page 17 &quot;12 of the 47 genes not specifically enriching any known GO biological Process&quot; is confusing. How can individual gene enrich for a GO process?</p></disp-quote><p><bold>Response.</bold> We agree that the statement was incorrectly phrased. We have changed it to state that “Only 12 of the 47 genes were not included in the gene sets of any enriched GO term.”</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>In their manuscript entitled &quot;Transcriptional immune suppression and upregulation of double stranded DNA damage and repair repertoires in ecDNA-containing tumors&quot; Lin et al. describe an important study on the transcriptional programs associated with the presence of extrachromosomal DNA in a cohort of 870 cancers of different origin. The authors find that compared to cancers lacking such amplifications, ecDNA+ cancers express higher levels of DNA damage repair-associated genes, but lower levels of immune-related gene programs.</p><p>This work is very timely and its findings have the potential to be very impactful, as the transcriptional context differences between ecDNA+ and ecDNA- cancers are currently largely unknown. The observation that immune programs are downregulated in ecDNA+ cancers may initiate new preclinical and translational studies that impact the way ecDNA+ cancers are treated in the future. Thus, this study has important theoretical implications that have the potential to substantially advance our understanding of ecDNA+ cancers.</p><p>Strengths</p><p>The authors provide compelling evidence for their conclusions based on large patient datasets. The methods they used and analyses are rigorous.</p><p>Weaknesses</p><p>The biological interpretation of the data remains observational. The direct implication of these genes in ecDNA(+) tumors is not tested experimentally.</p></disp-quote><p><bold>Response.</bold> We agree with the reviewer that experimental tests would be ideal. Towards that, there are some challenges. The immune system genes cannot be tested in cell line models as they need a tumor microenvironment. Tests of DSB repair mechanisms and cell cycle control can be performed in cell-lines, but not with the TCGA samples which are not available. Some of our collaborators are actively working on these topics, but that extensive experimental work is beyond the scope of this paper.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Public Review):</bold></p><p>Summary:</p><p>Using a combination of approaches, including automated feature selection and hierarchical clustering, the author identified a set of genes persistently associated with extrachromosomal DNA (ecDNA) presence across cancer types. The authors further validated the gene set identified using gene ontology enrichment analysis and identified that upregulated genes in extrachromosomal DNA-containing tumors are enriched in biological processes like DNA damage and cell proliferation, whereas downregulated genes are enriched in immune response processes.</p><p>Major comments:</p><p>(1) The authors presented a solid comparative analysis of ecDNA-containing and ecDNA-free tumors. An established automated feature selection approach, Boruta, was used to select differentially expressed genes (DEG) in ecDNA(+) and ecDNA(-) TCGA tumor samples, and the iterative selection process and two-tier multiple hypothesis testing ensured the selection of reliable DEGs. The author showed that the DEG selected using Boruta has stronger predictive power than genes with top log-fold changes.</p><p>(2) The author performed a thorough interpretation of the findings with GO enrichment analysis of biological processes enriched in the identified DEG set, and presented interesting findings, including the enrichment in DNA damage process among the genes upregulated in ecDNA(+) tumors.</p><p>(3) Overall, the authors achieved their aims with solid data mining and analysis approaches applied to public data tumor data sets.</p><p>(4) While it may not be the scope of this study, it will be interesting to at least have some justification for choosing Boruta over other feature selection methods, such as Recursive Feature Elimination (RFE) and backward stepwise selection.</p></disp-quote><p><bold>Response.</bold> We actually agree with the reviewer that some other feature selection methods could work just as well, and note that the Boruta analysis is not our creation, but a published feature selection method (Kursa, <italic>Journal of Statistical Software</italic> September 2010, Volume 36, Issue 11). We use Boruta to identify relevant genes, but the bulk of the paper is to understand the biological processes driven by that gene selection. Even if we had chosen another method that performed slightly better, it likely would not change the main conclusions. However, to address the reviewers concerns on over-reliance on one method, we added a different gene list created by a generalized linear model analysis, with the goal of checking if the expression of a gene could predict the ecDNA status of the sample <italic>after controlling for tumor subtype</italic>. Thus, we tested 5 different genelists in terms of their power in predicting ecDNA. While none of the lists is a great predictor of ecDNA status, the Core and CorEx gene lists are significantly better than the other lists. The Figure below replaces the previous Figure panels 2b and 2c.</p><fig id="sa4fig4" position="float"><label>Author response image 4.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig4-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(1) The authors showed that DESEQ-selected DEGs with top log-fold changes have less strong predictive power and speculated that this may be due to the fact that genes with top log-fold changes (LFC) are confined only to a small subset of samples. It will be interesting to select DEGs with top log-fold changes after first partitioning the tumor samples. For example, randomly partition the tumor samples, identify the DEGs with top LFC, combine the DEGs identified from each partition, then evaluate the predictive power of these DEGs against the Boruta-selected DEGs.</p></disp-quote><p><bold>Response.</bold> This is a great comment. We added a generalized linear model test for selecting genes whose expression is predictive of ecDNA status. The GLM list described above uses a standard methodology (Analysis of Variance) controls for tumor type as a covariate, and its predictive performance is only slightly better than the Top-|LFC| genes, while improving over a random gene set.</p><disp-quote content-type="editor-comment"><p>(2) While the authors showed that the presence of mutations was not able to classify ecDNA(+) and (-) tumor samples, it will be interesting to see if variant allele frequencies of the genes containing these mutations have predictive power.</p></disp-quote><p><bold>Response.</bold> This is a great suggestion. To address the reviewer’s question, we used allelic counts (REFs and ALTs) information from the MC3 variant callset, and calculated allele frequencies of all variants from samples where ecDNA status was available. Next, we conducted a Wilcoxon rank-sum test between VAFs of the ecDNA(+) group and VAFs of the ecDNA(-) group for every mutated gene. We found 1,073 genes with <italic>p</italic>&lt;0.05, but among them, only TP53 passed the multiple testing correction (<italic>padj</italic>&lt;0.05, Benjamini-Hochberg). As the results are identical to the tests based solely on presence of mutations, we decided not to include this data.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Recommendations For The Authors):</bold></p><p>(A) The presentation should be substantially streamlined.</p><p>(B) Preferably use a more intuitive simpler ML approach with fewer parameters to make it more credible. Because there are relatively few samples across numerous cancer types with greater variability in representation, a simpler procedure with transparent controls will be more convincing.</p></disp-quote><p><bold>Response.</bold> We accept the reviewer’s criticism in that other statistical or ML approaches could potentially have been used, and that some are simpler. However, the test used here directly addresses the question: <italic>Find a collection of genes whose expression value is predictive of ecDNA status in the sample.</italic> Because the underlying method in the Boruta analysis uses random forests, it can test predictive power without relying on a linearity assumption implicit in other methods. In this revision, we also compare against a Generalized Linear Model (regression analysis) and show that it is less suited to the specific task above. We address the reviewer concerns about specific parameter choices by showing robustness to the specific parameter. All details are provided in the initial questions, and in the revised manuscript.</p><disp-quote content-type="editor-comment"><p>(C) Avoid using any term implying causality unless you can bring in direct experimental evidence e.g. mutagenesis experiment followed by ecDNA measurement. Some places you use the word 'maintain ecDNA' and other places 'ecDNA impact'. But these are all associations. How can you distinguish causal genes from downstream effects without additional data?</p></disp-quote><p><bold>Response.</bold> We note that the word causal does not appear anywhere in the manuscript, and was not intended. Additionally we have revised the manuscript and are open to specific changes requested by the reviewer or the editors.</p><disp-quote content-type="editor-comment"><p>(D) Along these lines, if Boruta genes are indeed causal, one would expect Boruta-Up genes to be amplified more than expected in the ecDNA+; converse for Boruta-down genes.</p></disp-quote><p><bold>Response.</bold> We did not understand the reviewer’s question. By “amplified,” if the reviewer means “amplification of transcript level,” then that is exactly what the Boruta analysis is showing. Specifically, for each gene, we have the ability to pick a transcript level cut-off ‘<italic>t</italic>’ so that samples in which the expression is higher than <italic>t</italic> are more likely to be ecDNA(+). However, we are not claiming that there is causality, just that the transcript level is (weakly) predictive of the ecDNA status of the sample.</p><disp-quote content-type="editor-comment"><p>(E) A strawman control should be a simple regression-based gene identification that controls for ecDNA status and cancer type.</p></disp-quote><p><bold>Response.</bold> We agree that this was a very good suggestion. In the revision, we have applied a GLM, which controls for tumor type. Thus, we have 5 gene-lists (including the Core and CorEx genes). As described in the revised manuscript but also in response to the main comments above, none of the lists are a great predictor. However, the CorEx and Core genes are significantly better at predicting ecDNA status of a sample.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Recommendations For The Authors):</bold></p><p>Comments</p><p>(1) The analysis hinges on a classification of tumors into ecDNA(+) and ecDNA(-) using AmpliconClassifier. It would be good to know how robust the outcomes are with respect to the performance of AmpliconClassifier - how many false positives and negatives willAmpliconClassifier generate on this dataset and how would this influence the CorEx genes?</p></disp-quote><p><bold>Response.</bold> This is a very reasonable request. AA has been extensively tested on established cell-lines for its ability in predicting ecDNA status, and this information is published in multiple venues, including Kim, Nature genetics 2020, and shows precision 85% for recall 83%. For completeness, we have reproduced the relevant plot from that paper here, and the relevant text here, but are not including it in the manuscript.</p><p>“To evaluate the accuracy of the AmpliconArchitect predictions, we analyzed whole-genome sequencing data from a panel of 44 cancer cell lines, and examined tumor cells in metaphase. We used 35 unique fluorescence in-situ hybridization (FISH) probes in combination with matched centromeric probes (81 distinct “cell-line, probe” combinations) to determine the intranuclear location of amplicons (Supplementary Table 2). Following automated analysis &gt;1,600 images, we observed that 85% of amplicons characterized as ‘Circular’ by whole genome sequencing profile demonstrated an extrachromosomal fluorescent signal, representing the positive predictive value. Of the amplicons corresponding to extrachromosomally located FISH probes, 83% were classified as Circular, representing the sensitivity (Extended Data Fig. 1A).”</p><fig id="sa4fig5" position="float"><label>Author response image 5.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig5-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(2) It is unclear why genes are labeled Boruta genes when they are present in 10 out of 200 runs, this seems like an unexpectedly low number. How did the authors arrive at this number? Do the authors have any ground truth to estimate how well Boruta works in this setting and implementation?</p></disp-quote><p><bold>Response.</bold> This is a great question and asked by another reviewer as well. Given the weakness of an individual gene as a classifier, its repeated selection in multiple Boruta trials is already a significant event. By requiring a gene to be picked in 5% of the trials (10/200), we were selecting a small, but more robust list of genes. However, to further explore the reviewer’s concerns, we also applied 8 other selection criteria ranging from 5 (of 200 Boruta trials) to 200 of 200 Boruta trials. See Figure below. The number of CorEx genes expectedly decreases with increasing stringency. However, of the 187 GO terms that were enriched by UP-genes, 93 terms (50%) were enriched regardless of the cut-off (see Figure below), and 153 terms (82%) were enriched in at least 5 of the 8 cut-offs. Given that the remaining analysis works on the hierarchy of GO terms and finds 4 GO-categories (Mitotic Cell Cycle, G1/S, G2/M; cell-division; DSB DNA Damage response; and the HOX Gene cluster) enriched by UP-regulated genes, those conclusions would hold regardless of the specific cut-off.</p><fig id="sa4fig6" position="float"><label>Author response image 6.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig6-v1.tif"/></fig><p>The number of GO terms that were enriched by DOWN-regulated genes is smaller, only 73, and falls rapidly for higher cut-offs, with 25 at a cut-off of 15. Therefore we see fewer terms enriched for more stringent cut-offs. However, they all support immune processes. These results do suggest that there are fewer genes that are consistently down-regulated in ecDNA(+) cancers, and expression change in a small number of genes may be sufficient to promote conditions for ecDNA.</p><p>We have added the figure as a supplemental figure and have added the following text to the manuscript:</p><p>“Any CorEx gene is either a Core gene that was selected as a feature in at least 5% of 200 Boruta trials, or be highly co-expressed with a Core gene. Because the selection criterion of 5% is arbitrary, we also tested robustness with 8 other cut-offs ranging from 5-of-200 to 200-of-200 Boruta trials. The number of CorEx genes expectedly decreases with more stringent cut-offs.</p><p>However, of the 187 GO terms that were enriched by 262 CorEx UP-genes using 10 of 200 Boruta trials as the selection criteria, 93 terms (49.7%) were enriched for each cut-off (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>), and 155 terms (82.9%) were enriched in at least 5 of the 8 cut-offs. Given that our subsequent analyses utilized the hierarchy of GO terms and identified 4 GO-categories enriched by UP-regulated genes, the conclusions would hold regardless of the specific cut-off.”</p><disp-quote content-type="editor-comment"><p>(3) Authors extend the core gene set with co-expressed genes, arguing that &quot;gene C&quot; would not add predictive power in addition to &quot;gene B&quot; and is therefore not identified as a Boruta gene. However, from its description in the manuscript (summarized: &quot;Boruta [...] selects the highest feature importance score, s, of shadow features as a cut off, and returns features with a higher score than s.&quot;), it isn't immediately obvious to me why Boruta would not return both genes B and C. Maybe the authors could explain this better.</p></disp-quote><p><bold>Response.</bold> We consider the following.</p><p>(1) Consider 100 ecDNA(+) and 100 ecDNA(-) samples. Let the expression levels of genes B and C in the data-sets be as described in the figure below; y-axis is the gene expression, and x-axis is just a listing of all samples, with green color denoting ecDNA(+) samples and orange color denoting ecDNA(-) samples.</p><fig id="sa4fig7" position="float"><label>Author response image 7.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig7-v1.tif"/></fig><p>(2) Then, if we choose gene B and a transcript level of 1.25, we have a perfect prediction of ecDNA status because all samples where gene B has a transcript level higher than 1.25 are ecDNA(+) and otherwise they are ecDNA(-). Similarly, using Gene C, we can get perfect predictions. Thus, when Boruta has to select a gene, it will pick either Gene B or Gene C, because picking both will not improve prediction. We can therefore use Boruta to pick one gene, and then co-expression clustering to pick the other gene.</p><p>As an example, cluster #3 consists of 21 genes that were up-regulated in ecDNA(+) samples and enriched in cell-cycle related biological processes (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1D</xref>). While these genes were expressed similarly in ecDNA(+) samples, and separately, in ecDNA(-) samples, out of the 21 genes, only 9 genes were selected in at least 10 out of 200 Boruta trials (i.e., Core genes). Of the 12 remaining genes (i.e., CorEx genes), 8 genes were not selected by the Boruta method at all, 3 genes were selected in less than 5 out of 200 Boruta trials, and 1 gene was selected in 9 out of 200 Boruta trials.</p><fig id="sa4fig8" position="float"><label>Author response image 8.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig8-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(4) In Fig 2a, I would like to see the variability of the precision and recall in the main text, not only the maximum values. Authors could plot mean + standard deviation for precision and recall separately, or use S2a/b.</p></disp-quote><p><bold>Response.</bold> We have replaced Figures 2b and 2c with a combined figure (<xref ref-type="fig" rid="fig2">Figure 2B</xref>) that gives a box-plot describing the distribution of recall values for 5 gene lists: four from the original manuscript, and another gene list created using a Generalized Linear Model (GLM).</p><fig id="sa4fig9" position="float"><label>Author response image 9.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig9-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(5) Since the authors analyze bulk RNA, the gene expression signatures they notice could, in principle, originate from non-tumor cells as well. I do not believe this is the case, however, the paper would be strengthened by an analysis that shows that the difference in expression patterns of the Corex genes between ecDNA(+) and ecDNA(-)-samples does come from tumor cells. One way of showing this would be by using single-cell mRNA-sequencing data, and another way of showing this would be to show that Corex gene-expression correlates with tumor purity in bulk samples.</p></disp-quote><p><bold>Response.</bold> The reviewer is correct. Unfortunately, our analysis requires data with whole-genome sequencing (WGS) for ecDNA prediction, as well as RNA-seq for transcriptome profiling. The TCGA data-set is the only available data-set with a significant number of samples that includes both WGS and RNA-seq. They have not made tissue samples available for scRNA analysis, to our knowledge. The reviewer raises an important question regarding purity, but testing if CorEx gene expression correlates with tumor purity would require a large range of purity values, something that scientists would avoid when collecting samples.</p><p>However, the presence of non-cancer tissue (impurity) could reduce sensitivity of ecDNA detection, and therefore, change the results. To better investigate this, we started with a publication that investigated multiple tumor purity metrics and devised a composite score (CPE; Aran et al., 2015). Using their composite tumor purity, we find that ecDNA(-) samples have slightly lower purity than ecDNA(+) samples (<italic>p</italic>-value 0.0036; <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4A</xref>).</p><p>This result is not surprising because one would expect lower detection of ecDNA in less pure samples. The presence of undetected ecDNA in ecDNA(-) samples would confound the results by reducing the discriminating power of genes, but would not give false results. To test this, we measured the expression directionality in CorEx genes in all samples versus samples which had a high tumor purity (CPE 0.8). The results suggest that the <italic>p</italic>-values of directionality in the pure samples were highly correlated with the expression data from all samples (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4B</xref>).</p><fig id="sa4fig10" position="float"><label>Author response image 10.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig10-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(6) The biological interpretation of the data remains a bit too observational. Can the authors offer an interpretation of the enriched GO terms? And are any of these genes already implicated in ecDNA(+) tumors?</p></disp-quote><p><bold>Response.</bold> To answer the second question first, prior to our study, the focus was on genes that were amplified on ecDNA. Indeed many oncogenes known to be amplified in cancer are in fact amplified on ecDNA (Turner, Nature 2017, Kim Nature genetics 2020). This study is unique in that it identifies genes whose expression values are predictive of ecDNA(+) status. The Figure below lists 24 genes most frequently amplified on ecDNA from Kim, Nature Genetics 2020. With the exception of EGFR and CDK4, none of these 24 genes was included in the list of the 65 genes reported by us as the most frequently selected genes in the Boruta trials (lowest harmonic rank). Thus, most persistent CorEx genes do not lie on ecDNA. However, they all play important roles in biological processes relevant to cancer pathology including Immune Response, Mitotic cell Cycle, Cell division, and DSB repair. We agree with the reviewer that the results are observational (although statistically significant in populations), and some of our collaborators are actively working to experimentally validate some of these genes. The experimental work, however, is beyond the scope of this paper.</p><p>We have added the following statement to the manuscript. “Notably, of the 24 genes most frequently expressed on ecDNA,2 only EGFR and CDK4 were included in the list of 65 genes, suggesting that the most persistent CorEx genes do not themselves appear frequently on ecDNA.”</p><fig id="sa4fig11" position="float"><label>Author response image 11.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88895-sa4-fig11-v1.tif"/></fig><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Recommendations For The Authors):</bold></p><p>Minor comments:</p><p>(1) The authors performed gene ontology enrichment test but referred to it as gene set enrichment analysis. Usually gene set enrichment analysis does not refer to Fischer's exact test-based analysis but rather the one described in Subramanian et al 2005. The term correction should be made to avoid confusion.</p></disp-quote><p><bold>Response.</bold> We have rephrased text in the manuscript to prevent confusion between enrichment analysis on gene sets using an one-sided Fisher’s exact test and the Gene Set Enrichment Analysis (GSEA) method that exists as a software. We have also revised the header in the methods section from “Gene set enrichment analysis” to “Gene Ontology (GO) enrichment analysis”.</p><disp-quote content-type="editor-comment"><p>(2) A couple of figures could use more detailed labels and captions. In Figure 2c, it is unclear what the numbers 100 and 54 right next to the Cliff's Delta heatmap indicate. In Figures 3a and 4a, it is not immediately clear what the barplot on top of the heatmap indicates and there is no label for the y-axis.</p></disp-quote><p><bold>Response.</bold> These are good suggestions, and we have added descriptions to the figure captions.</p></body></sub-article></article>