<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">62548</article-id><article-id pub-id-type="doi">10.7554/eLife.62548</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Divergence in alternative polyadenylation contributes to gene regulatory differences between humans and chimpanzees</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-165075"><name><surname>Mittleman</surname><given-names>Briana E</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4979-4652</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-73687"><name><surname>Pott</surname><given-names>Sebastian</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4118-6150</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-166687"><name><surname>Warland</surname><given-names>Shane</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-206723"><name><surname>Barr</surname><given-names>Kenneth</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-0769-7053</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-198179"><name><surname>Cuevas</surname><given-names>Claudia</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-2844"><name><surname>Gilad</surname><given-names>Yoav</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-8284-8926</contrib-id><email>gilad@uchicago.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Genetics, Genomics and Systems Biology, University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Human Genetics, University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>Section of Genetic Medicine, Department of Medicine, University of Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Coop</surname><given-names>Graham</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, Davis</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Barkai</surname><given-names>Naama</given-names></name><role>Senior Editor</role><aff><institution>Weizmann Institute of Science</institution><country>Israel</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>17</day><month>02</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e62548</elocation-id><history><date date-type="received" iso-8601-date="2020-08-28"><day>28</day><month>08</month><year>2020</year></date><date date-type="accepted" iso-8601-date="2021-02-12"><day>12</day><month>02</month><year>2021</year></date></history><permissions><copyright-statement>© 2021, Mittleman et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Mittleman et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-62548-v2.pdf"/><abstract><p>While comparative functional genomic studies have shown that inter-species differences in gene expression can be explained by corresponding inter-species differences in genetic and epigenetic regulatory mechanisms, co-transcriptional mechanisms, such as alternative polyadenylation (APA), have received little attention. We characterized APA in lymphoblastoid cell lines from six humans and six chimpanzees by identifying and estimating the usage for 44,432 polyadenylation sites (PAS) in 9518 genes. Although APA is largely conserved, 1705 genes showed significantly different PAS usage (FDR 0.05) between species. Genes with divergent APA also tend to be differentially expressed, are enriched among genes showing differences in protein translation, and can explain a subset of observed inter-species protein expression differences that do not differ at the transcript level. Finally, we found that genes with a dominant PAS, which is used more often than other PAS, are particularly enriched for differentially expressed genes.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd><italic>P. troglodytes</italic></kwd><kwd>alternative polyadenylation</kwd><kwd>genomics</kwd><kwd>comparative</kwd><kwd>translation</kwd><kwd>gene expression</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd><kwd>Other</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100007234</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32GM09197</award-id><principal-award-recipient><name><surname>Mittleman</surname><given-names>Briana E</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>F31HL149259</award-id><principal-award-recipient><name><surname>Mittleman</surname><given-names>Briana E</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01HG010772</award-id><principal-award-recipient><name><surname>Gilad</surname><given-names>Yoav</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM13172</award-id><principal-award-recipient><name><surname>Gilad</surname><given-names>Yoav</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100006108</institution-id><institution>National Center for Advancing Translational Sciences</institution></institution-wrap></funding-source><award-id>K12HL119995</award-id><principal-award-recipient><name><surname>Pott</surname><given-names>Sebastian</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>A comparative analysis of human and chimpanzee polyadenylation site usage establishes alternative polyadenylation as another key mechanism underlying the genetic regulation of transcript and protein expression levels in primates.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Humans and our close primate relatives exhibit a striking array of phenotypic diversity despite sharing homologous proteins with nearly identical amino acid sequences (<xref ref-type="bibr" rid="bib40">King and Wilson, 1975</xref>). Understanding how this diversity is propagated from genomic sequence to mRNA and then to protein necessitates an understanding of the regulatory mechanisms that occur before, during, and after transcription. Studying gene regulatory features in humans and other primates has long provided opportunities to understand genome evolution and function. For example, studies comparing patterns of epigenetic marks in primates have provided mechanistic explanations that link genetic variation and divergence to differences in gene expression levels (<xref ref-type="bibr" rid="bib2">Banovich et al., 2014</xref>; <xref ref-type="bibr" rid="bib11">Cain et al., 2011</xref>; <xref ref-type="bibr" rid="bib50">McVicker et al., 2013</xref>; <xref ref-type="bibr" rid="bib58">Pai et al., 2011</xref>). Although many studies have focused on inter-species differences in the regulation of gene expression, fewer studies have addressed isoform-level variation, which contributes to differences in mRNA, translation, and protein levels between species (<xref ref-type="bibr" rid="bib9">Blekhman et al., 2010</xref>; <xref ref-type="bibr" rid="bib12">Calarco et al., 2007</xref>; <xref ref-type="bibr" rid="bib59">Pai et al., 2016</xref>).</p><p>The main mechanisms that contribute to mRNA isoform diversity are alternative splicing and alternative polyadenylation (APA). Alternative splicing produces different combinations of coding sequences in mature mRNA and protein. APA occurs at genes that have more than one polyadenylation site (PAS) and can result in mRNAs with different coding sequences or variable 3' UTR lengths. Like alternative splicing, APA that occurs within the gene body can affect protein sequence and function. (<xref ref-type="bibr" rid="bib41">Lee et al., 2018</xref>; <xref ref-type="bibr" rid="bib60">Pan et al., 2006</xref>; <xref ref-type="bibr" rid="bib70">Sandberg et al., 2008</xref>; <xref ref-type="bibr" rid="bib76">Tian and Manley, 2017</xref>; <xref ref-type="bibr" rid="bib78">Vasudevan et al., 2002</xref>; <xref ref-type="bibr" rid="bib84">Yao et al., 2018</xref>). Unlike with alternative splicing, a PAS in a coding region will lead to a truncated isoform rather than a different combination of included exons. APA that occurs outside of the coding sequence, in the 3' UTRs, can lead to differential inclusion of protein-binding motifs that can affect translational efficiency, mRNA stability, and mRNA localization (<xref ref-type="bibr" rid="bib49">Mayr, 2017</xref>; <xref ref-type="bibr" rid="bib76">Tian and Manley, 2017</xref>). Yet, despite its potential to produce tremendous variation in mRNA and protein regulation, few studies have explored the contribution of APA to regulatory divergence between species. Indeed, our current understanding of APA conservation in mammals comes from few comparative studies of humans and rodents (<xref ref-type="bibr" rid="bib1">Ara et al., 2006</xref>; <xref ref-type="bibr" rid="bib80">Wang et al., 2018a</xref>). However, these studies used sequence conservation rather than direct measurements of PAS usage to characterize APA (<xref ref-type="bibr" rid="bib80">Wang et al., 2018a</xref>). Thus, it remains possible that many mammalian PAS are functionally divergent despite having similar sequences.</p><p>To gain insight into APA conservation in humans and chimpanzees and understand how differences in APA contribute to gene regulation, we performed 3' sequencing (3' Seq) of mRNA isolated from nuclei collected from human and chimpanzee lymphoblastoid cell lines (LCLs). We integrated PAS usage measurements with RNA-sequencing (RNA-seq) data collected from the same cell lines to understand the relationship between APA and gene expression levels. Finally, we used ribosome profiling and protein measurements previously collected in the same panel of human and chimpanzee LCLs to explore the effects of APA on protein levels (<xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>). We took this approach because an understanding of how APA isoform usage varies among primates could help explain why some human and chimpanzee genes are differentially expressed at either the mRNA or protein levels, but not both.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Describing APA in human and chimpanzee LCLs</title><p>We performed 3' Seq of mRNA from six human and six chimpanzee LCLs, which we have previously used to study a variety of other functional genomic phenotypes, such as ribosome profiling to infer translation levels and mass spectrometry to measure protein levels (<xref ref-type="bibr" rid="bib11">Cain et al., 2011</xref>; <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>; <xref ref-type="bibr" rid="bib85">Zhou et al., 2014</xref>). We collected mRNA separately from whole cells and isolated nuclei. The two cellular fractions serve as biological replicates, which we mainly used to examine the quality of our data (see 'Materials and methods'). By collecting data from isolated nuclei, we were able to capture polyadenylated transcripts before they became undetectable due to other regulatory processes, such as isoform-specific decay (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>).</p><p>We mapped human 3' Seq reads to the GRCh38 reference genome (<xref ref-type="bibr" rid="bib71">Schneider et al., 2017</xref>) and chimpanzee 3' Seq reads to the panTro6 reference genome (<xref ref-type="bibr" rid="bib16">Chimpanzee Sequencing and Analysis Consortium, 2005</xref>) (see 'Materials and methods'). 3' Seq relies on a poly(dT) primer to target the poly(A) tail of mRNA molecules; however, it can also misprime by binding a sequence of genomic adenines. To account for mispriming of off-target genomic sequences, we removed reads that mapped to genomic regions containing ≥70% adenine or six consecutive adenine bases in the 10 bp directly upstream of the mapped location (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>; <xref ref-type="bibr" rid="bib72">Sheppard et al., 2013</xref>; <xref ref-type="bibr" rid="bib75">Tian et al., 2005</xref>) (see 'Materials and methods'). In addition, we treated all ambiguous nucleotide positions as adenines to ensure that differences in reference genome quality did not bias the detection of PAS or mispriming events (see 'Materials and methods'). As expected, the filtered aligned sequences, in both species, were enriched at transcription end sites and showed a similar distribution along orthologous 3' UTRs (see 'Materials and methods', <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Next, we used a custom peak calling method to ascertain PAS in humans and chimpanzees separately (see 'Materials and methods').</p><p>To compare PAS usage across species, we needed to identify the orthologous genomic regions of all PAS in our dataset, regardless of the species in which they were originally annotated. As we were unable to confidently identify orthologous PAS at base pair resolution (inferring synteny at base pair resolution in non-coding regions is challenging; Broad Institute <xref ref-type="bibr" rid="bib46">Lindblad-Toh et al., 2011</xref>), we extended each PAS by 100 bp upstream and downstream. We then used a reciprocal liftover pipeline to obtain an inclusive set of PAS regions with which we could confidently compare PAS usage between species (see 'Materials and methods'). Prior to the filtering described below, we identified 445,944 orthologous regions.</p><p>To quantify PAS usage, we first assigned each PAS to a gene using the hg38 RefSeq annotation (<xref ref-type="bibr" rid="bib64">Pruitt et al., 2004</xref>). We then computed the usage for each PAS in each individual as the fraction of reads mapping to one PAS over the total number of reads mapping to any PAS for the same gene (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>). We excluded PAS in lowly expressed genes (log<sub>2</sub>(CPM) ≤2 in four or more individuals) or with less than 5% usage, as measurements from sparse data are highly susceptible to random error (see 'Materials and methods'). We observed a strong correlation between PAS usage in mRNA from the nuclear and total cell fractions in all but one cell line (human NA18499; <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>), which we subsequently excluded from the study. We re-identified PAS after removing all data from NA18499 and re-quantified PAS usage using nuclear 3' Seq data from five human and six chimpanzee LCLs. Using this analysis pipeline, we identified a total of 44,432 PAS in 9518 genes, which we used for all downstream analyses. On a genome-wide scale, we found that mean PAS usage is highly correlated between species (Pearson’s correlation, 0.9, p&lt;2.2×10<sup>−16</sup>, <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>). However, as expected, 41.8% of the variation in PAS usage (as explained by the top principal component of the data) is highly correlated with species (Pearson’s correlation 0.99, p=2.95×10<sup>−8</sup>, <xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>), indicating substantial divergence in PAS usage.</p><p>We used a number of analyses to confirm that our ability to detect PAS was not biased by gene expression level or species. If our ability to detect PAS was biased by gene expression, we might expect a positive correlation between gene expression level and the number of PAS we detected. In our data, the number of PAS per gene is negatively correlated with gene expression level in both species (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>; human: Pearson’s correlation −0.17, p&lt;2.2×10<sup>−16</sup>; chimpanzee: Pearson’s correlation −0.19, p&lt;2.2×10<sup>−16</sup>). If our ability to detect PAS were biased by species, we would expect to identify more PAS per gene in one species over the other. This is neither the case genome-wide nor when we test each gene independently. We identified, on average, 3.87 PAS per gene in humans and 3.46 PAS per gene in chimpanzees. On average, per gene, the number of PAS in human minus the number of PAS in chimpanzee is 0.39 with a median value of 0 (<xref ref-type="fig" rid="fig1s7">Figure 1—figure supplement 7</xref>). Moreover, as expected, the physical distribution of PAS across genes is conserved, with the majority of PAS located in 3' UTRs (17,688, 40% in chimpanzee; and 17,620, 40% in human) and a considerable proportion located in introns (14,095, 32% in chimpanzee; and 14,119, 32% in human) (<xref ref-type="fig" rid="fig1">Figure 1A</xref>).</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Sequence conservation of polyadenylation sites (PAS) between humans and chimpanzees.</title><p>(<bold>a</bold>) Genic locations for 44,074 PAS identified in chimpanzee (left) and 44,130 PAS identified in human (right). (<bold>b</bold>) Mean PhyloP scores for PAS regions (yellow) as well as three 200 bp regions upstream and downstream (orange). (<bold>c</bold>) Proportion of human and chimpanzee PAS regions with each of the 12 annotated signal site motifs from <xref ref-type="bibr" rid="bib5">Beaudoing et al., 2000</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Density of merged human and chimpanzee 3' sequencing (3' Seq).</title><p>(<bold>A</bold>) Coverage of six chimpanzee, nuclear 3' Seq reads along Refseq transcripts. (<bold>B</bold>) Coverage of five human, nuclear 3' Seq reads along Refseq transcripts. (<bold>C</bold>) Coverage of six chimpanzee, nuclear 3' Seq reads orthologous 3' UTRs. (<bold>D</bold>) Coverage of five human, nuclear 3' Seq reads orthologous 3' UTRs.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp1-v2.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Model representation of usage calculation.</title><p>Representation of polyadenylation sites (PAS) usage calculation. Usage is a ratio of reads at each PAS to the number of reads mapping to any PAS in the same gene. Reproduced from <xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp2-v2.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>NA18499 removed from analysis due to low correlation between fractions.</title><p>Pearson's correlation between PAS usage calculated using nuclear 3' sequencing (3' Seq) libraries and total mRNA 3' Seq libraries, calculated using sites reaching 5% in one species in both fractions. (Plot from previous git commit 30ff122 on April 9, 2020, <ext-link ext-link-type="uri" xlink:href="https://github.com/brimittleman/Comparative_APA">https://github.com/brimittleman/Comparative_APA</ext-link>; <xref ref-type="bibr" rid="bib52">Mittleman, 2021a</xref>). </p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp3-v2.tif"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Polyadenylation sites (PAS) usage is highly correlated across species.</title><p>(<bold>A</bold>) Correlation between human and chimpanzee PAS usage for 44,432 PAS. Red line is a 1:1 line. Linear regression line and Pearson's correlation plotted in blue. (<bold>B</bold>) Pairwise correlation for human and chimpanzee PAS usage.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp4-v2.tif"/></fig><fig id="fig1s5" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 5.</label><caption><title>Variation in polyadenylation sites (PAS) usage.</title><p>(<bold>A</bold>) Plot of first two principal components (PCs) calculated by a PC analysis (PAC) on PAS usage (44,432 PAS). Chimpanzee samples are shown in red and human samples are shown in blue. (<bold>B</bold>) Heatmap representing correlation between technical factors and PCs. Y axis factors include: Species, Extraction date, Collection person, AverageAlive (average of two live dead calculations at time of collection), RIN score, RNA concentration. Explanation of factors and values in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp5-v2.tif"/></fig><fig id="fig1s6" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 6.</label><caption><title>Polyadenylation sites (PAS) detection likely not biased by expression level.</title><p>(<bold>A</bold>) Normalized gene expression plotted against the number of PAS detected at 5% usage in chimpanzee. (<bold>B</bold>) Normalized gene expression plotted against the number of PAS detected at 5% usage in human. The R package ggpubr was used to plot linear regression lines and calculate Pearson's correlations.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp6-v2.tif"/></fig><fig id="fig1s7" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 7.</label><caption><title>Polyadenylation sites (PAS) detection likely not biased by species.</title><p>Histogram of the number of PAS detected at 5% usage in human minus the number of 5% usage in chimpanzees. Red vertical line represents mean difference (0.39).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp7-v2.tif"/></fig><fig id="fig1s8" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 8.</label><caption><title><xref ref-type="fig" rid="fig1">Figure 1B</xref> separated by genic location.</title><p>Mean PhyloP scores for polyadenylation sites (PAS) regions (yellow) and 200 base pair bins upstream and downstream of PAS (orange). A one-sided Wilcoxon test was used to test for increased PhyloP in PAS regions (coding region: p&lt;2.2×10<sup>−16</sup>; 5 kb downstream of genes: p=7.02×10<sup>−6</sup>; intron: p=0.99; 3' UTR: p&lt;2.2×10<sup>−16</sup>; 5' UTR: p=0.011).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp8-v2.tif"/></fig><fig id="fig1s9" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 9.</label><caption><title>Polyadenylation sites (PAS) regions are conserved in the mammalian lineage.</title><p>Mean PhyloP scores for PAS regions (yellow) and 200 base pair bins upstream and downstream of PAS (orange) (<bold>A</bold>) <xref ref-type="fig" rid="fig1">Figure 1B</xref> recapitulated with PhyloP scores calculated over 20 vertebrate species. (<bold>B</bold>) <xref ref-type="fig" rid="fig1s8">Figure 1—figure supplement 8</xref> recapitulated with PhyloP scores calculated over 20 vertebrate species. One-sided Wilcoxon test was used to test for increased PhyloP in PAS regions (cds-coding region: p&lt;2.2×10<sup>−16</sup>; end-5 kb downstream of genes: p&lt;2.2×10<sup>−16</sup>; intron: p&lt;2.2×10<sup>−16</sup>; utr3-3' UTR: p&lt;2.2×10<sup>−16</sup>; utr5-5' UTR: p&lt;2.2×10<sup>−16</sup>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp9-v2.tif"/></fig><fig id="fig1s10" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 10.</label><caption><title>PAS with AATAAA and ATTAAA are used more often.</title><p>Mean PAS usage of the top two signal site motifs in human and chimpanzee plotted by annotated signal site.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp10-v2.tif"/></fig><fig id="fig1s11" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 11.</label><caption><title>Chimp-specific polyadenylation sites (PAS) likely due to loss of signal site in human lineage.</title><p>(<bold>A</bold>) Integrative Genomics Viewer (IGV) track for example of a chimpanzee-specific PAS in MAN2B2 gene. Top track is merged coverage from five human nuclear 3' sequencing (3' Seq) libraries. Chimp track is merged coverage from six chimpanzee nuclear 3' Seq libraries lifted to human genome with CrossMap. (<bold>B</bold>) Sequence alignment for region upstream of proximal PAS from UCSC Genome Browser black box indicates the signal site location. Canonical signal site is the ancestral state and was lost in the human lineage.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp11-v2.tif"/></fig><fig id="fig1s12" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 12.</label><caption><title>Reciprocal liftover pipeline.</title><p>Reciprocal liftover pipeline for unfiltered polyadenylation sites (PAS) including the number of sites remaining at each step. Liftover using UCSC liftover tool and chain files downloaded from UCSC Genome Browser.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig1-figsupp12-v2.tif"/></fig></fig-group><p>To assess sequence conservation in PAS regions, we downloaded PhyloP scores computed over 100 vertebrate genomes from the UCSC Genome Browser and calculated mean PhyloP scores in PAS regions. Higher mean PhyloP scores correspond to regions of higher sequence conservation and thus slower evolution (<xref ref-type="bibr" rid="bib63">Pollard et al., 2010</xref>). Overall, sequence elements at PAS are more conserved than 200 base pair surrounding regions 1 kb on either side of the PAS (<xref ref-type="fig" rid="fig1">Figure 1B</xref>, Wilcoxon rank sum test, p&lt;2.2×10<sup>−16</sup>). This pattern also holds independently for PAS in all genic locations other than introns (<xref ref-type="fig" rid="fig1s8">Figure 1—figure supplement 8</xref>). By repeating these analyses with PhyloP score computed over 20 vertebrates, we show the PAS are also conserved compared to the surrounding regions in the mammalian lineage (<xref ref-type="fig" rid="fig1s9">Figure 1—figure supplement 9</xref>).</p><p>We identified 302 and 357 human- and chimpanzee-specific PAS, respectively (see 'Materials and methods'). Suggesting biological significance, compared to the genes in which we identified a PAS in both species, the genes with species-specific PAS are enriched for a number of general cellular processes (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>, see 'Materials and methods'). It has been previously shown that most PAS are directly preceded by 1 of 12 annotated sequence motifs that recruit cleavage and polyadenylation machinery to mRNA molecules as they are transcribed (<xref ref-type="bibr" rid="bib5">Beaudoing et al., 2000</xref>). We asked if creation or disruption of a signal site motif could be responsible for species-specific PAS by mapping signal site motifs in both human and chimpanzee for each PAS region. Although human and chimpanzee PAS regions are equally likely to contain each of the 12 annotated signal sites (<xref ref-type="fig" rid="fig1">Figure 1C</xref>), only the top two most commonly used motifs, AATAAA and ATTAAA, are associated with increased PAS usage (<xref ref-type="fig" rid="fig1s10">Figure 1—figure supplement 10</xref>). Not only have other studies revealed a similar preference for these two motifs, cryo-electric microscopy analysis of the recognition machinery suggests that the mutation from the canonical signal site to the second most used site requires a smaller RNA rearrangement than other variations of the motif (<xref ref-type="bibr" rid="bib5">Beaudoing et al., 2000</xref>; <xref ref-type="bibr" rid="bib74">Sun et al., 2018</xref>). Thus, we considered only the presence or absence of these two motifs in subsequent analyses. We classified sites as having a species-specific signal site if we only identified the AATAAA or ATTAAA motif in one species. Of the 302 human-specific PAS, 14 have human-specific signal sites and 6 have chimpanzee-specific signal sites. Of the 357 chimpanzee-specific PAS, 24 have a chimpanzee-specific signal site and 6 have a human-specific signal site. These numbers are small; still, species-specific signal sites are more abundant than expected by chance among species-specific PAS in human (5.7×, hypergeometric test, p=2.30×10<sup>−7</sup>) and in chimpanzee (8.3×, hypergeometric test, p=3.2×10<sup>−15</sup>), suggesting that signal site changes can explain a subset of differences in PAS usage. For example, we identified a chimpanzee-specific PAS about 1 kb upstream of a PAS used in both species in the 3' UTR of <italic>MAN2B2</italic>. The ancestral signal site conserved in chimpanzee is AATAAA; however, there has been a T to C transition in the human lineage (<xref ref-type="bibr" rid="bib7">Blanchette et al., 2004</xref>; <xref ref-type="fig" rid="fig1s11">Figure 1—figure supplement 11</xref>). This transition is likely responsible for the loss of PAS in humans. The MAN2B2 gene involved with lysosomal degradation of glycoproteins, and in 2019 a physician diagnosed a patient with immune deficiency as a result of a loss of function mutation in the gene (<xref ref-type="bibr" rid="bib79">Verheijen et al., 2020</xref>).</p></sec><sec id="s2-2"><title>Characterizing inter-species differences in PAS usage</title><p>While a few hundred PAS are species-specific, the majority of PAS (98.5%) were identified in both species. We thus sought to characterize quantitative differences in APA patterns between human and chimpanzee by estimating the difference in the usage of individual PAS in each species. To do so, we used the leafcutter differential splicing tool (<xref ref-type="bibr" rid="bib43">Li et al., 2018</xref>), which allowed us to test for differences in normalized PAS usage fractions while accounting for gene structure (see 'Materials and methods'). Using this approach, at a false discovery rate (FDR) of 5% we identified 2342 PAS (in 1705 genes) whose usage differs by 20% or more between the species (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). We applied an arbitrary effect size cutoff to focus on larger inter-species differences, which are more likely to have functional consequences. The list of all PAS whose usage differs between the species, regardless of the effect size, is available in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Alternative polyadenylation (APA) is functionally conserved in both species.</title><p>(<bold>a</bold>) Proportion of polyadenylation sites (PAS) and genes differentially used at PAS and isoform diversity level. (Left) Divergent PAS are the 2342 PAS differentially used at 5% FDR. Conserved are the PAS not differentially used at 5% FDR. Not tested PAS were removed from analysis by leafcutter tool. (Middle) PAS level differentially used PAS reported at the gene level. Divergent genes are the 1705 genes with PAS differentially used at 5% FDR. Conserved genes are the genes with no PAS differentially used at 5% FDR. (Right) Divergent genes are the 881 genes with differences in isoform diversity between species at a 5% FDR. Conserved genes are genes without differences in isoform diversity. Genes with one PAS were not tested. (<bold>b</bold>) Proportion of tested genes with a dominant PAS in either species according to a range of cutoffs. Number of genes are reported in bars. The bars are colored by the dominance cutoff on the X axis. (<bold>c</bold>) Proportion of the number of genes with a dominant in either species that share the top used PAS according to each dominance cutoff. Number of genes with a dominant PAS in either species are reported in bars. The bars are colored by the dominance cutoff on the X axis.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Genic location of polyadenylation sites (PAS) differentially used between human and chimpanzee.</title><p>Differentially used PAS (5% FDR) between human chimpanzee by genic annotation.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Location of polyadenylation sites (PAS) within orthologous 3' UTRs.</title><p>Proportion of sites differentially used or conserved by whether they are the first (yellow), middle (blue), or last PAS (red) in orthologous exons.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp2-v2.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Genes with differentially used polyadenylation sites (PAS) are enriched for genes with apaQTL.</title><p>Ten thousand random subsamples of genes tested for differential alternative polyadenylation (APA) and overlap with genes with apaQTLs (APA quantitative trait loci) (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). Red line represents the actual overlap between genes with differential usage of at least one PAS and apaQTL genes.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp3-v2.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Information content measurement densities.</title><p>(<bold>A</bold>) Density of Shannon indices for all tested genes in human and chimpanzee <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>−</mml:mo><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>. (<bold>B</bold>) Density of Simpson indices for all tested genes in human and chimpanzee <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>−</mml:mo><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp4-v2.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Relationship between Shannon index and polyadenylation sites (PAS) number.</title><p>Shannon information index plotted against the number of PAS detect for each gene. Pearson’s correlation and significance in black.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp5-v2.tif"/></fig><fig id="fig2s6" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 6.</label><caption><title>Relationship between Simpson diversity index and polyadenylation sites (PAS) number.</title><p>Simpson diversity index plotted against the number of PAS detect for each gene. Pearson’s correlation and significance in black.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp6-v2.tif"/></fig><fig id="fig2s7" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 7.</label><caption><title>Intersection between genes with polyadenylation sites (PAS) and isoform diversity differences.</title><p>One thousand two hundred and fifty-one genes have significant differences in PAS usage between human and chimpanzee (left): 454 genes have significant differences in APA between humans and chimpanzee in PAS usage and in isoform diversity (middle) and 427 genes with differences in isoform diversity level only (right).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp7-v2.tif"/></fig><fig id="fig2s8" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 8.</label><caption><title>Polyadenylation sites (PAS) that do not lift from human to chimp.</title><p>Of the 10,077 PAS that do not reciprocally lift from human to chimp, distribution of the analysis step in which sites are filtered out. Most are lost due to not mapping to genes or due to low usage (likely noise).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig2-figsupp8-v2.tif"/></fig></fig-group><p>To better understand the mechanisms that underlie inter-species differences in PAS usage, and the potential functional impact of such differences, we considered the APA data in different contexts. First, we noticed that the spatial distribution of differentially used PAS reflects the distribution of all PAS; namely, differentially used PAS are most often located in 3' UTRs, followed by introns (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). Within the 3' UTR, however, differentially used PAS are more frequently the first ones compared with PAS that are used similarly in the two species (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>, difference in proportion test, p=0.0015). This pattern is intriguing, because changes in the usage of the first PAS in the 3' UTR may have the largest overall impact on the transcript length, and hence potentially the largest functional impact. However, it is also possible that we are more likely to detect differences in the usage in the first PAS in the 3' UTR because this site is transcribed earlier, and our estimate of usage is relative to all other sites in each gene.</p><p>We therefore sought evidence that differences in PAS usage may have functional consequences. In a previous study, we identified genetic variants associated with variation in PAS usage (apaQTLs) in a panel of 52 human LCLs (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). We found that genes with inter-species differences in PAS usage are highly enriched for apaQTLs (160, empirical p-value based on 10,000 permutations = 0.001, <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplements 3</xref> and <xref ref-type="fig" rid="fig2s1">1</xref>, 3×; hypergeometric test, p=0.0009). This observation indicates that inter-species differences in APA usage can often be found in genes whose regulation varies also at the population level, generally suggesting relaxation of evolutionary constraint on the regulation of such genes. We next considered sequence divergence at PAS by obtaining PhyloP scores for all PAS flanking regions (200 bp, as explained above). If many changes in PAS usage are genetically controlled, we would expect genomic regions of differentially used PAS to be less conserved than regions containing PAS that have similar usage. Indeed, differentially used sites are enriched for regions with negative mean PhyloP scores (1.02×, hypergeometric test, p=0.02). This observation indicates that sequence divergence is often associated with differences in PAS usage, and that the majority of PAS usage in humans and chimpanzees may be generally conserved due to evolutionary constraint.</p><p>We next asked, more specifically, if signal site changes are likely to lead to differences in PAS usage. We addressed this question by performing two analyses. First, we focused on the 82 differentially used PAS with a signal site that is annotated in only one of the species. We found that the presence of a species-specific signal site is associated with increased PAS usage, as might be expected (human enrichment 3.82×, hypergeometric p=1.37×10<sup>−10</sup>; chimpanzee enrichment 3.02×, p=3.91×10<sup>−8</sup>). Second, we considered the presence of G/U-rich elements, which are known signals to the molecular machinery for polyadenylation (<xref ref-type="bibr" rid="bib17">Colgan and Manley, 1997</xref>). Specifically, we considered the proportion of uracil bases in the PAS regions. Despite a high correlation in overall uracil content in both species (Pearson’s correlation 0.99, p&lt;2.2×10<sup>−16</sup>), the usage of PAS with greater uracil density in one species is more likely than expected by chance to be upregulated in that species (chimpanzee 1.04× enrichment, p<italic>=</italic>0.03; human 1.06× enrichment, p=0.03). Though species-specific signal sites explain a modest proportion of inter-species differences in PAS usage, these cases demonstrate the link between sequence evolution and conservation of PAS usage.</p></sec><sec id="s2-3"><title>The relationship between differences in APA and gene expression</title><p>Our analysis to this point indicates that inter-species differences in PAS usage are often genetically controlled, but generally we have not found strong evidence that they are functionally important. We explored this further by considering the APA data in the context of gene expression data that we collected from the same six human and six chimpanzee LCLs (see 'Materials and methods' for data collection procedures and low-level analysis of the RNA-seq data). We found no meaningful correlation between inter-species differences in gene expression levels and changes in polyadenylation site usage (ΔPAU) in 7462 genes for which we had both types of data (Pearson’s correlation = −0.06, p=3.1×10<sup>−7</sup>, <xref ref-type="fig" rid="fig3">Figure 3A</xref>). We then separately considered the data for the 3' UTR and intronic PAS, because we previously found a different relationship between PAS usage in these genic regions and gene expression levels (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). Indeed, we found that inter-species differences in the usage of intronic and 3' UTR PAS are loosely correlated with differences in expression effect size between the species at an equal magnitude but in opposite directions (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Increased usage of intronic sites is correlated with increased expression levels, while increased usage of 3' UTR sites is correlated with decreased expression.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Polyadenylation sites (PAS) usage differences for intronic and 3' UTR PAS correlate with differential (DE) effect sizes at similar magnitudes but in opposite directions.</title><p>(<bold>a</bold>) Changes in polyadenylation site usage (ΔPAU) for top intronic or 3' UTR PAS per gene (see 'Materials and methods') plotted against DE effect size from differential expression analysis. Pearson’s correlation plotted, Spearman’s correlation R = −0.053. (<bold>b</bold>) ΔPAU for top intronic or 3' UTR PAS per gene (see 'Materials and methods') plotted against DE effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. Pearson’s correlation plotted, Spearman’s correlation intronic R = 0.046, Spearman’s correlation 3' UTR R = −0.054. (<bold>c</bold>) ΔPAU for top intronic or 3' UTR PAS per gene (see 'Materials and methods') plotted against DE effect size from differential expression analysis. Pearson’s correlation plotted, Spearman’s correlation R = −0.16. (<bold>d</bold>) ΔPAU for top intronic or 3' UTR PAS per gene (see 'Materials and methods') plotted against DE effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. Pearson’s correlation plotted, Spearman’s correlation intronic R = 0.22, Spearman’s correlation 3' UTR R = −0.22. In all panels, we calculated the linear regression. In all panels, negative ΔPAU and DE effect sizes represent upregulation in chimpanzees. In panels <bold>b</bold> and <bold>d</bold>, we colored the points and regressions by genic location.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title><xref ref-type="fig" rid="fig3">Figure 3</xref> relationships expanded to total usage.</title><p>(<bold>A</bold>) Total mRNA changes in polyadenylation site usage (ΔPAU) for top intronic or 3' UTR polyadenylation sites (PAS) per gene plotted against differential effect size from differential expression analysis. (<bold>B</bold>) Total mRNA ΔPAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. (<bold>C</bold>) Total mRNA ΔPAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis. (<bold>D</bold>) Total mRNA ΔPAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. In all panels, we calculated the linear regression and Pearson's correlation with the r package ggpubr. In <bold>B and D</bold>, we colored the points and regression line by genic location. In all panels, negative ΔPAU and DE effect sizes represent upregulation in chimpanzees.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Differentially used 3' UTR polyadenylation sites (PAS) have higher AU element content.</title><p>(<bold>A</bold>) Genes with at least one differentially used 3' UTR PAS have more AU elements in their 3' UTRs. (<bold>B</bold>). Genes with at least one differentially used 3' UTR PAS have a higher proportion of their 3' UTRs covered by AU elements.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig3-figsupp2-v2.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title><xref ref-type="fig" rid="fig3">Figure 3</xref> without genes affected by liftover.</title><p>(<bold>A</bold>) Changes in polyadenylation site usage (∆PAU) for top intronic or 3' UTR polyadenylation sites (PAS) per gene plotted against differential effect size from differential expression analysis. (<bold>B</bold>) ∆PAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. (<bold>C</bold>) ∆PAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis. (<bold>D</bold>) ∆PAU for top intronic or 3' UTR PAS per gene plotted against differential effect size from differential expression analysis for genes with significant differences in each phenotype at 5% FDR. In all panels, we calculated the linear regression and Pearson's correlation with the r package ggpubr. In <bold>B and D</bold>, we colored the points and regression line by genic location. In all panels, negative ∆PAU and DE effect sizes represent upregulation in chimpanzees.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig3-figsupp3-v2.tif"/></fig><fig id="fig3s4" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 4.</label><caption><title>Differential expression quality control plots.</title><p>(<bold>A</bold>) First two principal components (PCs) in gene expression variation. (<bold>B</bold>) Heatmap representing correlation between technical factors and PCs. Explanation of Y axis factors and values available in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>. (<bold>C</bold>) Percent of live cells as calculated by trypan blue staining at collection is not confounded by species. (<bold>D</bold>) RIN scores reported by bioanalyzer at RNA-seq library generation are not confounded by species. (<bold>E</bold>) RNA concentrations reported by bioanalyzer at RNA-seq library generation are not confounded by species. (<bold>F</bold>) Cell concentrations at time of collection are not confounded by species.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig3-figsupp4-v2.tif"/></fig></fig-group><p>To further investigate the small but significant correlation above, we focused on 3796 genes that were classified as differentially expressed between humans and chimpanzees at 5% FDR (see 'Materials and methods'). We found that genes with at least one differentially used PAS between the species are more likely to be classified as differentially expressed than expected by chance (610 genes, 1.12× enrichment, hypergeometric test, p=3.18×10<sup>−5</sup>). Within the differentially expressed gene set, the genes with at least one differentially used PAS are enriched for RNA-processing pathways, such as RNA catabolic processes and RNA metabolic processes (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). The genes are also enriched in RNA-processing cellular compartments such as ribosomes and ribonucleoprotein complexes (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>, see 'Materials and methods'). Examining the subset of 610 genes, we observed a modest but significant negative correlation between differential expression effect size and ΔPAU when we considered all PAS (Pearson’s correlation = −0.15, p=0.0023, <xref ref-type="fig" rid="fig3">Figure 3C</xref>). Separating the analysis by PAS, genic location revealed, again, an opposite direction of the correlation between gene expression and the usage of either 3' UTR or intronic PAS (<xref ref-type="fig" rid="fig3">Figure 3D</xref>). These observations are consistent when we use PAS data based on 3' Seq data from whole cells instead of from the nuclear fractions, suggesting that the observed relationship is not due to nuclear export failure (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>, see 'Materials and methods').</p><p>To provide possible mechanistic insight into the relationship between PAS usage and gene expression, we identified AU-rich elements (AREs) in 3' UTRs in both human and chimpanzee. AREs in 3' UTRs have been linked to destabilization of mRNA transcripts and translation repression (<xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>; <xref ref-type="bibr" rid="bib55">Moore et al., 2014</xref>; <xref ref-type="bibr" rid="bib73">Siegel et al., 2020</xref>). AREs are recognized by a diverse group of binding proteins, leading to multiple models for why the pathway exists (reviewed in <xref ref-type="bibr" rid="bib3">Barreau et al., 2005</xref>). Of note, AREs and the associated binding proteins have been associated with exosome, stress granules, and P-bodies suggesting that AREs are important for response to physiological cell stress signals (<xref ref-type="bibr" rid="bib3">Barreau et al., 2005</xref>; <xref ref-type="bibr" rid="bib35">Kedersha and Anderson, 2002</xref>). We found that the 3' UTRs of genes that show an inter-species difference in 3' UTR PAS usage have a higher number (Wilcoxon test, p&lt;10<sup>−16</sup>, <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>) and density (p=5.2×10<sup>−6</sup>, <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>) of AREs compared with genes in which the 3' UTR PAS is similarly used in the two species.</p></sec><sec id="s2-4"><title>Considering overall APA diversity</title><p>We explored the relationship between inter-species differences in APA and gene expression by using a different perspective. We hypothesized that we could gain more insight into regulatory variation by summarizing the PAS diversity for a given gene using a single statistic, rather than by analyzing the usage of each site separately. To do so, we measured isoform diversity using Simpson’s D (D), a metric traditionally employed by ecologists to measure taxon diversity between environments (<xref ref-type="bibr" rid="bib56">Morris et al., 2014</xref>). In our system, higher D values indicate that the usage is spread more evenly across all PAS for a gene, while low D values suggest the one PAS is more dominant than others (see 'Materials and methods'). As expected, in both humans and chimpanzees, D values are correlated with the number of PAS per gene (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplements 4</xref>, <xref ref-type="fig" rid="fig2s5">5</xref> and <xref ref-type="fig" rid="fig2s6">6</xref>; human Pearson’s correlation 0.62, p&lt;2.2×10<sup>−16</sup>; chimpanzee Pearson’s correlation 0.63, p&lt;2.2×10<sup>−16</sup>).</p><p>Using Simpson’s D values calculated for each gene in each individual, we identified (at 5% FDR) 881 genes with significant differences in isoform diversity between species (<xref ref-type="fig" rid="fig2">Figure 2A</xref>, see 'Materials and methods'). Of these, 426 are genes for which we did not previously detect an inter-species difference in PAS usage, indicating that Simpson’s D is capturing an additional dimension of, or is more sensitive to, APA variation between species (<xref ref-type="fig" rid="fig2s7">Figure 2—figure supplement 7</xref>; for example, see <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>).</p><p>We proceeded by focusing on genes with low isoform diversity, suggesting a single dominant PAS. We calculated a dominance metric for each gene as the difference in mean usage between the first and second most used PAS (we used different cutoffs to classify dominance; see 'Materials and methods'). We found that the classification dominant PAS is highly consistent across species, a result that is quite robust with respect to the approach used to classify PAS as dominant (<xref ref-type="fig" rid="fig2">Figure 2B and C</xref>). While the dominant PAS is the same for most genes in humans and chimpanzees, differences in the usage of a dominant PAS are likely to contribute more to differential APA that have functional consequences between species than differences in other PAS. Indeed, regardless of the specific cutoff we used to define dominant PAS, when the dominant PAS is not the same in humans and chimpanzees, the corresponding genes are more likely to be differentially expressed between the species compared with genes where the dominant PAS is the same in both species, (for cutoffs between 0.2 and 0.7, all [p&lt;0.005]), and even compared with genes in which only a non-dominant PAS is differentially used (p&gt;0.8 for all cutoffs; <xref ref-type="fig" rid="fig4">Figure 4A,B</xref>).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Difference in dominant polyadenylation sites (PAS) between species likely drives differences in expression.</title><p>(<bold>a</bold>) Enrichment of genes with the different (left) of same (right) dominant PAS by dominant cutoff in differentially expressed genes. (<bold>b</bold>) −log<sub>10 </sub>(p-values) for enrichments in (a) calculated with hypergeometric tests. Horizontal line represents a p-value of 0.05. The bars are colored by the dominance cutoff on the X axis.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title><xref ref-type="fig" rid="fig4">Figure 4</xref> without genes affected by liftover.</title><p>(<bold>A</bold>) Enrichment of genes with the different (left) or same (right) dominant polyadenylation sites (PAS) by dominant cutoff in differentially expressed genes after removing genes likely affected by liftover. (<bold>B</bold>) -log<sub>10 </sub>(p-values) for enrichments in (A) calculated with hypergeometric tests. Horizontal line represents p=0.05.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig4-figsupp1-v2.tif"/></fig></fig-group><p>In a previous study that collected mRNA from a larger panel of human, chimpanzee, and rhesus macaque LCLs, Khan et al. identified genes whose regulation likely evolves under directional selection in humans and chimpanzees (<xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>). We were able to consider RNA and protein expression data as well as APA data from 2532 genes. We found that 22 of the genes with significant inter-species differences in APA at both the site level and in isoform diversity are among those whose regulation likely evolves under directional selection in the chimpanzee lineage, a 1.6× enrichment over what is expected by chance (hypergeometric test, p=0.015). While we did not identify any significant gene ontology (GO) categories for these genes, 5 of the 22 genes are associated with protein transport (CAPZA1, TPM3, TMED2, GOSR2, AP3S1) and 2 are ribosomal subunits (RPL13 and RPL7L1). We did not find a similar enrichment when we considered genes whose regulation evolved under selection in humans, but the sample size is rather small.</p></sec><sec id="s2-5"><title>Variation in APA and differences in protein expression</title><p>Given the well-characterized molecular connection between APA and the regulation of protein translation, we hypothesized that genes with inter-species differences in APA are also more likely to be differentially translated between the species (<xref ref-type="bibr" rid="bib20">Di Giammartino et al., 2011</xref>; <xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>; <xref ref-type="bibr" rid="bib76">Tian and Manley, 2017</xref>). To examine this, we obtained estimates of protein translation based on ribosome profiling data that were collected from human and chimpanzee LCLs by <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>. At a 5% FDR, Wang et al. identified 1993 differentially translated genes between humans and chimpanzees. Genes with significant inter-species differences in isoform diversity, but without significant differences in the usage at individual PAS, are enriched among the differentially translated gene set (1.21×, hypergeometric test, p=0.011; <xref ref-type="fig" rid="fig5">Figure 5A,B</xref>). The genes with differentially used PAS are 32× enriched within the differentially translated genes for genes involved in translation initiation (hypergeometric test, FDR 0.0091, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>, see 'Materials and methods').</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Polyadenylation sites (PAS)-level differences in alternative polyadenylation (APA) may drive differences in expression while isoform diversity differences likely drive translation differences.</title><p>(<bold>a</bold>) Enrichment of genes with isoform-level differences (ID), differences in APA at PAS level (PAS), or at both levels (Both) within differential expressed genes and differentially translated genes. Differentially translated genes reported by <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>. (<bold>b</bold>) −log<sub>10 </sub>(p-values) for enrichments in <bold>A</bold> calculated with hypergeometric tests. Horizontal line represents a p-value of 0.05.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Gene with significant differences in isoform diversity only.</title><p>Human and chimpanzee usage for five polyadenylation sites (PAS) identified in the IVNS1ABP gene. None of the PAS measured have significant differences in the usage at 5% FDR. IVNS1ABP is not differentially expressed.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig5-figsupp1-v2.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Relationship between changes in polyadenylation site usage (∆PAU) and differential translation (TE) effect sizes.</title><p>(<bold>A</bold>) ΔPAU for top 3' UTR and intronic polyadenylation sites (PAS) plotted against differential TE effect size as reported by Wang et al. ∆PAU for top 3' UTR and intronic PAS plotted against TE effect size as reported by Wang et al. separated by genic location. (<bold>C</bold>) ∆PAU for top 3' UTR and intronic PAS with significant differences in the usage plotted against TE effect size for significant genes (5% FWER) as reported by Wang et al. (<bold>D</bold>) ∆PAU for top 3' UTR and intronic PAS with significant differences in the usage plotted against TE effect size for significant genes (5% FWER) as reported by Wang et al. separated by genic location. Linear regression line was plotted and Pearson's correlation was calculated for data in each panel.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig5-figsupp2-v2.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>Relationship between alternative polyadenylation (APA) differences and protein decay mark.</title><p>(<bold>A</bold>) Absolute value of changes in polyadenylation site usage (∆PAU) for 3' UTR polyadenylation sites (PAS) with significant difference at site level plotted against the number of ubiquitination marks in the gene standardized by the number of amino acids. Regression line and Pearson's correlation are plotted in red. (<bold>B</bold>) Absolute value of ∆PAU for intronic PAS with significant difference at site level plotted against the number of ubiquitination marks in the gene standardized by the number of amino acids. Regression line and Pearson's correlation are plotted in blue.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig5-figsupp3-v2.tif"/></fig><fig id="fig5s4" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 4.</label><caption><title><xref ref-type="fig" rid="fig5">Figure 5</xref> without genes affected by liftover.</title><p>(<bold>A</bold>) Enrichment of genes with differences in isoform diversity, polyadenylation sites (PAS) usage, or both within differentially expressed genes and differentially translated genes after removing genes likely affected by liftover. Differentially translated genes reported by Wang et al. (<bold>B</bold>) −log<sub>10 </sub>(p-values) for enrichments in <bold>A</bold> calculated with hypergeometric tests. Horizontal line represents p=0.05.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig5-figsupp4-v2.tif"/></fig></fig-group><p>We next investigated the relationship between ΔPAU in humans and chimpanzees and the effect sizes for differences in protein translation between the species (<xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>). Considering the most differentially used 3' UTR or intronic PAS per gene (see 'Materials and methods'), we identified a significant correlation between inter-species differences in translation and ΔPAU for 3' UTR PAS, with a stronger correlation among genes with significant differences in both APA and translation (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). As expected, and to some extent we view this as a control analysis, we did not identify a significant correlation between intronic PAS ΔPAU values and differences in translation (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>).</p><p>Given the apparent impact of PAS usage on protein translation, we next considered direct measurements of protein expression data from 3391 genes in LCLs from humans and chimpanzees (<xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>). Using summary statistics from this study, we found 1263 genes to be differentially expressed at the protein level between the species (FDR of 5%). As the protein measurements are restricted to these 3391 genes, we do not have enough power to ask if genes with inter-species differences in APA are also more likely to be differentially expressed at the protein level. However, we did find a positive correlation between the absolute value of 3' UTR ΔPAU and the standardized number of ubiquitination sites for the same gene (Pearson’s correlation, R = 0.15, p=5.0×10<sup>−7</sup>, <xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>, see 'Materials and methods'), consistent with the observation that 3' UTR PAS are targets for the regulation of protein decay (<xref ref-type="bibr" rid="bib23">Dubnikov et al., 2017</xref>; <xref ref-type="bibr" rid="bib67">Ravid and Hochstrasser, 2008</xref>). Thus, we next focused on the 506 genes with significant inter-species differences in protein expression and an absence of corresponding differences in transcript expression levels that we also tested for differences in APA. Khan et al. reasonably hypothesized that inter-species differences in translation could account for the emergence of differences in protein expression levels when there are no regulatory differences at the RNA level, but they were unable to point to specific mechanisms. These genes are particularly interesting in the context of our current study, because APA which results in changes to 3' UTR length may be more likely to result in differences in protein expression without affecting the expression level of the mRNA.</p><p>Indeed, we found 76 genes with inter-species differences in APA that are also differentially expressed at the protein but not at the RNA level between humans and chimpanzees (<xref ref-type="fig" rid="fig6">Figure 6A,B</xref>). In these 76 genes, inter-species differences in PAS usage are enriched at the 3' UTR (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>). The 76 genes are likely of functional relevance, compared to all of the genes, with at least one differentially used PAS being enriched for cellular components such as the protease complex and endopeptidase complex. The set is also enriched for processes such as the regulation of DNA-templated transcription and amino acid activation. (For a full list, see <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>, see 'Materials and methods'.) Finally, to assess whether APA contributes to differences in gene regulation by affecting translation efficiency or protein degradation, we asked whether genes with differential protein expression were also differentially translated. Of the 149 genes with significant differences in APA and protein expression, Wang et al. reported translation measurements for 142 (<xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>). Only 34 genes displayed significant differences in translation efficiency, suggesting that isoform-specific post-translational modification of protein levels is largely responsible for protein-level differences (<xref ref-type="fig" rid="fig6">Figure 6C,D</xref>).</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Alternative polyadenylation (APA) differences explain genes differentially expressed at protein level but not in mRNA.</title><p>APA likely mediates functional differences post-translationally. (<bold>a</bold>) Number of genes with isoform-level differences (ID), differences in APA at polyadenylation sites (PAS) level (PAS), or at both levels (Both) differentially expressed in protein (5% FDR) but not mRNA (5% FDR). Genes with differentially expressed protein reported in <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>. (<bold>b</bold>) Proportion of genes with differential usage at PAS level (1251 genes), isoform diversity level (426 genes), or both (454 genes) differentially expressed in protein (5% FDR) but not mRNA (5% FDR). (<bold>c</bold>) Genes reported in (a) separated by genes differentially translated at 5% FDR. Differentially translated genes reported in <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>. (<bold>d</bold>) Genes differentially expressed in protein but not in mRNA, colored by differences in APA. Proportion of genes in the set differentially translated at 5% FDR.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig6-v2.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Enrichment of 3' UTR polyadenylation sites (PAS) in genes differentially expressed in protein but not in mRNA.</title><p>Genic location enrichments for the PAS in genes differentially expressed at protein level but not mRNA level among all differentially used PAS. p-values were calculated with a hypergeometric test.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig6-figsupp1-v2.tif"/></fig><fig id="fig6s2" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 2.</label><caption><title><xref ref-type="fig" rid="fig6">Figure 6</xref> without genes affected by liftover.</title><p>(<bold>A</bold>) Number of genes with differences in isoform diversity, polyadenylation sites (PAS) usage, or both differentially expressed in protein (5% FDR) but not in mRNA (5% FDR). Genes differentially expressed in protein from Khan et al. (<bold>B</bold>) Proportion of genes with differential isoform diversity, PAS usage, or both that are differentially expressed in protein (5% FDR) but not mRNA (5% FDR). (<bold>C</bold>) Genes reported in (A), separated by genes differentially translated at 5% FDR. Differentially translated gene reported in Wang et al. (<bold>D</bold>) Genes differentially expressed in protein but not in mRNA, colored by differences in APA. Proportion of genes in the set differentially translated at 5% FDR.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-fig6-figsupp2-v2.tif"/></fig></fig-group></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Comparative primate functional genomic studies have contributed to our understanding of the gene regulatory processes that underlie genotype-phenotype relationships. A common goal of these studies is to understand the general properties and level of conservation of specific regulatory phenotypes, such as gene expression or DNA methylation levels. Multiple data types collected from the same cell lines or tissues can then be analyzed together to generate hypotheses about how gene regulatory processes contribute to inter-species differences in morphology, physiology, cognitive phenotypes, disease susceptibility, and other traits (<xref ref-type="bibr" rid="bib6">Blake et al., 2020</xref>; <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib69">Romero et al., 2018</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>; <xref ref-type="bibr" rid="bib85">Zhou et al., 2014</xref>). Moreover, unlike functional genomic studies within humans, comparative studies require only modest sample sizes to identify regulatory effects. This is because genetic variation between species is greater than genetic variation within species. Thus, regulatory differences that distinguish humans from other primates tend to have larger effect sizes than regulatory differences that distinguish between individual humans (<xref ref-type="bibr" rid="bib32">Housman and Gilad, 2020</xref>).</p><p>Importantly, inter-species differences identified using a comparative approach may be important not only for understanding primate evolution but may implicate candidate loci for further investigation in humans. For example, genomic regions that are conserved in primates may point to loci that are likely to have negative functional consequences in humans, potentially with effects on disease risk (<xref ref-type="bibr" rid="bib31">Housman et al., 2019</xref>). Identification of genomic regions under adaptation in humans is also critical, as they may point to causal mechanisms for human-specific traits, including diseases that are specific to humans (<xref ref-type="bibr" rid="bib28">Gokhman et al., 2020</xref>; <xref ref-type="bibr" rid="bib82">Ward and Gilad, 2019</xref>).</p><p>We characterized APA in human and chimpanzee LCLs to begin to understand the role that co-transcriptional mechanisms play in the evolution of gene regulation in primates. Our group has previously studied a variety of other gene regulatory phenotypes in primate LCLs (<xref ref-type="bibr" rid="bib11">Cain et al., 2011</xref>; <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>; <xref ref-type="bibr" rid="bib85">Zhou et al., 2014</xref>). We and others have demonstrated that gene regulatory phenotypes in these cell lines recapitulate many regulatory patterns seen in primary tissues (<xref ref-type="bibr" rid="bib13">Caliskan et al., 2011</xref>; <xref ref-type="bibr" rid="bib14">Çalışkan et al., 2014</xref>; <xref ref-type="bibr" rid="bib38">Khaitovich et al., 2006</xref>). Not only did our use of primate LCLs allow us to circumvent many of the practical and ethical issues associated with primate research, but it also allowed us to integrate 3' Seq and gene expression data from this study in the context of previously collected ribosome occupancy and protein expression data from primate LCLs (<xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>). Together, these data allowed us to study the contribution of APA to inter-species differences in transcript and protein expression levels. We recognize the limitations of this study with respect to physiologically interesting phenotypes. The genome-wide map of APA events in human and chimpanzee has allowed us to infer global mechanisms connecting APA to other gene regulatory processes. By expanding the study of primate APA into other cell types and dynamic processes, future studies will be able to connect the mechanisms described here to primate phenotypes of interest.</p><p>APA is an important molecular mechanism with regard to both the evolution of gene regulation and physiological traits. On a long-term evolutionary scale, both 3' UTR length and the proportion of genes exhibiting APA have scaled with genome size and complexity. The expansion of APA is believed to have introduced biological complexity independent of an increase in the number of distinct genes (<xref ref-type="bibr" rid="bib48">Mayr, 2016</xref>; <xref ref-type="bibr" rid="bib49">Mayr, 2017</xref>). As usage of multiple isoforms has been maintained, it is likely that distinct isoforms have divergent functions that are maintained by balancing selection. For example, APA facilitates post-transcriptional regulation of a <italic>Drosophila Hox</italic> gene through maintenance of two isoforms differentially targeted by multiple miRNAs (<xref ref-type="bibr" rid="bib61">Patraquim et al., 2011</xref>). In turn, genome-wide changes in APA during differentiation of stem cells to terminal cell types direct isoform-specific gene regulation that is important for development in a range of species, including humans (<xref ref-type="bibr" rid="bib29">Hilgers et al., 2011</xref>; <xref ref-type="bibr" rid="bib34">Ji et al., 2009</xref>; <xref ref-type="bibr" rid="bib42">Li et al., 2012</xref>). Further, dysregulation of tumor suppressor genes through intronic polyadenylation is known to contribute to cancer pathogenesis (<xref ref-type="bibr" rid="bib22">Dubbury et al., 2018</xref>; <xref ref-type="bibr" rid="bib41">Lee et al., 2018</xref>). We hypothesize that a better understanding of APA in primates will aid in the understanding of APA evolution and its contribution to human-specific phenotypes.</p><sec id="s3-1"><title>APA is mostly conserved, especially dominant sites</title><p>We measured APA from 3' Seq data by calculating a ratio of isoforms terminating at one PAS compared to isoforms terminating at other PAS for the same gene. We then compared PAS usage ratios between species. To expand our understanding of APA conservation, we also calculated an isoform diversity statistic (Simpson’s D) for each gene in each species. Because Simpson’s D captures both the number of PAS isoforms and their usage, we were able to evaluate small regulatory changes spread across many PAS, rather than only focus on large changes at individual PAS. While previous studies have used Shannon index to quantify isoform diversity (<xref ref-type="bibr" rid="bib59">Pai et al., 2016</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>), we found Simpson’s D to be less correlated with PAS number, making it less sensitive to the number of PAS per gene (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplements 5</xref> and <xref ref-type="fig" rid="fig2s6">6</xref>). In addition, by placing more weight on dominant PAS, Simpson’s D more closely mirrors our current biological understanding of APA, wherein dominant PAS play a larger role in downstream gene regulation (<xref ref-type="bibr" rid="bib56">Morris et al., 2014</xref>).</p><p>In general, we found that both individual PAS usage and isoform diversity are highly conserved between human and chimpanzee. Consistent with comparative studies of APA in humans and rodents, which used genomic synteny to identify conserved PAS, we found higher conservation among genes with a single PAS (<xref ref-type="bibr" rid="bib1">Ara et al., 2006</xref>; <xref ref-type="bibr" rid="bib80">Wang et al., 2018a</xref>) and showed that sequence variation in PAS signal sites and the surrounding U-rich regions contributes to inter-species differences in APA (<xref ref-type="bibr" rid="bib80">Wang et al., 2018a</xref>). Because we characterized APA in closely related primates, our study provides additional insight into APA divergence at both the gene level and the species level, revealing functional changes that contribute to differences in downstream gene regulation. For example, we observed that when genes use one PAS markedly more often than others, said dominant PAS tends to be the same in both human and chimpanzee. It is likely that strong selection pressures have acted on these genes, resulting in continual usage of the same dominant isoform. This could imply that the dominant isoform is functionally important and alternative isoforms are potentially associated with reduced fitness. However, non-dominant isoforms also show evidence of conservation. Thus, it remains possible that there is a threshold at which the level of expression of the alternative isoforms begins to impede gene function.</p></sec><sec id="s3-2"><title>APA is associated with gene expression divergence</title><p>Our study also revealed that the majority of differentially used PAS between species are located in 3' UTRs. We showed that, across species, increased intronic PAS usage is associated with a modest increased mRNA expression levels, while increased 3' UTR PAS usage is correlated with a modest decrease in mRNA expression. In a previous study, we found that human alternative polyadenylation quantitative trait loci (apaQTL) alleles associated with increased intronic PAS usage were correlated with <italic>decreased</italic> mRNA expression levels (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). This is not the first molecular phenotype wherein a within-species study revealed alternative regulatory models compared to an inter-species analysis. For example, Pai et al. reported tissue-specific differential methylation to be almost exclusively inversely correlated with gene expression patterns between human and chimpanzee (<xref ref-type="bibr" rid="bib24">Enard et al., 2004</xref>; <xref ref-type="bibr" rid="bib58">Pai et al., 2011</xref>; <xref ref-type="bibr" rid="bib83">Weber et al., 2007</xref>), whereas Banovich et al. discovered genetic variation associated with DNA methylation variation that was both directly and inversely correlated with expression quantitative trait loci (eQTLs). At this time, we cannot provide evidence for a mechanistic explanation for these contradictory observations. We hypothesize the following: Transcripts terminating in introns are likely subject to nonstop decay (NSD). By studying APA variation within humans, we probably captured the effects of intronic termination. Across species, however, increased intronic PAS may simply track the overall expression level of the gene. Specifically, increased usage of intronic PAS may result in truncated isoforms that do not contain 3' UTR <italic>cis</italic> regulatory elements that would normally signal mRNA decay. If these truncated isoforms are no longer targets of mRNA decay, this could cause some of these genes to appear upregulated. We hypothesize that the within-species effects related to NSD were likely overshadowed by inter-species differences, which typically have much larger effect sizes than differences observed within a population (<xref ref-type="bibr" rid="bib32">Housman and Gilad, 2020</xref>).</p><p>Alternatively, the large effect sizes for differential usage of 3' UTR PAS could also be driving the relationship between differential APA and differential expression. In line with this hypothesis, we found increases in AU destabilizing elements and ubiquitination marks for genes with divergent 3' UTR PAS. However, since PAS usage is calculated as a ratio, we may have detected changes in intronic PAS usage solely as a mathematical consequence of changes in 3' UTR PAS. Functional follow-up on the genes with PAS detected as differentially used between and within species would be necessary to explore the relative importance of each of these regulatory pathways and to disentangle the results from both studies.</p><p>In past studies, we and others have estimated the proportion of variation in gene expression explained by different regulatory mechanism. For example, we have previously tested for differences in expression before and after accounting for another regulatory mechanism or used formal mediation analyses (<xref ref-type="bibr" rid="bib6">Blake et al., 2020</xref>; <xref ref-type="bibr" rid="bib8">Blekhman et al., 2009</xref>; <xref ref-type="bibr" rid="bib11">Cain et al., 2011</xref>; <xref ref-type="bibr" rid="bib25">Eres et al., 2019</xref>). We would have liked to perform similar analysis in the current study, regarding the role that APA plays in the overall regulatory divergence between the species. However, APA was measured using ratios of alternative mRNA isoforms and thus, effect sizes for APA and differential expression are on different scales and we cannot use a standard mediation approach to formally calculate the proportion of expression variation explained by APA. That said, we are generally convinced that APA contributes to differences in mRNA expression overall because 764 of 3796 (20.1%) differentially expressed genes also have significant differences in APA.</p></sec><sec id="s3-3"><title>Inter-species differences in APA explain protein-specific regulatory divergence</title><p>Though genes are ultimately expressed as proteins, many studies (including current studies by our group) still measure mRNA expression as an implicit proxy for protein expression levels. As a justification for this approach, we typically point to the fact that after accounting for technical considerations, the correlation between mRNA levels and protein abundance is quite high genome-wide, specifically, across genes (<xref ref-type="bibr" rid="bib10">Buccitelli and Selbach, 2020</xref>; <xref ref-type="bibr" rid="bib18">Csárdi et al., 2015</xref>). However, we also know that at the level of a single gene, across individuals or tissues, the correlation between mRNA and protein measurements tends to be much lower (<xref ref-type="bibr" rid="bib4">Battle et al., 2015</xref>; <xref ref-type="bibr" rid="bib10">Buccitelli and Selbach, 2020</xref>). This suggests that a number of molecular mechanisms decouple mRNA and protein expression levels post-transcriptionally. Clearly, we do not yet fully understand the post-transcriptional and translational mechanisms that shape the proteome (<xref ref-type="bibr" rid="bib10">Buccitelli and Selbach, 2020</xref>).</p><p>Within human populations and between primates, there is a large number of genes that are differentially expressed at the mRNA level but not as proteins. By directly measuring translation levels for these genes, previous studies have proposed that post-translational protein buffering can explain the decreased variation at the protein level (<xref ref-type="bibr" rid="bib4">Battle et al., 2015</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>). Conversely, there are also genes that are more variable at the protein levels than at the mRNA levels (<xref ref-type="bibr" rid="bib4">Battle et al., 2015</xref>; <xref ref-type="bibr" rid="bib15">Chick et al., 2016</xref>; <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>). Our previous work demonstrated that some protein-specific QTLs are also highly correlated with differences in APA (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). Considered all of these observations, we expanded this analysis and demonstrated that genes that are differentially expressed as proteins, but not at the mRNA level, between human and chimpanzee, tend to have divergent APA patterns. We also found that the divergent protein levels are likely due to post-translational molecular mechanisms. While we cannot directly test the mechanism here, we hypothesize that APA could lead to variation in protein levels as a consequence of protein autoregulation, by differentially including RNA and protein-binding motifs (<xref ref-type="bibr" rid="bib10">Buccitelli and Selbach, 2020</xref>; <xref ref-type="bibr" rid="bib19">de Bie and Ciechanover, 2011</xref>; <xref ref-type="bibr" rid="bib57">Müller-McNicoll et al., 2019</xref>). Alternatively, APA could contribute to temporal and spatial differences in protein expression, which would affect our ability to quantify protein with traditional techniques (<xref ref-type="bibr" rid="bib10">Buccitelli and Selbach, 2020</xref>; <xref ref-type="bibr" rid="bib76">Tian and Manley, 2017</xref>).</p><p>In conclusion, a better understanding of co-transcriptional gene regulatory mechanisms, such as APA, may point to additional mechanisms contributing to the decoupling of mRNA and protein abundance and more generally, enhance our understanding of how variation percolates through genetic variants, mRNA, and protein to ultimately affect human phenotypic diversity.</p></sec></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Cell culture and collections</title><p>We grew six human and six chimpanzee Epstein-Barr virus-transformed LCLs in glutamine-depleted RPMI (RPMI 1640 1× from Corning [15–040 CM]), completed with 20% FBS, 2 mM GlutaMax (Gibco [35050–061]), 100 IU/mL penicillin, and 100 µg/mL streptomycin. We cultured all cells at 37°C at 5% CO<sub>2</sub>. We passaged each cell line a minimum of three times, then maintained cells at 1 × 10<sup>6</sup> cells/mL in preparation for collection. Cell line numbers and details can be found in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>. The human lines were derived from Yoruba individuals collected as part of the HapMap project and can be ordered through the Coriell Institute (<xref ref-type="bibr" rid="bib33">International HapMap Consortium, 2005</xref>). The sample IDs and Research Resource Identifiers for the human cell lines are the following: NA18498: CVCL_P466, NA18499: CVCL_P457, NA18502: CVCL_P459, NA18504: CVCL_P460, NA18510: CVCL_P461, NA18523: CVCL_P468. Chimpanzee LCLs were originally transformed from individuals from the New Iberia Research Center (University of Louisiana at Lafayette), Coriell IPBR repository and Arizona State University (<xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>). The cell lines have previously been used for similar studies of primate gene regulation (<xref ref-type="bibr" rid="bib11">Cain et al., 2011</xref>; <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>; <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>; <xref ref-type="bibr" rid="bib85">Zhou et al., 2014</xref>). Cell lines have been authenticated with RNA-seq and have tested negative for mycoplasm.</p><p>Once all cells lines reached 1 × 10<sup>6</sup> cells/mL, we used the collection and RNA extraction method detailed in <xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref> . to extract whole-cell and nuclear mRNA. Briefly, we collected 30 million cells in two 15 million cell aliquots. We extracted nuclei from one aliquot per line using the nuclear isolation protocol outlined by <xref ref-type="bibr" rid="bib47">Mayer and Churchman, 2016</xref>. We extracted mRNA in two fraction- and species-matched batches, using the miRNeasy kit (Qiagen) according to manufacturer's instructions, including the DNase step to remove genomic DNA. We quantified mRNA and tested quality using a nanodrop. Details of mRNA processing for each line, including concentrations and quality, can be found in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>.</p></sec><sec id="s4-2"><title>3' Seq to identify PAS and quantify site usage</title><p>We generated 3' Seq libraries from whole-cell and nuclear-isolated mRNA from six chimpanzee and six human individuals using the QuantSeq Rev 3' mRNA-Seq Library Prep Kit (<xref ref-type="bibr" rid="bib54">Moll et al., 2014</xref>) according to the manufacturer’s instructions. We sequenced all libraries on the Illumina NextSeq500 at the University of Chicago Genomics Core facility using single-end 50 bp sequencing.</p><p>We mapped human 3' Seq libraries to GRCh38 (<xref ref-type="bibr" rid="bib71">Schneider et al., 2017</xref>) and chimpanzee libraries to panTro6 (<xref ref-type="bibr" rid="bib16">Chimpanzee Sequencing and Analysis Consortium, 2005</xref>) using the STAR RNA-seq aligner with default settings (<xref ref-type="bibr" rid="bib21">Dobin et al., 2013</xref>). Similar to our previous work, we removed reads with evidence of internal priming resulting from the poly(dT) primer. We filtered reads proceeded by 6 As or 7 of 10 As in the base pairs directly upstream of the mapped location (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>; <xref ref-type="bibr" rid="bib72">Sheppard et al., 2013</xref>; <xref ref-type="bibr" rid="bib75">Tian et al., 2005</xref>). To ensure that differences in low quality bases would not bias our results, we treated any N in the genome annotation as an A. All raw read counts, mapped read counts, and filtered read counts can be found in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>.</p><p>We first identified an inclusive set of PAS in each species separately. We used the same in-house peak caller described in <xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>, annotating each PAS as the most 3' base in each peak. The initial PAS set included 340,023 in human and 303,249 in chimps. We extended PAS 100 bp upstream and 100 bp downstream and used a reciprocal liftover pipeline to identify an inclusive set of orthologous PAS. We downloaded chain files from UCSC Genome Browser (<xref ref-type="bibr" rid="bib37">Kent et al., 2002</xref>). Details of the pipeline and number of PAS passing each step can be found in <xref ref-type="fig" rid="fig1s12">Figure 1—figure supplement 12</xref>.</p><p>Due to gene annotation differences between species, we annotated all orthologous PAS to the human NCBI RefSeq annotation downloaded from UCSC Genome Browser (<xref ref-type="bibr" rid="bib64">Pruitt et al., 2004</xref>). We used a hierarchical model to assign PAS to genic locations (<xref ref-type="bibr" rid="bib45">Lin et al., 2012</xref>; <xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). We prioritized annotations in the following order: 3' UTRs (UTR5), 5 kb downstream of genes (end), exons (cds), 5' UTRs (UTR5), and introns (intron). We quantified reads mapping to each annotated PAS for each individual in both the total RNA libraries and nuclear RNA libraries using featureCounts with the -s strand specificity flag (<xref ref-type="bibr" rid="bib44">Liao et al., 2014</xref>). We calculated usage for each PAS in each library as a ratio of reads mapping to the PAS divided by the number of reads mapping to any PAS in the same gene (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>). We implemented two filtering steps to remove PAS with ratios likely biased by low site count or low gene count separately in each fraction.</p><p>Next, we filtered out sites with less than 5% usage in both species in the nuclear fraction. We then merged nuclear counts across all PAS in each gene. We removed PAS in genes not passing a cutoff of log<sub>2</sub>(CPM) &gt;2 in at least 8 of the 12 individuals. After applying these filters, we were left with 44,432 PAS. As a quality control metric, we compared PAS usage calculated from the nuclear fraction to PAS usage calculated from whole-cell fraction for each individual (we used the same methods to identify and quantify PAS usage in the whole-cell 3' Seq data). We expected a high correlation between PAS usage in each fraction. Further, we expected a similar correlation in human and chimpanzee individuals (<xref ref-type="bibr" rid="bib51">Mittleman et al., 2020</xref>). Human individual NA18499 had significantly lower across-species correlation than the other individuals and was therefore removed from the analysis (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>).</p><p>To ensure gene expression level did not introduce ascertainment bias, we tested the relationship between PAS number and normalized gene expression. In both species, the number of PAS is negatively correlated with normalized gene expression (human: Pearson’s correlation = −0.19, p&lt;2.2×10<sup>−16</sup>; chimpanzee: Pearson’s correlation = −0.17, p&lt;2.2×10<sup>−16</sup>; <xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>). We expected species to contribute the most amount of variation to PAS usage. We ran principal component analysis (PCA) on the filtered nuclear PAS usage. PC1 accounts for 41.8% of the variation and is highly correlated with species (R<sup>2</sup> = 0.68). PC2 accounts for 13.1% of the variation and is moderately correlated with RNA extraction technician (R<sup>2</sup> = 0.38) and extraction day (R<sup>2</sup> = 0.28). As both of these variables are balanced with respect to species, we do not believe they bias the results (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>). We identified 302 sites used at a rate of 5% or greater in humans and 0% in chimpanzees, which we designated as human-specific. We identified 357 sites used at a rate of 5% or greater in chimpanzee and 0% in humans, which we designated as chimp-specific.</p><p>We acknowledge the possibility that unlifted PAS may affect the downstream analyses; therefore, we removed genes for which PAS ratios may be affected. Specifically, we annotated and calculated usage for the human PAS, including the 10,077 PAS that do not reciprocally lift to the chimpanzee genome. After removing PAS in genes previously identified as lowly expressed and PAS with usage below 5%, 386 PAS in 353 genes remain (<xref ref-type="fig" rid="fig2s8">Figure 2—figure supplement 8</xref>). We removed these 353 genes and recreated main figures 3-6. (<xref ref-type="fig" rid="fig3">Figures 3</xref>–<xref ref-type="fig" rid="fig6">6</xref>, <xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, <xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4</xref>, <xref ref-type="fig" rid="fig6s2">Figure 6—figure supplement 2</xref>).</p></sec><sec id="s4-3"><title>Orthologous 3' UTRs</title><p>We identified a set of orthologous UTRs using the orthologous exon file described in the differential expression analysis section of the 'Materials and methods'. We merged all regions annotated as 3' UTR by gene. If a gene had multiple non-continuous annotations, we selected the most 3' region as the orthologous UTR. We used deepTools compute matrix and plotHeatmap functions to plot merged human and chimpanzee reads along the orthologous 3' UTR set (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>; <xref ref-type="bibr" rid="bib66">Ramírez et al., 2016</xref>). For all genes with PAS only in 3' UTRs, we assigned PAS to single, first, middle, and last, as previously described (<xref ref-type="bibr" rid="bib80">Wang et al., 2018a</xref>).</p></sec><sec id="s4-4"><title>Analysis of sequence conservation around PAS</title><p>We used PhyloP scores to measure sequence-level conservation. We downloaded the hg38 100-way vertebrate PhyloP bigwig file from the UCSC table browser (<xref ref-type="bibr" rid="bib63">Pollard et al., 2010</xref>). We computed scores for PAS regions as well as 200 bp intervals by taking the mean of the base pair scores. We removed any region with missing data from the analysis. We tested for differences in mean phloP scores using Wilcoxon rank sum tests.</p><p>We tested for the presence of the polyadenylation signal site motif in the 200 bp PAS regions. We used the bedtools nuc tool with the strand-specific flag to test for the presence of each of the 12 previously annotated motifs for each PAS in both species (<xref ref-type="bibr" rid="bib5">Beaudoing et al., 2000</xref>; <xref ref-type="bibr" rid="bib65">Quinlan and Hall, 2010</xref>). If a PAS had multiple motifs, we used a hierarchical model to choose the site based on the number of PAS with each identified motif (order: AATAAA, ATTAAA, AAAAAG, AAAAAA, TATAAA, AATATA, AGTAAA, AATACA, GATAAA, AATAGA, CATAAA, ACTAAA). The proportion of PAS with each signal site motif matched across species (<xref ref-type="fig" rid="fig2">Figure 2</xref>). To ask if the presence or absence of a signal site explained species specificity or site-level differences, we restricted our analysis to the top two signal sites. These two motifs are the only sites where the presence of a signal is associated with increased usage of the site in both species (<xref ref-type="fig" rid="fig1s10">Figure 1—figure supplement 10</xref>). For the 359 PAS with one of these two signal sites present only in chimpanzees, average usage was higher in chimpanzees than in humans (p=0.025). For the 361 PAS with one of these two signal sites present only in humans, average usage was higher in humans (p=2.0×10<sup>−4</sup>). We used hypergeometric tests to evaluate enrichment of differentially used PAS and species-specific PAS in the set of PAS with signal sites in only one species.</p><p>We also examined the proportion of U nucleotides in each PAS region. We used the bedtools nuc with the -s flag for strand specificity (<xref ref-type="bibr" rid="bib65">Quinlan and Hall, 2010</xref>). We tested if PAS with differences in U content are enriched for differentially used PAS using a hypergeometric test.</p></sec><sec id="s4-5"><title>Differential APA</title><sec id="s4-5-1"><title>PAS-level differences</title><p>We quantified reads mapping to each PAS using the featureCounts tool with the -s strand specificity tool (<xref ref-type="bibr" rid="bib44">Liao et al., 2014</xref>). We tested for site-level differences between human and chimpanzee using the leafcutter leafcutter_ds.R tool with standard settings (<xref ref-type="bibr" rid="bib43">Li et al., 2018</xref>). We tested for differences in both the total and nuclear fractions. We tested 43,038 PAS in 8422 genes in the nuclear fraction and 41,914 PAS in 8333 genes in the total fraction. We classified PAS as differentially used if the gene reached significance at 5% FDR and the PAS had a ΔPAU greater than 20% (absolute value [ΔPAU]&gt;0.2). A negative ΔPAU indicates increased usage in chimpanzees and ΔPAU indicates increased usage in humans. The top PAS per gene is the PAS with the most significant difference between species; ties were broken using mean usage for all individuals in both species.</p></sec><sec id="s4-5-2"><title>Isoform diversity differences</title><p>We calculated Shannon information content (<inline-formula><mml:math id="inf3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>−</mml:mo><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>) and Simpson index (<inline-formula><mml:math id="inf4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>) using mean usage of each PAS in humans and chimpanzees, where <inline-formula><mml:math id="inf5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mstyle></mml:math></inline-formula> is the usage of the ith of <italic>s</italic> sites in the gene. We used Simpson index to assess isoform diversity because the correlation between Simpson index and number of PAS is lower than the correlation between Shannon information content and the number of PAS per gene (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplements 5</xref>, <xref ref-type="fig" rid="fig2s6">6</xref>). To identify genes with differences in isoform diversity, we recalculated Simpson index per gene per individual and tested for differences between species with Wilcoxon tests. We reported genes with differences at 5% FDR.</p></sec><sec id="s4-5-3"><title>Conservation of dominant PAS</title><p>We consider a gene to have a dominant PAS if the within species average usage of the top used PAS is greater than the second most used site by 0.4. We reported results for cutoffs between 0.1 and 0.9. If a gene had a dominant PAS in either species, we included the top used site for both species when testing if genes use the same or different dominant PAS between species. We tested for enrichment of genes using the same or different dominant PAS with differentially expressed genes using hypergeometric tests.</p></sec></sec><sec id="s4-6"><title>Differential expression analysis</title><p>We generated unstranded RNA-seq libraries using the Illumina TruSeq Total RNA kit according to the manufacturer’s instructions using the total mRNA collected from all 12 individuals (Illumina, San Diego, CA, USA). We sequenced RNA-seq libraries at the University of Chicago Genomics Core facility using the single-end 50 bp protocol on one lane of the Illumina HiSeq 4000 machine. RNA quality and concentration at the time of library prep and number of sequenced reads per library are available in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>. We mapped the human libraries to GRhg38 (<xref ref-type="bibr" rid="bib71">Schneider et al., 2017</xref>) and chimpanzee libraries to panTro6 (<xref ref-type="bibr" rid="bib16">Chimpanzee Sequencing and Analysis Consortium, 2005</xref>) and quantified reads mapping to orthologous exons.</p><p>To generate an updated orthologous exon file for the most recent chimpanzee genome assembly (panTro6), we followed the procedure reported in Pavlovic et al. with slight modifications (<xref ref-type="bibr" rid="bib62">Pavlovic et al., 2018</xref>). We started with human (GRCh38) exon definitions from Ensembl version 98. We filtered this set of definitions for biotypes ‘protein_coding’ using the command mkgtf from cellranger (10× genomics). We then removed exon segments that were in exon definitions for multiple genes. This broke some exons into smaller unique exons. We then removed exons smaller than 10 bp. We took the final set of exons (1,371,917 exons from 20,338 genes) and extracted their sequences from the genome Ensembl GRCh38.p12. We used BLAT version 35 to identify orthologous sequences within the chimpanzee genome (panTro6) (<xref ref-type="bibr" rid="bib36">Kent, 2002</xref>). We removed hits with indels larger than 25 bp (using a function blatOutIndelIdent from <ext-link ext-link-type="uri" xlink:href="https://bitbucket.org/ee_reh_neh/orthoexon">https://bitbucket.org/ee_reh_neh/orthoexon</ext-link>). We then extracted the panTro6 sequences that had the highest sequence identity. We ran BLAT on this orthologous exon set to find matches in both the human and chimpanzee genomes. We removed exons that did not return the original location in humans or chimpanzees, as well as exons that mapped to multiple places with higher than 90% sequence identity. We removed exons from different human genes that mapped to overlapping regions in the chimpanzee genome. Finally, we removed exons that mapped to a different contig than the majority of exons from each gene. This resulted in a set of 1,250,820 orthologous exons from 19,515 genes.</p><p>We mapped on average 18.6 million reads to orthologous exons. We collapsed orthologous exons to quantify raw gene expression for each gene in each individual. We standardized counts and filtered out genes that did not pass the criteria of log<sub>2</sub>(CPM) values greater than 1 in 8 of the 12 individuals. To prepare counts for differential expression modeling, we used the Voom function with the quantile normalization method in the limma R package (<xref ref-type="bibr" rid="bib68">Ritchie et al., 2015</xref>). We used principal component analysis (PCA) to test for batch effects. PC1 explains 35.1% of the variation and is highly correlated with species (R<sup>2</sup> = 0.98) (<xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>). Collected metadata such as the percent of live cells at collection, cell concentration at collection, RIN score, and RNA concentration do not segregate by species (<xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>). We modeled species as a fixed effect and called genes as differentially expressed at a 5% FDR. The results from our differential expression analysis, including effect sizes and significance values, are available in <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>.</p></sec><sec id="s4-7"><title>Integration of translation and protein data</title><p>We downloaded differentially translated genes and their effect sizes from Additional file 5 of <xref ref-type="bibr" rid="bib81">Wang et al., 2018b</xref>. Wang et al. modeled differential translation using ribosome profiling of four human, four chimpanzee, and four rhesus macaque LCLs. For all integrations, we conditioned on the 6407 genes tested in the Wang et al. study and in our APA analysis. We tested for enrichments using a one-sided hypergeometric test implemented in R. We tested for correlations in effect sizes between site-level ΔPAU and translation HvC effect sizes by first filtering for the top PAS (see top PAS method above, <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). We report Pearson’s correlations calculated in R.</p><p>We downloaded differential protein-level genes, effect sizes, and directional selection classifications from <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref> of <xref ref-type="bibr" rid="bib39">Khan et al., 2013</xref>. Khan et al. modeled differential protein expression of 3390 genes using high-resolution mass spectrometry of stable isotope labeling by amino acids in cell culture (SILAC) collected from five human, five chimpanzee, and five rhesus macaque LCLs.</p></sec><sec id="s4-8"><title>Supplemental functional data</title><p>We downloaded human protein length (in number of amino acids) for proteins annotated as reviewed for high confidence from UniProtKB (<xref ref-type="bibr" rid="bib77">UniProt Consortium, 2019</xref>). We downloaded ubiquitination protein modification data from PhoshoSitePlus version 050320 (<xref ref-type="bibr" rid="bib30">Hornbeck et al., 2015</xref>). For all analyses in which we used interaction or ubiquitination data, we normalized the values by number of amino acids. To identify 3' UTR AREs in human RefSeq annotated 3' UTRs, we used the transcriptome_properties.py script published in <xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>, available at <ext-link ext-link-type="uri" xlink:href="https://github.com/stephenfloor/tripseq-analysis">https://github.com/stephenfloor/tripseq-analysis</ext-link> (<xref ref-type="bibr" rid="bib53">Mittleman, 2021b</xref>; copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:d7c631f71dd7ab3a9d40cbce627c5fda9281e4e7;origin=https://github.com/stephenfloor/tripseq-analysis;visit=swh:1:snp:c7a7c70e5b66e638bde1708a5782ffd2d0417b34;anchor=swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957/">swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957</ext-link>) with the –au-elements flag (<xref ref-type="bibr" rid="bib26">Floor, 2017</xref>; <xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>). According to <xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>, the fraction of AU elements is the percentage of the 3' UTR with repeating AU elements of 5nt or more (<xref ref-type="bibr" rid="bib27">Floor and Doudna, 2016</xref>).</p></sec><sec id="s4-9"><title>Gene set enrichments</title><p>We performed Fast Gene Set Enrichment Analysis (FGSEA) in R (minSize = 15, maxSize = 500, nperm = 100,000) to identify enriched gene ontology (GO) terms enriched within the genes differentially expressed with at least one PAS differentially used between species. We downloaded the C5: GO terms from MSigDB. Significant GO terms can be found in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>. To identify enriched GO terms in the remaining analyses, we used the GOrilla setting for two unranked lists of genes. To test for GO terms associated with species-specific PAS, we input the genes with a species-specific PAS as the target list and all genes with at least one identified PAS as the background set. For the differentially translated and differential protein-level background sets, we tested genes with at least one differentially used PAS as the target. We identified no significant GO terms in tdifferential protein-level analysis. Finally, among genes with at least one differentially used PAS, we tested for enriched GO terms for the genes differentially expressed in protein but not mRNA. Significant terms for each set can be found in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We thank N Gonzales for comments on the manuscript. We thank Y Li, M Ward, and G Housman for useful discussion. Funding: This work was supported by the US National Institutes of Health (R01HG010772 and R35GM13172 to YG). BEM supported by T32 GM09197 to the University of Chicago and F31HL149259 to BEM from National Heart, Lung, and Blood Institute of the National Institutes of Health. SP was in part supported by the National Center for Advancing Translational Sciences of the NIH (K12 HL119995). This work was completed in part with resources provided by the University of Chicago Research Computing Center.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Funding acquisition, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Data curation, Investigation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Data curation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Formal analysis, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con5"><p>Resources, Data curation</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Resources, Supervision, Funding acquisition, Methodology, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Gene ontology (GO) enrichment results.</title><p>Regulatory processes, components, and functions identified through GOrilla as enriched at 5% FDR. Table includes the species-specific PAS genes (species-specific PAS), genes differentially translated with at least one differentially used PAS (TranslationandAPA), and the genes differentially expressed in protein but not mRNA that also have at least one differentially used PAS (APA and protein, no expression difference). GO term, description, p-value, and FDR q-value from GOrilla.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp1-v2.txt"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Polyadenylation sites (PAS) differential usage results.</title><p>Column names as described – PAS: polyadenylation site, gene: gene (cluster in leafcutter), PAS_logeffectsize: log effect size for differential usage of the PAS between human and chimpanzee, PAS_deltaPAU: difference in polyadenylation site usage between human and chimpanzee (delta PSI in leafcutter), Gene_logLR: log likelihood ratio for differential usage of any PAS in the gene (cluster likelihood ratio in leafcutter), Gene_adjustedP-value: adjusted p-value for differential usage of the any PAS in the gene (cluster p-value in leafcutter).</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp2-v2.txt"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Fast Gene Set Enrichment Analysis (FGSEA) results for enrichment genes differentially expressed with at least one differentially used polyadenylation sites (PAS).</title><p>FGSEA results using the C5: ontology gene set from the MSigDB collection. Pathway from C5 set, p-value adjusted for multiple testing, normalized enrichment score from FGSEA, and number of genes in the pathway.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp3-v2.txt"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>3' Sequencing metadata.</title><p>Column names as described – Species: Cell line species, Lines: Cell line ID, Fraction: Cellular fraction, CollectionDate: Date of cell harvest and nuclear isolation, Extraction_date: Date of RNA extraction, Collection_person: Author initial for who processed cell harvest and nuclear isolation, UndilutedAverage: Average of two cell count measurements 1 × 10<sup>6</sup>, AverageAlive: Average of two cell live dead counts – calculated with trypan blue stain, Concentration: Extracted RNA concentration (ng/mL), RIN: RIN score for extracted RNA, 260.280.Ratio: 260/280 ratio calculated on nanodrop, Library: 3' Seq library date, Reads: Number of sequenced reads, Mapped_wMP: Number of Mapped reads before removing reads likely due to misprimming, Mapped_Clean: Number of Mapped reads after removing reads likely due to misprimming.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp4-v2.txt"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Metadata for RNA-sequencing data.</title><p>Column names as described – Species: Cell line species, Lines: Cell line ID, Collection_person: Author initial for who processed cell harvest and nuclear isolation, UndilutedAverage: Average of two cell count measurements 1 × 10<sup>6</sup>, AverageAlive: Average of two cell live dead counts – calculated with trypan blue stain, CollectionDate: Date of cell harvest and nuclear isolation, Extraction: Date of RNA extraction, RIN: RIN score for extracted RNA, BioAConc: RNA concentration (ng/µL), Reads: Number of Sequenced reads, Mapped: Number of mapped reads, AssignedOrtho: Number of mapped reads assigned to orthologous exons.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp5-v2.txt"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>Differential expression results.</title><p>Column names as described – Differential expression results from limma. gene: tested gene, logFC: log twofold change in normalized gene expression, adj.P.Val: BH adjusted p-value from t test, B: Beta value, t: t statistic.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-62548-supp6-v2.txt"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-62548-transrepform-v2.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>Sequencing data available on GEO under accession GSE155245.</p><p>The following dataset was generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Mittleman</surname><given-names>BE</given-names></name><name><surname>Pott</surname><given-names>S</given-names></name><name><surname>Warland</surname><given-names>S</given-names></name><name><surname>Barr</surname><given-names>K</given-names></name><name><surname>Cuevas</surname><given-names>C</given-names></name><name><surname>Gilad</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Divergence in alternative polyadenylation contributes to gene regulatory differences between humans and chimpanzees</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="http://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155245">GSE155245</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ara</surname> <given-names>T</given-names></name><name><surname>Lopez</surname> <given-names>F</given-names></name><name><surname>Ritchie</surname> <given-names>W</given-names></name><name><surname>Benech</surname> <given-names>P</given-names></name><name><surname>Gautheret</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Conservation of alternative polyadenylation patterns in mammalian genes</article-title><source>BMC Genomics</source><volume>7</volume><elocation-id>189</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2164-7-189</pub-id><pub-id pub-id-type="pmid">16872498</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Banovich</surname> <given-names>NE</given-names></name><name><surname>Lan</surname> <given-names>X</given-names></name><name><surname>McVicker</surname> <given-names>G</given-names></name><name><surname>van de Geijn</surname> <given-names>B</given-names></name><name><surname>Degner</surname> <given-names>JF</given-names></name><name><surname>Blischak</surname> <given-names>JD</given-names></name><name><surname>Roux</surname> <given-names>J</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Methylation QTLs are associated with coordinated changes in transcription factor binding, histone modifications, and gene expression levels</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004663</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004663</pub-id><pub-id pub-id-type="pmid">25233095</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barreau</surname> <given-names>C</given-names></name><name><surname>Paillard</surname> <given-names>L</given-names></name><name><surname>Osborne</surname> <given-names>HB</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>AU-rich elements and associated factors: are there unifying principles?</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>7138</fpage><lpage>7150</lpage><pub-id pub-id-type="doi">10.1093/nar/gki1012</pub-id><pub-id pub-id-type="pmid">16391004</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Battle</surname> <given-names>A</given-names></name><name><surname>Khan</surname> <given-names>Z</given-names></name><name><surname>Wang</surname> <given-names>SH</given-names></name><name><surname>Mitrano</surname> <given-names>A</given-names></name><name><surname>Ford</surname> <given-names>MJ</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Genomic variation impact of regulatory variation from RNA to protein</article-title><source>Science</source><volume>347</volume><fpage>664</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1126/science.1260793</pub-id><pub-id pub-id-type="pmid">25657249</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Beaudoing</surname> <given-names>E</given-names></name><name><surname>Freier</surname> <given-names>S</given-names></name><name><surname>Wyatt</surname> <given-names>JR</given-names></name><name><surname>Claverie</surname> <given-names>JM</given-names></name><name><surname>Gautheret</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Patterns of variant polyadenylation signal usage in human genes</article-title><source>Genome Research</source><volume>10</volume><fpage>1001</fpage><lpage>1010</lpage><pub-id pub-id-type="doi">10.1101/gr.10.7.1001</pub-id><pub-id pub-id-type="pmid">10899149</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blake</surname> <given-names>LE</given-names></name><name><surname>Roux</surname> <given-names>J</given-names></name><name><surname>Hernando-Herraez</surname> <given-names>I</given-names></name><name><surname>Banovich</surname> <given-names>NE</given-names></name><name><surname>Perez</surname> <given-names>RG</given-names></name><name><surname>Hsiao</surname> <given-names>CJ</given-names></name><name><surname>Eres</surname> <given-names>I</given-names></name><name><surname>Cuevas</surname> <given-names>C</given-names></name><name><surname>Marques-Bonet</surname> <given-names>T</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A comparison of gene expression and DNA methylation patterns across tissues and species</article-title><source>Genome Research</source><volume>30</volume><fpage>250</fpage><lpage>262</lpage><pub-id pub-id-type="doi">10.1101/gr.254904.119</pub-id><pub-id pub-id-type="pmid">31953346</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blanchette</surname> <given-names>M</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name><name><surname>Riemer</surname> <given-names>C</given-names></name><name><surname>Elnitski</surname> <given-names>L</given-names></name><name><surname>Smit</surname> <given-names>AF</given-names></name><name><surname>Roskin</surname> <given-names>KM</given-names></name><name><surname>Baertsch</surname> <given-names>R</given-names></name><name><surname>Rosenbloom</surname> <given-names>K</given-names></name><name><surname>Clawson</surname> <given-names>H</given-names></name><name><surname>Green</surname> <given-names>ED</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name><name><surname>Miller</surname> <given-names>W</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Aligning multiple genomic sequences with the threaded blockset aligner</article-title><source>Genome Research</source><volume>14</volume><fpage>708</fpage><lpage>715</lpage><pub-id pub-id-type="doi">10.1101/gr.1933104</pub-id><pub-id pub-id-type="pmid">15060014</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blekhman</surname> <given-names>R</given-names></name><name><surname>Oshlack</surname> <given-names>A</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Segmental duplications contribute to gene expression differences between humans and chimpanzees</article-title><source>Genetics</source><volume>182</volume><fpage>627</fpage><lpage>630</lpage><pub-id pub-id-type="doi">10.1534/genetics.108.099960</pub-id><pub-id pub-id-type="pmid">19332884</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blekhman</surname> <given-names>R</given-names></name><name><surname>Marioni</surname> <given-names>JC</given-names></name><name><surname>Zumbo</surname> <given-names>P</given-names></name><name><surname>Stephens</surname> <given-names>M</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Sex-specific and lineage-specific alternative splicing in primates</article-title><source>Genome Research</source><volume>20</volume><fpage>180</fpage><lpage>189</lpage><pub-id pub-id-type="doi">10.1101/gr.099226.109</pub-id><pub-id pub-id-type="pmid">20009012</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buccitelli</surname> <given-names>C</given-names></name><name><surname>Selbach</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>mRNAs, proteins and the emerging principles of gene expression control</article-title><source>Nature Reviews Genetics</source><volume>21</volume><fpage>630</fpage><lpage>644</lpage><pub-id pub-id-type="doi">10.1038/s41576-020-0258-4</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cain</surname> <given-names>CE</given-names></name><name><surname>Blekhman</surname> <given-names>R</given-names></name><name><surname>Marioni</surname> <given-names>JC</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Gene expression differences among primates are associated with changes in a histone epigenetic modification</article-title><source>Genetics</source><volume>187</volume><fpage>1225</fpage><lpage>1234</lpage><pub-id pub-id-type="doi">10.1534/genetics.110.126177</pub-id><pub-id pub-id-type="pmid">21321133</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Calarco</surname> <given-names>JA</given-names></name><name><surname>Xing</surname> <given-names>Y</given-names></name><name><surname>Cáceres</surname> <given-names>M</given-names></name><name><surname>Calarco</surname> <given-names>JP</given-names></name><name><surname>Xiao</surname> <given-names>X</given-names></name><name><surname>Pan</surname> <given-names>Q</given-names></name><name><surname>Lee</surname> <given-names>C</given-names></name><name><surname>Preuss</surname> <given-names>TM</given-names></name><name><surname>Blencowe</surname> <given-names>BJ</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Global analysis of alternative splicing differences between humans and chimpanzees</article-title><source>Genes &amp; Development</source><volume>21</volume><fpage>2963</fpage><lpage>2975</lpage><pub-id pub-id-type="doi">10.1101/gad.1606907</pub-id><pub-id pub-id-type="pmid">17978102</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Caliskan</surname> <given-names>M</given-names></name><name><surname>Cusanovich</surname> <given-names>DA</given-names></name><name><surname>Ober</surname> <given-names>C</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The effects of EBV transformation on gene expression levels and methylation profiles</article-title><source>Human Molecular Genetics</source><volume>20</volume><fpage>1643</fpage><lpage>1652</lpage><pub-id pub-id-type="doi">10.1093/hmg/ddr041</pub-id><pub-id pub-id-type="pmid">21289059</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Çalışkan</surname> <given-names>M</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Ober</surname> <given-names>C</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The effect of freeze-thaw cycles on gene expression levels in lymphoblastoid cell lines</article-title><source>PLOS ONE</source><volume>9</volume><elocation-id>e107166</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0107166</pub-id><pub-id pub-id-type="pmid">25192014</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chick</surname> <given-names>JM</given-names></name><name><surname>Munger</surname> <given-names>SC</given-names></name><name><surname>Simecek</surname> <given-names>P</given-names></name><name><surname>Huttlin</surname> <given-names>EL</given-names></name><name><surname>Choi</surname> <given-names>K</given-names></name><name><surname>Gatti</surname> <given-names>DM</given-names></name><name><surname>Raghupathy</surname> <given-names>N</given-names></name><name><surname>Svenson</surname> <given-names>KL</given-names></name><name><surname>Churchill</surname> <given-names>GA</given-names></name><name><surname>Gygi</surname> <given-names>SP</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Defining the consequences of genetic variation on a proteome-wide scale</article-title><source>Nature</source><volume>534</volume><fpage>500</fpage><lpage>505</lpage><pub-id pub-id-type="doi">10.1038/nature18270</pub-id><pub-id pub-id-type="pmid">27309819</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>Chimpanzee Sequencing and Analysis Consortium</collab></person-group><year iso-8601-date="2005">2005</year><article-title>Initial sequence of the chimpanzee genome and comparison with the human genome</article-title><source>Nature</source><volume>437</volume><fpage>69</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1038/nature04072</pub-id><pub-id pub-id-type="pmid">16136131</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Colgan</surname> <given-names>DF</given-names></name><name><surname>Manley</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Mechanism and regulation of mRNA polyadenylation</article-title><source>Genes &amp; Development</source><volume>11</volume><fpage>2755</fpage><lpage>2766</lpage><pub-id pub-id-type="doi">10.1101/gad.11.21.2755</pub-id><pub-id pub-id-type="pmid">9353246</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Csárdi</surname> <given-names>G</given-names></name><name><surname>Franks</surname> <given-names>A</given-names></name><name><surname>Choi</surname> <given-names>DS</given-names></name><name><surname>Airoldi</surname> <given-names>EM</given-names></name><name><surname>Drummond</surname> <given-names>DA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Accounting for experimental noise reveals that mRNA levels, amplified by post-transcriptional processes, largely determine steady-state protein levels in yeast</article-title><source>PLOS Genetics</source><volume>11</volume><elocation-id>e1005206</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1005206</pub-id><pub-id pub-id-type="pmid">25950722</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>de Bie</surname> <given-names>P</given-names></name><name><surname>Ciechanover</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Ubiquitination of E3 ligases: self-regulation of the ubiquitin system via Proteolytic and non-proteolytic mechanisms</article-title><source>Cell Death &amp; Differentiation</source><volume>18</volume><fpage>1393</fpage><lpage>1402</lpage><pub-id pub-id-type="doi">10.1038/cdd.2011.16</pub-id><pub-id pub-id-type="pmid">21372847</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Di Giammartino</surname> <given-names>DC</given-names></name><name><surname>Nishida</surname> <given-names>K</given-names></name><name><surname>Manley</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Mechanisms and consequences of alternative polyadenylation</article-title><source>Molecular Cell</source><volume>43</volume><fpage>853</fpage><lpage>866</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2011.08.017</pub-id><pub-id pub-id-type="pmid">21925375</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dobin</surname> <given-names>A</given-names></name><name><surname>Davis</surname> <given-names>CA</given-names></name><name><surname>Schlesinger</surname> <given-names>F</given-names></name><name><surname>Drenkow</surname> <given-names>J</given-names></name><name><surname>Zaleski</surname> <given-names>C</given-names></name><name><surname>Jha</surname> <given-names>S</given-names></name><name><surname>Batut</surname> <given-names>P</given-names></name><name><surname>Chaisson</surname> <given-names>M</given-names></name><name><surname>Gingeras</surname> <given-names>TR</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>STAR: ultrafast universal RNA-seq aligner</article-title><source>Bioinformatics</source><volume>29</volume><fpage>15</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id><pub-id pub-id-type="pmid">23104886</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dubbury</surname> <given-names>SJ</given-names></name><name><surname>Boutz</surname> <given-names>PL</given-names></name><name><surname>Sharp</surname> <given-names>PA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>CDK12 regulates DNA repair genes by suppressing intronic polyadenylation</article-title><source>Nature</source><volume>564</volume><fpage>141</fpage><lpage>145</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0758-y</pub-id><pub-id pub-id-type="pmid">30487607</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dubnikov</surname> <given-names>T</given-names></name><name><surname>Ben-Gedalya</surname> <given-names>T</given-names></name><name><surname>Cohen</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Protein quality control in health and disease</article-title><source>Cold Spring Harbor Perspectives in Biology</source><volume>9</volume><elocation-id>a023523</elocation-id><pub-id pub-id-type="doi">10.1101/cshperspect.a023523</pub-id><pub-id pub-id-type="pmid">27864315</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Enard</surname> <given-names>W</given-names></name><name><surname>Fassbender</surname> <given-names>A</given-names></name><name><surname>Model</surname> <given-names>F</given-names></name><name><surname>Adorján</surname> <given-names>P</given-names></name><name><surname>Pääbo</surname> <given-names>S</given-names></name><name><surname>Olek</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Differences in DNA methylation patterns between humans and chimpanzees</article-title><source>Current Biology</source><volume>14</volume><fpage>R148</fpage><lpage>R149</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2004.01.042</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eres</surname> <given-names>IE</given-names></name><name><surname>Luo</surname> <given-names>K</given-names></name><name><surname>Hsiao</surname> <given-names>CJ</given-names></name><name><surname>Blake</surname> <given-names>LE</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Reorganization of 3D genome structure may contribute to gene regulatory evolution in primates</article-title><source>PLOS Genetics</source><volume>15</volume><elocation-id>e1008278</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1008278</pub-id><pub-id pub-id-type="pmid">31323043</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Floor</surname> <given-names>SN</given-names></name></person-group><year iso-8601-date="2017">2017</year><data-title>tripseq-analysis</data-title><source>Github</source><version designator="3e823ab">3e823ab</version><ext-link ext-link-type="uri" xlink:href="https://github.com/stephenfloor/tripseq-analysis">https://github.com/stephenfloor/tripseq-analysis</ext-link></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Floor</surname> <given-names>SN</given-names></name><name><surname>Doudna</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Tunable protein synthesis by transcript isoforms in human cells</article-title><source>eLife</source><volume>5</volume><elocation-id>e10921</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.10921</pub-id><pub-id pub-id-type="pmid">26735365</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gokhman</surname> <given-names>D</given-names></name><name><surname>Nissim-Rafinia</surname> <given-names>M</given-names></name><name><surname>Agranat-Tamir</surname> <given-names>L</given-names></name><name><surname>Housman</surname> <given-names>G</given-names></name><name><surname>García-Pérez</surname> <given-names>R</given-names></name><name><surname>Lizano</surname> <given-names>E</given-names></name><name><surname>Cheronet</surname> <given-names>O</given-names></name><name><surname>Mallick</surname> <given-names>S</given-names></name><name><surname>Nieves-Colón</surname> <given-names>MA</given-names></name><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Alpaslan-Roodenberg</surname> <given-names>S</given-names></name><name><surname>Novak</surname> <given-names>M</given-names></name><name><surname>Gu</surname> <given-names>H</given-names></name><name><surname>Osinski</surname> <given-names>JM</given-names></name><name><surname>Ferrando-Bernal</surname> <given-names>M</given-names></name><name><surname>Gelabert</surname> <given-names>P</given-names></name><name><surname>Lipende</surname> <given-names>I</given-names></name><name><surname>Mjungu</surname> <given-names>D</given-names></name><name><surname>Kondova</surname> <given-names>I</given-names></name><name><surname>Bontrop</surname> <given-names>R</given-names></name><name><surname>Kullmer</surname> <given-names>O</given-names></name><name><surname>Weber</surname> <given-names>G</given-names></name><name><surname>Shahar</surname> <given-names>T</given-names></name><name><surname>Dvir-Ginzberg</surname> <given-names>M</given-names></name><name><surname>Faerman</surname> <given-names>M</given-names></name><name><surname>Quillen</surname> <given-names>EE</given-names></name><name><surname>Meissner</surname> <given-names>A</given-names></name><name><surname>Lahav</surname> <given-names>Y</given-names></name><name><surname>Kandel</surname> <given-names>L</given-names></name><name><surname>Liebergall</surname> <given-names>M</given-names></name><name><surname>Prada</surname> <given-names>ME</given-names></name><name><surname>Vidal</surname> <given-names>JM</given-names></name><name><surname>Gronostajski</surname> <given-names>RM</given-names></name><name><surname>Stone</surname> <given-names>AC</given-names></name><name><surname>Yakir</surname> <given-names>B</given-names></name><name><surname>Lalueza-Fox</surname> <given-names>C</given-names></name><name><surname>Pinhasi</surname> <given-names>R</given-names></name><name><surname>Reich</surname> <given-names>D</given-names></name><name><surname>Marques-Bonet</surname> <given-names>T</given-names></name><name><surname>Meshorer</surname> <given-names>E</given-names></name><name><surname>Carmel</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Differential DNA methylation of vocal and facial anatomy genes in modern humans</article-title><source>Nature Communications</source><volume>11</volume><elocation-id>1189</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-020-15020-6</pub-id><pub-id pub-id-type="pmid">32132541</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hilgers</surname> <given-names>V</given-names></name><name><surname>Perry</surname> <given-names>MW</given-names></name><name><surname>Hendrix</surname> <given-names>D</given-names></name><name><surname>Stark</surname> <given-names>A</given-names></name><name><surname>Levine</surname> <given-names>M</given-names></name><name><surname>Haley</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Neural-specific elongation of 3' UTRs during <italic>Drosophila</italic> development</article-title><source>PNAS</source><volume>108</volume><fpage>15864</fpage><lpage>15869</lpage><pub-id pub-id-type="doi">10.1073/pnas.1112672108</pub-id><pub-id pub-id-type="pmid">21896737</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hornbeck</surname> <given-names>PV</given-names></name><name><surname>Zhang</surname> <given-names>B</given-names></name><name><surname>Murray</surname> <given-names>B</given-names></name><name><surname>Kornhauser</surname> <given-names>JM</given-names></name><name><surname>Latham</surname> <given-names>V</given-names></name><name><surname>Skrzypek</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>PhosphoSitePlus, 2014: mutations, PTMs and recalibrations</article-title><source>Nucleic Acids Research</source><volume>43</volume><fpage>D512</fpage><lpage>D520</lpage><pub-id pub-id-type="doi">10.1093/nar/gku1267</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Housman</surname> <given-names>G</given-names></name><name><surname>Havill</surname> <given-names>LM</given-names></name><name><surname>Quillen</surname> <given-names>EE</given-names></name><name><surname>Comuzzie</surname> <given-names>AG</given-names></name><name><surname>Stone</surname> <given-names>AC</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Assessment of DNA methylation patterns in the bone and cartilage of a nonhuman primate model of osteoarthritis</article-title><source>Cartilage</source><volume>10</volume><fpage>335</fpage><lpage>345</lpage><pub-id pub-id-type="doi">10.1177/1947603518759173</pub-id><pub-id pub-id-type="pmid">29457464</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Housman</surname> <given-names>G</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Prime time for primate functional genomics</article-title><source>Current Opinion in Genetics &amp; Development</source><volume>62</volume><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1016/j.gde.2020.04.007</pub-id><pub-id pub-id-type="pmid">32544775</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>International HapMap Consortium</collab></person-group><year iso-8601-date="2005">2005</year><article-title>A haplotype map of the human genome</article-title><source>Nature</source><volume>437</volume><fpage>1299</fpage><lpage>1320</lpage><pub-id pub-id-type="doi">10.1038/nature04226</pub-id><pub-id pub-id-type="pmid">16255080</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>Z</given-names></name><name><surname>Lee</surname> <given-names>JY</given-names></name><name><surname>Pan</surname> <given-names>Z</given-names></name><name><surname>Jiang</surname> <given-names>B</given-names></name><name><surname>Tian</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Progressive lengthening of 3' untranslated regions of mRNAs by alternative polyadenylation during mouse embryonic development</article-title><source>PNAS</source><volume>106</volume><fpage>7028</fpage><lpage>7033</lpage><pub-id pub-id-type="doi">10.1073/pnas.0900028106</pub-id><pub-id pub-id-type="pmid">19372383</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kedersha</surname> <given-names>N</given-names></name><name><surname>Anderson</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Stress granules: sites of mRNA triage that regulate mRNA stability and translatability</article-title><source>Biochemical Society Transactions</source><volume>30</volume><fpage>963</fpage><lpage>969</lpage><pub-id pub-id-type="doi">10.1042/bst0300963</pub-id><pub-id pub-id-type="pmid">12440955</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kent</surname> <given-names>WJ</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>BLAT--the BLAST-like alignment tool</article-title><source>Genome Research</source><volume>12</volume><fpage>656</fpage><lpage>664</lpage><pub-id pub-id-type="doi">10.1101/gr.229202</pub-id><pub-id pub-id-type="pmid">11932250</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kent</surname> <given-names>WJ</given-names></name><name><surname>Sugnet</surname> <given-names>CW</given-names></name><name><surname>Furey</surname> <given-names>TS</given-names></name><name><surname>Roskin</surname> <given-names>KM</given-names></name><name><surname>Pringle</surname> <given-names>TH</given-names></name><name><surname>Zahler</surname> <given-names>AM</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>The human genome browser at UCSC</article-title><source>Genome Research</source><volume>12</volume><fpage>996</fpage><lpage>1006</lpage><pub-id pub-id-type="doi">10.1101/gr.229102</pub-id><pub-id pub-id-type="pmid">12045153</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khaitovich</surname> <given-names>P</given-names></name><name><surname>Enard</surname> <given-names>W</given-names></name><name><surname>Lachmann</surname> <given-names>M</given-names></name><name><surname>Pääbo</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Evolution of primate gene expression</article-title><source>Nature Reviews Genetics</source><volume>7</volume><fpage>693</fpage><lpage>702</lpage><pub-id pub-id-type="doi">10.1038/nrg1940</pub-id><pub-id pub-id-type="pmid">16921347</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname> <given-names>Z</given-names></name><name><surname>Ford</surname> <given-names>MJ</given-names></name><name><surname>Cusanovich</surname> <given-names>DA</given-names></name><name><surname>Mitrano</surname> <given-names>A</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Primate transcript and protein expression levels evolve under compensatory selection pressures</article-title><source>Science</source><volume>342</volume><fpage>1100</fpage><lpage>1104</lpage><pub-id pub-id-type="doi">10.1126/science.1242379</pub-id><pub-id pub-id-type="pmid">24136357</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>King</surname> <given-names>MC</given-names></name><name><surname>Wilson</surname> <given-names>AC</given-names></name></person-group><year iso-8601-date="1975">1975</year><article-title>Evolution at two levels in humans and chimpanzees</article-title><source>Science</source><volume>188</volume><fpage>107</fpage><lpage>116</lpage><pub-id pub-id-type="doi">10.1126/science.1090005</pub-id><pub-id pub-id-type="pmid">1090005</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>SH</given-names></name><name><surname>Singh</surname> <given-names>I</given-names></name><name><surname>Tisdale</surname> <given-names>S</given-names></name><name><surname>Abdel-Wahab</surname> <given-names>O</given-names></name><name><surname>Leslie</surname> <given-names>CS</given-names></name><name><surname>Mayr</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Widespread intronic polyadenylation inactivates tumour suppressor genes in leukaemia</article-title><source>Nature</source><volume>561</volume><fpage>127</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0465-8</pub-id><pub-id pub-id-type="pmid">30150773</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y</given-names></name><name><surname>Sun</surname> <given-names>Y</given-names></name><name><surname>Fu</surname> <given-names>Y</given-names></name><name><surname>Li</surname> <given-names>M</given-names></name><name><surname>Huang</surname> <given-names>G</given-names></name><name><surname>Zhang</surname> <given-names>C</given-names></name><name><surname>Liang</surname> <given-names>J</given-names></name><name><surname>Huang</surname> <given-names>S</given-names></name><name><surname>Shen</surname> <given-names>G</given-names></name><name><surname>Yuan</surname> <given-names>S</given-names></name><name><surname>Chen</surname> <given-names>L</given-names></name><name><surname>Chen</surname> <given-names>S</given-names></name><name><surname>Xu</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Dynamic landscape of tandem 3' UTRs during zebrafish development</article-title><source>Genome Research</source><volume>22</volume><fpage>1899</fpage><lpage>1906</lpage><pub-id pub-id-type="doi">10.1101/gr.128488.111</pub-id><pub-id pub-id-type="pmid">22955139</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>YI</given-names></name><name><surname>Knowles</surname> <given-names>DA</given-names></name><name><surname>Humphrey</surname> <given-names>J</given-names></name><name><surname>Barbeira</surname> <given-names>AN</given-names></name><name><surname>Dickinson</surname> <given-names>SP</given-names></name><name><surname>Im</surname> <given-names>HK</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Annotation-free quantification of RNA splicing using LeafCutter</article-title><source>Nature Genetics</source><volume>50</volume><fpage>151</fpage><lpage>158</lpage><pub-id pub-id-type="doi">10.1038/s41588-017-0004-9</pub-id><pub-id pub-id-type="pmid">29229983</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>Y</given-names></name><name><surname>Smyth</surname> <given-names>GK</given-names></name><name><surname>Shi</surname> <given-names>W</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>featureCounts: an efficient general purpose program for assigning sequence reads to genomic features</article-title><source>Bioinformatics</source><volume>30</volume><fpage>923</fpage><lpage>930</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btt656</pub-id><pub-id pub-id-type="pmid">24227677</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>Y</given-names></name><name><surname>Li</surname> <given-names>Z</given-names></name><name><surname>Ozsolak</surname> <given-names>F</given-names></name><name><surname>Kim</surname> <given-names>SW</given-names></name><name><surname>Arango-Argoty</surname> <given-names>G</given-names></name><name><surname>Liu</surname> <given-names>TT</given-names></name><name><surname>Tenenbaum</surname> <given-names>SA</given-names></name><name><surname>Bailey</surname> <given-names>T</given-names></name><name><surname>Monaghan</surname> <given-names>AP</given-names></name><name><surname>Milos</surname> <given-names>PM</given-names></name><name><surname>John</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>An in-depth map of polyadenylation sites in Cancer</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>8460</fpage><lpage>8471</lpage><pub-id pub-id-type="doi">10.1093/nar/gks637</pub-id><pub-id pub-id-type="pmid">22753024</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lindblad-Toh</surname> <given-names>K</given-names></name><name><surname>Garber</surname> <given-names>M</given-names></name><name><surname>Zuk</surname> <given-names>O</given-names></name><name><surname>Lin</surname> <given-names>MF</given-names></name><name><surname>Parker</surname> <given-names>BJ</given-names></name><name><surname>Washietl</surname> <given-names>S</given-names></name><name><surname>Kheradpour</surname> <given-names>P</given-names></name><name><surname>Ernst</surname> <given-names>J</given-names></name><name><surname>Jordan</surname> <given-names>G</given-names></name><name><surname>Mauceli</surname> <given-names>E</given-names></name><name><surname>Ward</surname> <given-names>LD</given-names></name><name><surname>Lowe</surname> <given-names>CB</given-names></name><name><surname>Holloway</surname> <given-names>AK</given-names></name><name><surname>Clamp</surname> <given-names>M</given-names></name><name><surname>Gnerre</surname> <given-names>S</given-names></name><name><surname>Alföldi</surname> <given-names>J</given-names></name><name><surname>Beal</surname> <given-names>K</given-names></name><name><surname>Chang</surname> <given-names>J</given-names></name><name><surname>Clawson</surname> <given-names>H</given-names></name><name><surname>Cuff</surname> <given-names>J</given-names></name><name><surname>Di Palma</surname> <given-names>F</given-names></name><name><surname>Fitzgerald</surname> <given-names>S</given-names></name><name><surname>Flicek</surname> <given-names>P</given-names></name><name><surname>Guttman</surname> <given-names>M</given-names></name><name><surname>Hubisz</surname> <given-names>MJ</given-names></name><name><surname>Jaffe</surname> <given-names>DB</given-names></name><name><surname>Jungreis</surname> <given-names>I</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name><name><surname>Kostka</surname> <given-names>D</given-names></name><name><surname>Lara</surname> <given-names>M</given-names></name><name><surname>Martins</surname> <given-names>AL</given-names></name><name><surname>Massingham</surname> <given-names>T</given-names></name><name><surname>Moltke</surname> <given-names>I</given-names></name><name><surname>Raney</surname> <given-names>BJ</given-names></name><name><surname>Rasmussen</surname> <given-names>MD</given-names></name><name><surname>Robinson</surname> <given-names>J</given-names></name><name><surname>Stark</surname> <given-names>A</given-names></name><name><surname>Vilella</surname> <given-names>AJ</given-names></name><name><surname>Wen</surname> <given-names>J</given-names></name><name><surname>Xie</surname> <given-names>X</given-names></name><name><surname>Zody</surname> <given-names>MC</given-names></name><name><surname>Worley</surname> <given-names>KC</given-names></name><name><surname>Kovar</surname> <given-names>CL</given-names></name><name><surname>Muzny</surname> <given-names>DM</given-names></name><name><surname>Gibbs</surname> <given-names>RA</given-names></name><name><surname>Warren</surname> <given-names>WC</given-names></name><name><surname>Mardis</surname> <given-names>ER</given-names></name><name><surname>Weinstock</surname> <given-names>GM</given-names></name><name><surname>Wilson</surname> <given-names>RK</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name><name><surname>Margulies</surname> <given-names>EH</given-names></name><name><surname>Herrero</surname> <given-names>J</given-names></name><name><surname>Green</surname> <given-names>ED</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name><name><surname>Siepel</surname> <given-names>A</given-names></name><name><surname>Goldman</surname> <given-names>N</given-names></name><name><surname>Pollard</surname> <given-names>KS</given-names></name><name><surname>Pedersen</surname> <given-names>JS</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A high-resolution map of human evolutionary constraint using 29 mammals</article-title><source>Nature</source><volume>478</volume><fpage>476</fpage><lpage>482</lpage><pub-id pub-id-type="doi">10.1038/nature10530</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mayer</surname> <given-names>A</given-names></name><name><surname>Churchman</surname> <given-names>LS</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genome-wide profiling of RNA polymerase transcription at Nucleotide resolution in human cells with native elongating transcript sequencing</article-title><source>Nature Protocols</source><volume>11</volume><fpage>813</fpage><lpage>833</lpage><pub-id pub-id-type="doi">10.1038/nprot.2016.047</pub-id><pub-id pub-id-type="pmid">27010758</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mayr</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Evolution and biological roles of alternative 3'UTRs</article-title><source>Trends in Cell Biology</source><volume>26</volume><fpage>227</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1016/j.tcb.2015.10.012</pub-id><pub-id pub-id-type="pmid">26597575</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mayr</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Regulation by 3'-Untranslated regions</article-title><source>Annual Review of Genetics</source><volume>51</volume><fpage>171</fpage><lpage>194</lpage><pub-id pub-id-type="doi">10.1146/annurev-genet-120116-024704</pub-id><pub-id pub-id-type="pmid">28853924</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVicker</surname> <given-names>G</given-names></name><name><surname>van de Geijn</surname> <given-names>B</given-names></name><name><surname>Degner</surname> <given-names>JF</given-names></name><name><surname>Cain</surname> <given-names>CE</given-names></name><name><surname>Banovich</surname> <given-names>NE</given-names></name><name><surname>Raj</surname> <given-names>A</given-names></name><name><surname>Lewellen</surname> <given-names>N</given-names></name><name><surname>Myrthil</surname> <given-names>M</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Identification of genetic variants that affect histone modifications in human cells</article-title><source>Science</source><volume>342</volume><fpage>747</fpage><lpage>749</lpage><pub-id pub-id-type="doi">10.1126/science.1242429</pub-id><pub-id pub-id-type="pmid">24136359</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mittleman</surname> <given-names>BE</given-names></name><name><surname>Pott</surname> <given-names>S</given-names></name><name><surname>Warland</surname> <given-names>S</given-names></name><name><surname>Zeng</surname> <given-names>T</given-names></name><name><surname>Mu</surname> <given-names>Z</given-names></name><name><surname>Kaur</surname> <given-names>M</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name><name><surname>Li</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Alternative polyadenylation mediates genetic regulation of gene expression</article-title><source>eLife</source><volume>9</volume><elocation-id>e57492</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.57492</pub-id><pub-id pub-id-type="pmid">32584258</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Mittleman</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2021">2021a</year><data-title>Comparative APA Analysis</data-title><source>Github</source><version designator="Afc57f4">Afc57f4</version><ext-link ext-link-type="uri" xlink:href="https://github.com/brimittleman/Comparative_APA">https://github.com/brimittleman/Comparative_APA</ext-link></element-citation></ref><ref id="bib53"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Mittleman</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2021">2021b</year><data-title>tripseq-analysis</data-title><source>Software Heritage</source><version designator="swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957">swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957</version><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:d7c631f71dd7ab3a9d40cbce627c5fda9281e4e7;origin=https://github.com/stephenfloor/tripseq-analysis;visit=swh:1:snp:c7a7c70e5b66e638bde1708a5782ffd2d0417b34;anchor=swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957/">https://archive.softwareheritage.org/swh:1:dir:d7c631f71dd7ab3a9d40cbce627c5fda9281e4e7;origin=https://github.com/stephenfloor/tripseq-analysis;visit=swh:1:snp:c7a7c70e5b66e638bde1708a5782ffd2d0417b34;anchor=swh:1:rev:3e823abcca5b8c1e5e89dd9bd4c49e8673b3e957/</ext-link></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moll</surname> <given-names>P</given-names></name><name><surname>Ante</surname> <given-names>M</given-names></name><name><surname>Seitz</surname> <given-names>A</given-names></name><name><surname>Reda</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>QuantSeq 3′ mRNA sequencing for RNA quantification</article-title><source>Nature Methods</source><volume>11</volume><elocation-id>972</elocation-id><pub-id pub-id-type="doi">10.1038/nmeth.f.376</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname> <given-names>AE</given-names></name><name><surname>Chenette</surname> <given-names>DM</given-names></name><name><surname>Larkin</surname> <given-names>LC</given-names></name><name><surname>Schneider</surname> <given-names>RJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Physiological networks and disease functions of RNA-binding protein AUF1: rna-binding protein AUF1</article-title><source>Wiley Interdisciplinary Reviews RNA</source><volume>5</volume><fpage>549</fpage><lpage>564</lpage><pub-id pub-id-type="doi">10.1002/wrna.1230</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morris</surname> <given-names>EK</given-names></name><name><surname>Caruso</surname> <given-names>T</given-names></name><name><surname>Buscot</surname> <given-names>F</given-names></name><name><surname>Fischer</surname> <given-names>M</given-names></name><name><surname>Hancock</surname> <given-names>C</given-names></name><name><surname>Maier</surname> <given-names>TS</given-names></name><name><surname>Meiners</surname> <given-names>T</given-names></name><name><surname>Müller</surname> <given-names>C</given-names></name><name><surname>Obermaier</surname> <given-names>E</given-names></name><name><surname>Prati</surname> <given-names>D</given-names></name><name><surname>Socher</surname> <given-names>SA</given-names></name><name><surname>Sonnemann</surname> <given-names>I</given-names></name><name><surname>Wäschke</surname> <given-names>N</given-names></name><name><surname>Wubet</surname> <given-names>T</given-names></name><name><surname>Wurst</surname> <given-names>S</given-names></name><name><surname>Rillig</surname> <given-names>MC</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Choosing and using diversity indices: insights for ecological applications from the german biodiversity exploratories</article-title><source>Ecology and Evolution</source><volume>4</volume><fpage>3514</fpage><lpage>3524</lpage><pub-id pub-id-type="doi">10.1002/ece3.1155</pub-id><pub-id pub-id-type="pmid">25478144</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Müller-McNicoll</surname> <given-names>M</given-names></name><name><surname>Rossbach</surname> <given-names>O</given-names></name><name><surname>Hui</surname> <given-names>J</given-names></name><name><surname>Medenbach</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Auto-regulatory feedback by RNA-binding proteins</article-title><source>Journal of Molecular Cell Biology</source><volume>11</volume><fpage>930</fpage><lpage>939</lpage><pub-id pub-id-type="doi">10.1093/jmcb/mjz043</pub-id><pub-id pub-id-type="pmid">31152582</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pai</surname> <given-names>AA</given-names></name><name><surname>Bell</surname> <given-names>JT</given-names></name><name><surname>Marioni</surname> <given-names>JC</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A genome-wide study of DNA methylation patterns and gene expression levels in multiple human and chimpanzee tissues</article-title><source>PLOS Genetics</source><volume>7</volume><elocation-id>e1001316</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1001316</pub-id><pub-id pub-id-type="pmid">21383968</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pai</surname> <given-names>AA</given-names></name><name><surname>Baharian</surname> <given-names>G</given-names></name><name><surname>Pagé Sabourin</surname> <given-names>A</given-names></name><name><surname>Brinkworth</surname> <given-names>JF</given-names></name><name><surname>Nédélec</surname> <given-names>Y</given-names></name><name><surname>Foley</surname> <given-names>JW</given-names></name><name><surname>Grenier</surname> <given-names>JC</given-names></name><name><surname>Siddle</surname> <given-names>KJ</given-names></name><name><surname>Dumaine</surname> <given-names>A</given-names></name><name><surname>Yotova</surname> <given-names>V</given-names></name><name><surname>Johnson</surname> <given-names>ZP</given-names></name><name><surname>Lanford</surname> <given-names>RE</given-names></name><name><surname>Burge</surname> <given-names>CB</given-names></name><name><surname>Barreiro</surname> <given-names>LB</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Widespread shortening of 3' Untranslated regions and increased exon inclusion are evolutionarily conserved features of innate immune responses to infection</article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1006338</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1006338</pub-id><pub-id pub-id-type="pmid">27690314</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>Z</given-names></name><name><surname>Zhang</surname> <given-names>H</given-names></name><name><surname>Hague</surname> <given-names>LK</given-names></name><name><surname>Lee</surname> <given-names>JY</given-names></name><name><surname>Lutz</surname> <given-names>CS</given-names></name><name><surname>Tian</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>An intronic polyadenylation site in human and mouse CstF-77 genes suggests an evolutionarily conserved regulatory mechanism</article-title><source>Gene</source><volume>366</volume><fpage>325</fpage><lpage>334</lpage><pub-id pub-id-type="doi">10.1016/j.gene.2005.09.024</pub-id><pub-id pub-id-type="pmid">16316725</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patraquim</surname> <given-names>P</given-names></name><name><surname>Warnefors</surname> <given-names>M</given-names></name><name><surname>Alonso</surname> <given-names>CR</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Evolution of hox post-transcriptional regulation by alternative polyadenylation and microRNA modulation within 12 <italic>Drosophila</italic> genomes</article-title><source>Molecular Biology and Evolution</source><volume>28</volume><fpage>2453</fpage><lpage>2460</lpage><pub-id pub-id-type="doi">10.1093/molbev/msr073</pub-id><pub-id pub-id-type="pmid">21436120</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pavlovic</surname> <given-names>BJ</given-names></name><name><surname>Blake</surname> <given-names>LE</given-names></name><name><surname>Roux</surname> <given-names>J</given-names></name><name><surname>Chavarria</surname> <given-names>C</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A comparative assessment of human and chimpanzee iPSC-derived cardiomyocytes with primary heart tissues</article-title><source>Scientific Reports</source><volume>8</volume><elocation-id>15312</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-018-33478-9</pub-id><pub-id pub-id-type="pmid">30333510</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pollard</surname> <given-names>KS</given-names></name><name><surname>Hubisz</surname> <given-names>MJ</given-names></name><name><surname>Rosenbloom</surname> <given-names>KR</given-names></name><name><surname>Siepel</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Detection of nonneutral substitution rates on mammalian phylogenies</article-title><source>Genome Research</source><volume>20</volume><fpage>110</fpage><lpage>121</lpage><pub-id pub-id-type="doi">10.1101/gr.097857.109</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pruitt</surname> <given-names>KD</given-names></name><name><surname>Tatusova</surname> <given-names>T</given-names></name><name><surname>Maglott</surname> <given-names>DR</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>NCBI Reference Sequence (RefSeq): a curated non-redundant sequence database of genomes, transcripts and proteins</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>D501</fpage><lpage>D504</lpage><pub-id pub-id-type="doi">10.1093/nar/gki025</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname> <given-names>AR</given-names></name><name><surname>Hall</surname> <given-names>IM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title><source>Bioinformatics</source><volume>26</volume><fpage>841</fpage><lpage>842</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id><pub-id pub-id-type="pmid">20110278</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ramírez</surname> <given-names>F</given-names></name><name><surname>Ryan</surname> <given-names>DP</given-names></name><name><surname>Grüning</surname> <given-names>B</given-names></name><name><surname>Bhardwaj</surname> <given-names>V</given-names></name><name><surname>Kilpert</surname> <given-names>F</given-names></name><name><surname>Richter</surname> <given-names>AS</given-names></name><name><surname>Heyne</surname> <given-names>S</given-names></name><name><surname>Dündar</surname> <given-names>F</given-names></name><name><surname>Manke</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>deepTools2: a next generation web server for deep-sequencing data analysis</article-title><source>Nucleic Acids Research</source><volume>44</volume><fpage>W160</fpage><lpage>W165</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw257</pub-id><pub-id pub-id-type="pmid">27079975</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ravid</surname> <given-names>T</given-names></name><name><surname>Hochstrasser</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Diversity of degradation signals in the ubiquitin-proteasome system</article-title><source>Nature Reviews Molecular Cell Biology</source><volume>9</volume><fpage>679</fpage><lpage>689</lpage><pub-id pub-id-type="doi">10.1038/nrm2468</pub-id><pub-id pub-id-type="pmid">18698327</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ritchie</surname> <given-names>ME</given-names></name><name><surname>Phipson</surname> <given-names>B</given-names></name><name><surname>Wu</surname> <given-names>D</given-names></name><name><surname>Hu</surname> <given-names>Y</given-names></name><name><surname>Law</surname> <given-names>CW</given-names></name><name><surname>Shi</surname> <given-names>W</given-names></name><name><surname>Smyth</surname> <given-names>GK</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Limma powers differential expression analyses for RNA-sequencing and microarray studies</article-title><source>Nucleic Acids Research</source><volume>43</volume><elocation-id>e47</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkv007</pub-id><pub-id pub-id-type="pmid">25605792</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Romero</surname> <given-names>IG</given-names></name><name><surname>Gopalakrishnan</surname> <given-names>S</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Widespread conservation of chromatin accessibility patterns and transcription factor binding in human and chimpanzee induced pluripotent stem cells</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/466631</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sandberg</surname> <given-names>R</given-names></name><name><surname>Neilson</surname> <given-names>JR</given-names></name><name><surname>Sarma</surname> <given-names>A</given-names></name><name><surname>Sharp</surname> <given-names>PA</given-names></name><name><surname>Burge</surname> <given-names>CB</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Proliferating cells express mRNAs with shortened 3' untranslated regions and fewer microRNA target sites</article-title><source>Science</source><volume>320</volume><fpage>1643</fpage><lpage>1647</lpage><pub-id pub-id-type="doi">10.1126/science.1155390</pub-id><pub-id pub-id-type="pmid">18566288</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>VA</given-names></name><name><surname>Graves-Lindsay</surname> <given-names>T</given-names></name><name><surname>Howe</surname> <given-names>K</given-names></name><name><surname>Bouk</surname> <given-names>N</given-names></name><name><surname>Chen</surname> <given-names>HC</given-names></name><name><surname>Kitts</surname> <given-names>PA</given-names></name><name><surname>Murphy</surname> <given-names>TD</given-names></name><name><surname>Pruitt</surname> <given-names>KD</given-names></name><name><surname>Thibaud-Nissen</surname> <given-names>F</given-names></name><name><surname>Albracht</surname> <given-names>D</given-names></name><name><surname>Fulton</surname> <given-names>RS</given-names></name><name><surname>Kremitzki</surname> <given-names>M</given-names></name><name><surname>Magrini</surname> <given-names>V</given-names></name><name><surname>Markovic</surname> <given-names>C</given-names></name><name><surname>McGrath</surname> <given-names>S</given-names></name><name><surname>Steinberg</surname> <given-names>KM</given-names></name><name><surname>Auger</surname> <given-names>K</given-names></name><name><surname>Chow</surname> <given-names>W</given-names></name><name><surname>Collins</surname> <given-names>J</given-names></name><name><surname>Harden</surname> <given-names>G</given-names></name><name><surname>Hubbard</surname> <given-names>T</given-names></name><name><surname>Pelan</surname> <given-names>S</given-names></name><name><surname>Simpson</surname> <given-names>JT</given-names></name><name><surname>Threadgold</surname> <given-names>G</given-names></name><name><surname>Torrance</surname> <given-names>J</given-names></name><name><surname>Wood</surname> <given-names>JM</given-names></name><name><surname>Clarke</surname> <given-names>L</given-names></name><name><surname>Koren</surname> <given-names>S</given-names></name><name><surname>Boitano</surname> <given-names>M</given-names></name><name><surname>Peluso</surname> <given-names>P</given-names></name><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Chin</surname> <given-names>CS</given-names></name><name><surname>Phillippy</surname> <given-names>AM</given-names></name><name><surname>Durbin</surname> <given-names>R</given-names></name><name><surname>Wilson</surname> <given-names>RK</given-names></name><name><surname>Flicek</surname> <given-names>P</given-names></name><name><surname>Eichler</surname> <given-names>EE</given-names></name><name><surname>Church</surname> <given-names>DM</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Evaluation of GRCh38 and de novo haploid genome assemblies demonstrates the enduring quality of the reference assembly</article-title><source>Genome Research</source><volume>27</volume><fpage>849</fpage><lpage>864</lpage><pub-id pub-id-type="doi">10.1101/gr.213611.116</pub-id><pub-id pub-id-type="pmid">28396521</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sheppard</surname> <given-names>S</given-names></name><name><surname>Lawson</surname> <given-names>ND</given-names></name><name><surname>Zhu</surname> <given-names>LJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Accurate identification of polyadenylation sites from 3' end deep sequencing using a naive bayes classifier</article-title><source>Bioinformatics</source><volume>29</volume><fpage>2564</fpage><lpage>2571</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btt446</pub-id><pub-id pub-id-type="pmid">23962617</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Siegel</surname> <given-names>DA</given-names></name><name><surname>Tonqueze</surname> <given-names>OL</given-names></name><name><surname>Biton</surname> <given-names>A</given-names></name><name><surname>Zaitlen</surname> <given-names>N</given-names></name><name><surname>Erle</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Massively parallel analysis of human 3′ UTRs Reveals that AU-Rich Element Length and Registration Predict mRNA Destabilization</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.02.12.945063</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Y</given-names></name><name><surname>Zhang</surname> <given-names>Y</given-names></name><name><surname>Hamilton</surname> <given-names>K</given-names></name><name><surname>Manley</surname> <given-names>JL</given-names></name><name><surname>Shi</surname> <given-names>Y</given-names></name><name><surname>Walz</surname> <given-names>T</given-names></name><name><surname>Tong</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Molecular basis for the recognition of the human AAUAAA polyadenylation signal</article-title><source>PNAS</source><volume>115</volume><fpage>E1419</fpage><lpage>E1428</lpage><pub-id pub-id-type="doi">10.1073/pnas.1718723115</pub-id><pub-id pub-id-type="pmid">29208711</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>B</given-names></name><name><surname>Hu</surname> <given-names>J</given-names></name><name><surname>Zhang</surname> <given-names>H</given-names></name><name><surname>Lutz</surname> <given-names>CS</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>A large-scale analysis of mRNA polyadenylation of human and mouse genes</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>201</fpage><lpage>212</lpage><pub-id pub-id-type="doi">10.1093/nar/gki158</pub-id><pub-id pub-id-type="pmid">15647503</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>B</given-names></name><name><surname>Manley</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Alternative polyadenylation of mRNA precursors</article-title><source>Nature Reviews Molecular Cell Biology</source><volume>18</volume><fpage>18</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1038/nrm.2016.116</pub-id><pub-id pub-id-type="pmid">27677860</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>UniProt Consortium</collab></person-group><year iso-8601-date="2019">2019</year><article-title>UniProt: a worldwide hub of protein knowledge</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D506</fpage><lpage>D515</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1049</pub-id><pub-id pub-id-type="pmid">30395287</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vasudevan</surname> <given-names>S</given-names></name><name><surname>Peltz</surname> <given-names>SW</given-names></name><name><surname>Wilusz</surname> <given-names>CJ</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Non-stop decay--a new mRNA surveillance pathway</article-title><source>BioEssays</source><volume>24</volume><fpage>785</fpage><lpage>788</lpage><pub-id pub-id-type="doi">10.1002/bies.10153</pub-id><pub-id pub-id-type="pmid">12210514</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Verheijen</surname> <given-names>J</given-names></name><name><surname>Wong</surname> <given-names>SY</given-names></name><name><surname>Rowe</surname> <given-names>JH</given-names></name><name><surname>Raymond</surname> <given-names>K</given-names></name><name><surname>Stoddard</surname> <given-names>J</given-names></name><name><surname>Delmonte</surname> <given-names>OM</given-names></name><name><surname>Bosticardo</surname> <given-names>M</given-names></name><name><surname>Dobbs</surname> <given-names>K</given-names></name><name><surname>Niemela</surname> <given-names>J</given-names></name><name><surname>Calzoni</surname> <given-names>E</given-names></name><name><surname>Pai</surname> <given-names>S-Y</given-names></name><name><surname>Choi</surname> <given-names>U</given-names></name><name><surname>Yamazaki</surname> <given-names>Y</given-names></name><name><surname>Comeau</surname> <given-names>AM</given-names></name><name><surname>Janssen</surname> <given-names>E</given-names></name><name><surname>Henderson</surname> <given-names>L</given-names></name><name><surname>Hazen</surname> <given-names>M</given-names></name><name><surname>Berry</surname> <given-names>G</given-names></name><name><surname>Rosenzweig</surname> <given-names>SD</given-names></name><name><surname>Aldhekri</surname> <given-names>HH</given-names></name><name><surname>He</surname> <given-names>M</given-names></name><name><surname>Notarangelo</surname> <given-names>LD</given-names></name><name><surname>Morava</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Defining a new immune deficiency syndrome: man2b2-cdg</article-title><source>Journal of Allergy and Clinical Immunology</source><volume>145</volume><fpage>1008</fpage><lpage>1011</lpage><pub-id pub-id-type="doi">10.1016/j.jaci.2019.11.016</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R</given-names></name><name><surname>Zheng</surname> <given-names>D</given-names></name><name><surname>Yehia</surname> <given-names>G</given-names></name><name><surname>Tian</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018a</year><article-title>A compendium of conserved cleavage and polyadenylation events in mammalian genes</article-title><source>Genome Research</source><volume>28</volume><fpage>1427</fpage><lpage>1441</lpage><pub-id pub-id-type="doi">10.1101/gr.237826.118</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>SH</given-names></name><name><surname>Hsiao</surname> <given-names>CJ</given-names></name><name><surname>Khan</surname> <given-names>Z</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2018">2018b</year><article-title>Post-translational buffering leads to convergent protein expression levels between primates</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>83</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1451-z</pub-id><pub-id pub-id-type="pmid">29950183</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ward</surname> <given-names>MC</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A generally conserved response to hypoxia in iPSC-derived cardiomyocytes from humans and chimpanzees</article-title><source>eLife</source><volume>8</volume><elocation-id>e42374</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.42374</pub-id><pub-id pub-id-type="pmid">30958265</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weber</surname> <given-names>M</given-names></name><name><surname>Hellmann</surname> <given-names>I</given-names></name><name><surname>Stadler</surname> <given-names>MB</given-names></name><name><surname>Ramos</surname> <given-names>L</given-names></name><name><surname>Pääbo</surname> <given-names>S</given-names></name><name><surname>Rebhan</surname> <given-names>M</given-names></name><name><surname>Schübeler</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Distribution, silencing potential and evolutionary impact of promoter DNA methylation in the human genome</article-title><source>Nature Genetics</source><volume>39</volume><fpage>457</fpage><lpage>466</lpage><pub-id pub-id-type="doi">10.1038/ng1990</pub-id><pub-id pub-id-type="pmid">17334365</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>C</given-names></name><name><surname>Chen</surname> <given-names>G</given-names></name><name><surname>Song</surname> <given-names>C</given-names></name><name><surname>Keefe</surname> <given-names>J</given-names></name><name><surname>Mendelson</surname> <given-names>M</given-names></name><name><surname>Huan</surname> <given-names>T</given-names></name><name><surname>Sun</surname> <given-names>BB</given-names></name><name><surname>Laser</surname> <given-names>A</given-names></name><name><surname>Maranville</surname> <given-names>JC</given-names></name><name><surname>Wu</surname> <given-names>H</given-names></name><name><surname>Ho</surname> <given-names>JE</given-names></name><name><surname>Courchesne</surname> <given-names>P</given-names></name><name><surname>Lyass</surname> <given-names>A</given-names></name><name><surname>Larson</surname> <given-names>MG</given-names></name><name><surname>Gieger</surname> <given-names>C</given-names></name><name><surname>Graumann</surname> <given-names>J</given-names></name><name><surname>Johnson</surname> <given-names>AD</given-names></name><name><surname>Danesh</surname> <given-names>J</given-names></name><name><surname>Runz</surname> <given-names>H</given-names></name><name><surname>Hwang</surname> <given-names>SJ</given-names></name><name><surname>Liu</surname> <given-names>C</given-names></name><name><surname>Butterworth</surname> <given-names>AS</given-names></name><name><surname>Suhre</surname> <given-names>K</given-names></name><name><surname>Levy</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Author correction: genome-wide mapping of plasma protein QTLs identifies putatively causal genes and pathways for cardiovascular disease</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>3853</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-06231-z</pub-id><pub-id pub-id-type="pmid">30228274</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X</given-names></name><name><surname>Cain</surname> <given-names>CE</given-names></name><name><surname>Myrthil</surname> <given-names>M</given-names></name><name><surname>Lewellen</surname> <given-names>N</given-names></name><name><surname>Michelini</surname> <given-names>K</given-names></name><name><surname>Davenport</surname> <given-names>ER</given-names></name><name><surname>Stephens</surname> <given-names>M</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Epigenetic modifications are associated with inter-species gene expression variation in primates</article-title><source>Genome Biology</source><volume>15</volume><elocation-id>547</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-014-0547-3</pub-id><pub-id pub-id-type="pmid">25468404</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.62548.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Coop</surname><given-names>Graham</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, Davis</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p>Thank you for submitting your article &quot;Divergence in alternative polyadenylation contributes to gene regulatory differences between humans and chimpanzees&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by three peer reviewers, and the evaluation has been overseen by a Reviewing Editor and Naama Barkai as the Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers and Reviewing Editor have discussed the reviews with one another and drafted this decision to help you prepare a revised submission. The reviewers and I all agree that the paper is suitable for publication in <italic>eLife</italic>. From our discussion we agree that no additional experiments are necessary but that some additional analyses would help flesh out the biological conclusions of the paper. The reviewers all read over each other’s comments and thought the requested analyses seemed reasonable. I have included the full reviews below, for you to provide a point-by-point response to.</p><p><italic>Reviewer #1:</italic></p><p>The paper is a well-executed and thorough analysis of PAS usage in LCLs between humans and chimpanzees. The research is technically sound, and this is additionally demonstrated by the accompanying supplemental analyses, and the use of additional datasets which permitted studying how differential PAS usage relates to protein level expression. Yet the paper lacks direct connections to actual phenotypic and biological differences between humans and chimpanzees, and in this regard reads more of a technical work (perhaps suitable in its current form in a journal such as Genome Research or Nucleic Acids Research), than the type of study that is published on a specific biological phenotype as often presented in <italic>eLife</italic>. Along with additional issues listed below, are suggestions for adjusting the analyses and text to help bridge the gap to phenotype:</p><p>1) Throughout the Results sections the authors present a myriad of lists of gene and PAS usage sites that result from different ways of cutting the data and connecting PAS usage to isoform and protein expression. Can these lists be explored in more detail, perhaps through functional gene set enrichment analyses and/or the use of the GREAT? Analyzing the sets in at least this manner might help to connect PAS usage differences to actual biology between humans/chimpanzees as well as within each species.</p><p>2) While this reviewer greatly appreciated the assessment of PAS via 3' Seq along in tandem with mRNA expression, and in the context of their incorporation of published protein and isoform data on the same cell type, another main issue is the lack of any functional validation experiments for any of the insights they generate in each of the Results sections. This is a bit concerning considering the reliance on LCLs as a standalone cell type for this work.</p><p><italic>Reviewer #2:</italic></p><p>Mittleman and colleagues have used 3' RNA-seq to study change in polyadenylation site usage between humans and chimpanzees. The results I suspect will be of greatest interest are that PAS change is associated with change in both RNA and protein level gene expression. This was a very interesting study, and something I hadn't considered before. The data collected seems to be a perfect fit for their study and the general analysis approach is solid.</p><p>1) I had one major concern with the manuscript as it's written-throughout the results, effect sizes were consistently weak to moderate at best (1-1.5X enrichment, weak correlations, etc.). This in itself is fine. I strongly believe there is often too much emphasis on results with large effect sizes. That said, the Discussion makes some strong claims about their data without tempering their language to account for the effect size. Some additional background information contrasting their results with other aspects of the transcription process would likely help place the role of PAS change in gene expression divergence between species. For example, how strongly does change in ChIP-seq/ATAC-seq/DNase-seq signal associate with DE genes between species. Basically, statements like &quot;We showed that, across species, increased intronic PAS usage is associated with increased mRNA expression levels, while increased 3' UTR PAS usage is correlated with a decrease in mRNA expression&quot; could use some additional context to highlight the relative significance of this result.</p><p>2) &quot;Indeed, we found that inter-species differences in the usage of intronic and 3' UTR PAS correlate with differences in expression effect size between the species at an equal magnitude, but in opposite directions (Figure 3B). Increased usage of intronic sites is correlated with increased expression levels, while increased usage of 3' UTR sites is correlated with decreased expression.&quot;</p><p>It's unclear why the uncategorized correlation (-0.06) was deemed &quot;not meaningful&quot;, yet the categorized correlations (-0.077, 0.073) were considered to be &quot;correlated&quot;. Neither of these seem particularly well correlated to me.</p><p>3) Figure 6 could use additional null models to determine whether these findings are outside of what might be expected at random. If the authors were to draw the same number of genes at random as found in each of the 3 classes (e.g. 1251 PAS genes) enough times to generate a reasonable null estimate, do their results of gene overlap and proportion fall into the tail of the null estimate?</p><p>4) Figure 1—figure supplement 4: There appears to be a strong bias in PAS usage favoring the chimpanzee samples. Can the authors explain this result? It isn't immediately obvious that we wouldn't expect a more normal distribution of PAS usage divergence between species.</p><p><italic>Reviewer #3:</italic></p><p>The manuscript by Mittleman and co-authors investigates alternative polyadenylation (APA) as a potential mechanism underlying the genetic regulation of transcript and protein expression levels in primates. Previous studies have identified genetic and epigenetic regulatory mechanisms underlying inter-species differences in gene expression. This is the first study focusing specifically on APA functional conservation/divergence between humans and chimpanzees. The manuscript describes APA in lymphoblastoid cell lines from six humans and six chimpanzees. The manuscript's main finding is that APA is largely conserved in humans and chimpanzees. Genes with significantly different PAS usage between the two species are enriched among differentially expressed genes, as well as among genes that show differences in protein translation between species. However, these results are based on relatively small subsets of genes. The manuscript is mainly focused on the molecular mechanisms and features of APA between species, while missing to investigate and discuss the biological role of the genes involved. This is problematic when trying to draw broader conclusions from analyses that focus on relatively small subset of genes, without any information on their biological relevance. For example, are the genes with differential APA and gene (or protein) expression relevant for divergent traits between the two species?</p><p>1) Differential PAS – The manuscripts reports 2,342 PAS, in 1,705 Genes, with differential usage between species. Additional information on the function of these genes should be reported as well as on the directionality of effect relative to the pathways and biological processes involved.</p><p>2) Signal site changes and PAS usage – The manuscript reports that the presence of a species-specific signal site is associated with increased PAS usage. Is this in the correct direction? In other words, does presence of the signal site in one species correspond to increased usage in the same species?</p><p>3) Differences in APA and gene expression – From the results in this section, it looks like the number of genes differentially expressed that also have differential PAS usage are a small subset. Is this because of lack of power (due to the small number of individuals included in the current study) or an actual biological phenomenon? How does this result compare to studies in humans of differential expression and differential APA in response to treatments or across cell types? What are the features and function of the genes represented in this subset?</p><p>4) The correlation in Figure 3B is only slightly higher than the one in Figure 3A, while the finding of opposite correlations depending on the location of the PAS used is interesting, these data do not support the interpretation. I would recommend moving this section to the supplements (supplemental figures) and keeping in the main text only the results of the analysis focused on genes differential expressed between species. Even in this case the trends aren't supported by strong results, so the test should be careful as to not overinterpret suggestive patterns. I would recommend to use Spearman's correlation rather than Pearson's because it is less sensitive to outliers. I would also suggest that the authors consider a scenario where the location of the PAS influences the direction of gene expression change, thus using a logistic model to test this hypothesis. Similar considerations apply also to the analysis of differential APA and differential protein expression.</p><p>5) Variation in APA and differences in protein expression – This section of the manuscript compares the APA data with published data of protein translation and expression in human and chimpanzee. While the enrichments and correlations reported in the first 2 paragraphs (and Figure 5) are of potential interest, the protein data are limited to a few thousand genes. How many of these genes can also be annotated in the APA data generated in this study? In other words, without the actual number of overlaps, it is difficult to assess whether the reported results are robust and widespread as opposed to limited to a few hundred genes. What are these genes functions? What can we learn about the evolution of human and chimpanzee-specific traits? Are these genes expected to show differential regulation between species? For the correlation analysis that considers intronic and 3'UTR PAS locations, please see my comment to the similar analysis done for differentially expressed genes. I recommend using Spearman's correlation and a logistic model.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.62548.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><p>We thank all of the reviewers for their thoughtful comments on our manuscript. All of the reviewers suggested that we include more biological information to support the mechanistic trends that we describe in the paper. In response, we performed a number of gene set enrichment analyses and added the results to the manuscript and supplement. We also provided additional information about the species-specific PAS example gene, and the genes previously annotated as subject to directional selection. The reviewers also asked us to perform functional validation of our RNA-seq data. In early RNA sequencing papers, authors performed validation experiments with qPCR. We now know that results from functional high throughput sequencing analyses can be trusted provided the appropriate quality control measures are used (as is the case for our study). The main aim of this work was to establish a genome-wide census of APA events in human and chimps and to infer global mechanisms that underlie species specific APA. Thus, our conclusions are generally based on aggregated observations from many sites and are robust against false positives and false negative errors. Like the reviewers, we acknowledge the importance of functional follow-up on our results to understand the effect of APA differences on phenotypes. However, the editors have agreed that follow-up experiments are outside the scope of this manuscript. We believe this study opens the door for using a similar study design to understand APA conservation in more cell types and dynamic biological processes. We added a few sentences to the discussion to acknowledge this limitation and reaffirm the goals of the study. Please see below for specific responses and changes we have made to the manuscript.</p><disp-quote content-type="editor-comment"><p>Reviewer #1:</p><p>The paper is a well-executed and thorough analysis of PAS usage in LCLs between humans and chimpanzees. The research is technically sound, and this is additionally demonstrated by the accompanying supplemental analyses, and the use of additional datasets which permitted studying how differential PAS usage relates to protein level expression. Yet the paper lacks direct connections to actual phenotypic and biological differences between humans and chimpanzees, and in this regard reads more of a technical work (perhaps suitable in its current form in a journal such as Genome Research or Nucleic Acids Research), than the type of study that is published on a specific biological phenotype as often presented in eLife. Along with additional issues listed below, are suggestions for adjusting the analyses and text to help bridge the gap to phenotype:</p><p>1) Throughout the Results sections the authors present a myriad of lists of gene and PAS usage sites that result from different ways of cutting the data and connecting PAS usage to isoform and protein expression. Can these lists be explored in more detail, perhaps through functional gene set enrichment analyses and/or the use of the GREAT? Analyzing the sets in at least this manner might help to connect PAS usage differences to actual biology between humans/chimpanzees as well as within each species.</p></disp-quote><p>We have now conducted a number of gene enrichment analyses using both fgsea and Gorilla (the details of which are found in the Materials and methods). We found that the differentially expressed genes within the dAPA genes are enriched for RNA processing pathways, such as RNA catabolic process and mRNA metabolic processes. We also found a 32X enrichment of translation initiation genes within the dAPA genes that are also differentially translated.</p><p>Using all of the dAPA genes as a background, we found functional enrichments for the genes that are differentially expressed in protein and not in mRNA. We identified small but significant enrichments for a number of sets related to vital cellular processes such as ribonucleotide binding, protein-containing complex binding, nuclear transport, and nucleocytoplasmic transport. We also found a number of gene regulatory components and processes enriched for genes with species-specific PAS compared to all genes where we identified a PAS.</p><p>We added these results to the Results and Materials and methods section of the paper. We also added two additional supplemental tables with the results.</p><disp-quote content-type="editor-comment"><p>2) While this reviewer greatly appreciated the assessment of PAS via 3' Seq along in tandem with mRNA expression, and in the context of their incorporation of published protein and isoform data on the same cell type, another main issue is the lack of any functional validation experiments for any of the insights they generate in each of the Results sections. This is a bit concerning considering the reliance on LCLs as a standalone cell type for this work.</p></disp-quote><p>It’s a balance. By using LCLs we were able to consider the APA data in the context of many other data sets collected from the same lines. Indeed, the greatest advantage of the LCLs is that they are a renewable resource that is available from multiple primate species. There is no other cell line that is available from multiple individuals from chimpanzees, other than our own panels of iPSCs. Certainly, we hope that future studies will consider other cell types based on this panel.</p><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>Mittleman and colleagues have used 3' RNA-seq to study change in polyadenylation site usage between humans and chimpanzees. The results I suspect will be of greatest interest are that PAS change is associated with change in both RNA and protein level gene expression. This was a very interesting study, and something I hadn't considered before. The data collected seems to be a perfect fit for their study and the general analysis approach is solid.</p><p>1) I had one major concern with the manuscript as it's written-throughout the results, effect sizes were consistently weak to moderate at best (1-1.5X enrichment, weak correlations, etc.). This in itself is fine. I strongly believe there is often too much emphasis on results with large effect sizes. That said, the Discussion makes some strong claims about their data without tempering their language to account for the effect size. Some additional background information contrasting their results with other aspects of the transcription process would likely help place the role of PAS change in gene expression divergence between species. For example, how strongly does change in ChIP-seq/ATAC-seq/DNase-seq signal associate with DE genes between species. Basically, statements like &quot;We showed that, across species, increased intronic PAS usage is associated with increased mRNA expression levels, while increased 3' UTR PAS usage is correlated with a decrease in mRNA expression&quot; could use some additional context to highlight the relative significance of this result.</p></disp-quote><p>Thanks for the comment. We added the word modest to the sentence.</p><disp-quote content-type="editor-comment"><p>2) &quot;Indeed, we found that inter-species differences in the usage of intronic and 3' UTR PAS correlate with differences in expression effect size between the species at an equal magnitude, but in opposite directions (Figure 3B). Increased usage of intronic sites is correlated with increased expression levels, while increased usage of 3' UTR sites is correlated with decreased expression.&quot;</p><p>It's unclear why the uncategorized correlation (-0.06) was deemed &quot;not meaningful&quot;, yet the categorized correlations (-0.077, 0.073) were considered to be &quot;correlated&quot;. Neither of these seem particularly well correlated to me.</p></disp-quote><p>We also agree that the correlations in Figure 3B are not strong. This is part of the reason we also present Figure 3D. When we subset on significant genes the correlation is stronger. We revised the language in this section to temper the results in 3B and demonstrate that the small correlation motivated our subsequent sub-setting of the data.</p><disp-quote content-type="editor-comment"><p>3) Figure 6 could use additional null models to determine whether these findings are outside of what might be expected at random. If the authors were to draw the same number of genes at random as found in each of the 3 classes (e.g. 1251 PAS genes) enough times to generate a reasonable null estimate, do their results of gene overlap and proportion fall into the tail of the null estimate?</p></disp-quote><p>We performed this analysis and found the actual number of differential APA genes that are differentially expressed in protein and not mRNA is not significantly higher than what would be expected by change (<xref ref-type="fig" rid="sa2fig1">Author response image 1</xref>). The protein data comes from Khan et al. who measured 3,390 proteins with high resolution mass spec, thus analysis is fairly low powered. We are only able to sample 661 genes from 2632 genes that we have all of the data for. In addition, we do not expect APA differences to explain all or even a majority of the genes differentially expressed in protein but not mRNA. We expect a number of additional mechanisms to lead to protein differences.</p><fig id="sa2fig1"><label>Author response image 1.</label><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-62548-resp-fig1-v2.tif"/></fig><disp-quote content-type="editor-comment"><p>4) Figure 1—figure supplement 4: There appears to be a strong bias in PAS usage favoring the chimpanzee samples. Can the authors explain this result? It isn't immediately obvious that we wouldn't expect a more normal distribution of PAS usage divergence between species.</p></disp-quote><p>We thank the reviewer for noticing this in the supplemental figure. We also found this unexpected at first, but we believe the result is an artifact of our QC and filtering.</p><p>PAS usage is a ratio of the reads mapping to one PAS over the number or reads mapping to any PAS assigned to the same gene. We calculated the usage values on an inclusive set of PAS then filtered to PAS reaching 5% in either species. We decided to identify and quantify sites this way to account for species and genomic region-specific noise. As a result of this choice, the overall usage for every gene may not add up to exactly 100%. This means that even though we identify (on average) the same number of PAS per gene in each species, the structure in this plot is not unexpected.</p><p>The bias toward chimpanzee in the supplemental figure shows that PAS usage is spread more evenly across human PAS than chimpanzee PAS. Before calculating PAS usage for PAS, we needed to assign sites to genes. Because the human annotation is more sophisticated than the chimpanzee annotation, we annotated all of the PAS to the human annotation. We acknowledge that if many PAS in chimpanzee fall outside of the human annotated genic regions, we would have lost those sites and the bias could result in the structure seen in the supplement. This would occur because included chimpanzee PAS would have inflated usage ratios. However, in reality we lost more of the sites originally discovered in human because they fell outside of the annotation than chimpanzee (22,278 in human vs. 18,954 in chimpanzee).</p><p>We also do not think the structure is a result of technical factors. We performed PCA on PAS usage (Figure 1—figure supplement 5). The top PC explains 41.8% of the variation and is highly correlated with species. The second PC explains 13.1% of the variation and is slighty correlated with extraction date and the author who collected the data. Because we balanced these technical factors with respect to species in the original study design (Supplementary file 4), the technical factors likely do not contribute to the structure the reviewer identified in Figure 1—figure supplement 4.</p><p>We account for any structure in the data introduced by the ratio characteristic of the phenotype in the differential PAS usage analysis. We normalized the usage values before testing for differences.</p><p>The PAS usage bias could also have affected the dominance analysis. However, anytime we refer to the dominant PAS in the manuscript, we show robustness of the results by presenting the results at a range of cutoffs.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>The manuscript by Mittleman and co-authors investigates alternative polyadenylation (APA) as a potential mechanism underlying the genetic regulation of transcript and protein expression levels in primates. Previous studies have identified genetic and epigenetic regulatory mechanisms underlying inter-species differences in gene expression. This is the first study focusing specifically on APA functional conservation/divergence between humans and chimpanzees. The manuscript describes APA in lymphoblastoid cell lines from six humans and six chimpanzees. The manuscript's main finding is that APA is largely conserved in humans and chimpanzees. Genes with significantly different PAS usage between the two species are enriched among differentially expressed genes, as well as among genes that show differences in protein translation between species. However, these results are based on relatively small subsets of genes. The manuscript is mainly focused on the molecular mechanisms and features of APA between species, while missing to investigate and discuss the biological role of the genes involved. This is problematic when trying to draw broader conclusions from analyses that focus on relatively small subset of genes, without any information on their biological relevance. For example, are the genes with differential APA and gene (or protein) expression relevant for divergent traits between the two species?</p><p>1) Differential PAS – The manuscripts reports 2,342 PAS, in 1,705 Genes, with differential usage between species. Additional information on the function of these genes should be reported as well as on the directionality of effect relative to the pathways and biological processes involved.</p></disp-quote><p>We added a number of gene set enrichments to the paper. Please see the general comments above.</p><disp-quote content-type="editor-comment"><p>2) Signal site changes and PAS usage – The manuscript reports that the presence of a species-specific signal site is associated with increased PAS usage. Is this in the correct direction? In other words, does presence of the signal site in one species correspond to increased usage in the same species?</p></disp-quote><p>Yes, this is the correct direction. Presence of a signal site in the species corresponds to increased usage of the site in the same species. This is best seen in our example figure, Figure 1—figure supplement 12.</p><disp-quote content-type="editor-comment"><p>3) Differences in APA and gene expression – From the results in this section, it looks like the number of genes differentially expressed that also have differential PAS usage are a small subset. Is this because of lack of power (due to the small number of individuals included in the current study) or an actual biological phenomenon? How does this result compare to studies in humans of differential expression and differential APA in response to treatments or across cell types? What are the features and function of the genes represented in this subset?</p></disp-quote><p>For the gene functions please see comments to other reviewers above.</p><disp-quote content-type="editor-comment"><p>4) The correlation in Figure 3B is only slightly higher than the one in Figure 3A, while the finding of opposite correlations depending on the location of the PAS used is interesting, these data do not support the interpretation. I would recommend moving this section to the supplements (supplemental figures) and keeping in the main text only the results of the analysis focused on genes differential expressed between species. Even in this case the trends aren't supported by strong results, so the test should be careful as to not overinterpret suggestive patterns. I would recommend to use Spearman's correlation rather than Pearson's because it is less sensitive to outliers. I would also suggest that the authors consider a scenario where the location of the PAS influences the direction of gene expression change, thus using a logistic model to test this hypothesis. Similar considerations apply also to the analysis of differential APA and differential protein expression.</p></disp-quote><p>We have added Spearman’s correlations to the legend in Figure 3. The results are consistent. We do not know how to perform a logistic regression because we do not know what continuous variable we would use.</p><disp-quote content-type="editor-comment"><p>5) Variation in APA and differences in protein expression – This section of the manuscript compares the APA data with published data of protein translation and expression in human and chimpanzee. While the enrichments and correlations reported in the first 2 paragraphs (and Figure 5) are of potential interest, the protein data are limited to a few thousand genes. How many of these genes can also be annotated in the APA data generated in this study? In other words, without the actual number of overlaps, it is difficult to assess whether the reported results are robust and widespread as opposed to limited to a few hundred genes. What are these genes functions? What can we learn about the evolution of human and chimpanzee-specific traits? Are these genes expected to show differential regulation between species? For the correlation analysis that considers intronic and 3'UTR PAS locations, please see my comment to the similar analysis done for differentially expressed genes. I recommend using Spearman's correlation and a logistic model.</p></disp-quote><p>For the gene functions please see comments to other reviewers above. We do not have a great way to annotate if a gene is expected to show differential regulation between species other than by overlapping the genes with other regulatory traits as we have done here. We note in the results the number of genes that we have APA and protein data for (3,391) and the number of genes we tested for expression and APA (7,462).</p></body></sub-article></article>