<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">82556</article-id><article-id pub-id-type="doi">10.7554/eLife.82556</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group><subj-group subj-group-type="heading"><subject>Structural Biology and Molecular Biophysics</subject></subj-group></article-categories><title-group><article-title>Structure-guided isoform identification for the human transcriptome</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-288908"><name><surname>Sommer</surname><given-names>Markus J</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-3414-1875</contrib-id><email>markusjsommer@gmail.com</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-289978"><name><surname>Cha</surname><given-names>Sooyoung</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7211-4603</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-114036"><name><surname>Varabyou</surname><given-names>Ales</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-289979"><name><surname>Rincon</surname><given-names>Natalia</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-289980"><name><surname>Park</surname><given-names>Sukhwan</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-289981"><name><surname>Minkin</surname><given-names>Ilia</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-289982"><name><surname>Pertea</surname><given-names>Mihaela</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes" id="author-289983"><name><surname>Steinegger</surname><given-names>Martin</given-names></name><email>martin.steinegger@snu.ac.kr</email><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="other" rid="fund7"/><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes" id="author-108273"><name><surname>Salzberg</surname><given-names>Steven L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8859-7432</contrib-id><email>salzberg@jhu.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Department of Biomedical Engineering, Johns Hopkins School of Medicine and Whiting School of Engineering</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Center for Computational Biology, Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04h9pn542</institution-id><institution>School of Biological Sciences, Seoul National University</institution></institution-wrap><addr-line><named-content content-type="city">Seoul</named-content></addr-line><country>Republic of Korea</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04h9pn542</institution-id><institution>Artificial Intelligence Institute, Seoul National University</institution></institution-wrap><addr-line><named-content content-type="city">Seoul</named-content></addr-line><country>Republic of Korea</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Department of Computer Science, Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff><aff id="aff6"><label>6</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04h9pn542</institution-id><institution>Institute of Molecular Biology and Genetics, Seoul National University</institution></institution-wrap><addr-line><named-content content-type="city">Seoul</named-content></addr-line><country>Republic of Korea</country></aff><aff id="aff7"><label>7</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00za53h95</institution-id><institution>Department of Biostatistics, Johns Hopkins University</institution></institution-wrap><addr-line><named-content content-type="city">Baltimore</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Dötsch</surname><given-names>Volker</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04cvxnb49</institution-id><institution>Goethe University</institution></institution-wrap><country>Germany</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Dötsch</surname><given-names>Volker</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04cvxnb49</institution-id><institution>Goethe University</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>15</day><month>12</month><year>2022</year></pub-date><pub-date pub-type="collection"><year>2022</year></pub-date><volume>11</volume><elocation-id>e82556</elocation-id><history><date date-type="received" iso-8601-date="2022-08-09"><day>09</day><month>08</month><year>2022</year></date><date date-type="accepted" iso-8601-date="2022-12-13"><day>13</day><month>12</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at .</event-desc><date date-type="preprint" iso-8601-date="2022-06-09"><day>09</day><month>06</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.06.08.495354"/></event></pub-history><permissions><copyright-statement>© 2022, Sommer et al</copyright-statement><copyright-year>2022</copyright-year><copyright-holder>Sommer et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-82556-v3.pdf"/><abstract><p>Recently developed methods to predict three-dimensional protein structure with high accuracy have opened new avenues for genome and proteome research. We explore a new hypothesis in genome annotation, namely whether computationally predicted structures can help to identify which of multiple possible gene isoforms represents a functional protein product. Guided by protein structure predictions, we evaluated over 230,000 isoforms of human protein-coding genes assembled from over 10,000 RNA sequencing experiments across many human tissues. From this set of assembled transcripts, we identified hundreds of isoforms with more confidently predicted structure and potentially superior function in comparison to canonical isoforms in the latest human gene database. We illustrate our new method with examples where structure provides a guide to function in combination with expression and evolutionary evidence. Additionally, we provide the complete set of structures as a resource to better understand the function of human genes and their isoforms. These results demonstrate the promise of protein structure prediction as a genome annotation tool, allowing us to refine even the most highly curated catalog of human proteins. More generally we demonstrate a practical, structure-guided approach that can be used to enhance the annotation of any genome.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>protein structure prediction</kwd><kwd>genome annotation</kwd><kwd>mouse</kwd><kwd>transcriptomics</kwd><kwd>proteomics</kwd><kwd>machine learning</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01-HG006677</award-id><principal-award-recipient><name><surname>Salzberg</surname><given-names>Steven L</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35-GM130151</award-id><principal-award-recipient><name><surname>Salzberg</surname><given-names>Steven L</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100003725</institution-id><institution>National Research Foundation of Korea</institution></institution-wrap></funding-source><award-id>2019R1-A6A1-A10073437</award-id><principal-award-recipient><name><surname>Steinegger</surname><given-names>Martin</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100003725</institution-id><institution>National Research Foundation of Korea</institution></institution-wrap></funding-source><award-id>2020M3-A9G7-103933</award-id><principal-award-recipient><name><surname>Steinegger</surname><given-names>Martin</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100003725</institution-id><institution>National Research Foundation of Korea</institution></institution-wrap></funding-source><award-id>2021-R1C1-C102065</award-id><principal-award-recipient><name><surname>Steinegger</surname><given-names>Martin</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100003725</institution-id><institution>National Research Foundation of Korea</institution></institution-wrap></funding-source><award-id>2021-M3A9-I4021220</award-id><principal-award-recipient><name><surname>Steinegger</surname><given-names>Martin</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100002551</institution-id><institution>Seoul National University</institution></institution-wrap></funding-source><award-id>Creative-Pioneering Researchers Program</award-id><principal-award-recipient><name><surname>Steinegger</surname><given-names>Martin</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The ability to accurately predict a protein's structure gives us an entirely new way to annotate the human genome.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>More than 20 years after the initial publication of the human genome, the scientific community is still trying to determine the complete set of human protein-coding genes. Although the number of genes is converging around 20,000, we do not yet have agreement on the precise number. The true number of different isoforms of human genes – variations due to alternative splicing, alternative transcription initiation sites, and alternative transcription termination sites – is even less certain. Currently, the major human gene annotation databases each contain well over 100,000 protein-coding transcripts (<xref ref-type="bibr" rid="bib15">Howe et al., 2021</xref>; <xref ref-type="bibr" rid="bib34">O’Leary et al., 2016</xref>; <xref ref-type="bibr" rid="bib14">Harrow et al., 2012</xref>; <xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>; <xref ref-type="bibr" rid="bib42">Salzberg, 2018</xref>), but the sets of transcripts vary widely among them. The disagreement between human transcriptome databases was clearly demonstrated when, in 2018, GENCODE and RefSeq were shown to agree on fewer than 50,000 of the nearly 300,000 total transcripts in their human annotations (<xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>).</p><p>Although the functions of many human genes are known, elucidating gene function remains a complex and time-consuming task. Given that at least 92% of human genes express more than one isoform (<xref ref-type="bibr" rid="bib54">Wang et al., 2008</xref>), and that the human transcriptome contains an average of seven or more unique transcripts per protein-coding gene (<xref ref-type="bibr" rid="bib48">Tung et al., 2020</xref>), the only feasible way to determine which isoforms are functional on a genome-wide scale is by using computational methods. Until now, the primary tools used to investigate gene function were sequence alignment and gene expression. Alignment relies on the long-established observation that if a protein is conserved in other species, then it is likely to be functional, particularly if the conservation extends to distantly related species (<xref ref-type="bibr" rid="bib25">Lindblad-Toh, 2011</xref>). This rule applies to isoforms as well: if we can find evidence that a particular sequence – e.g., a protein that uses an alternative exon – is present in species that diverged tens of millions of years ago, then the conservation of the sequence argues in favor of its function.</p><p>In a similar vein, the use of RNA sequencing (RNA-seq) to detect gene expression also provides clues to function: if a transcript is consistently expressed in multiple samples, it is more likely to be functional than one for which little expression evidence can be found. Genes may encode multiple transcripts that fold into distinct isoforms with well-defined functions (<xref ref-type="bibr" rid="bib54">Wang et al., 2008</xref>), but recent work has shown that most assembled human transcripts are found at very low levels in the transcriptomes of individual tissues (<xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>), and may simply reflect biological noise, products of intrinsically stochastic biochemical reactions (<xref ref-type="bibr" rid="bib8">Eling et al., 2019</xref>; <xref ref-type="bibr" rid="bib39">Ponting and Haerty, 2022</xref>). The large majority of assembled isoforms are unlikely to be functional, and indeed only a small percentage are included in current human genome annotation databases (<xref ref-type="bibr" rid="bib15">Howe et al., 2021</xref>; <xref ref-type="bibr" rid="bib34">O’Leary et al., 2016</xref>; <xref ref-type="bibr" rid="bib14">Harrow et al., 2012</xref>; <xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>; <xref ref-type="bibr" rid="bib42">Salzberg, 2018</xref>). Because transcription is noisy, the observation of transcription in RNA-seq data is insufficient evidence to conclude that a sequence is functional (<xref ref-type="bibr" rid="bib35">Palazzo and Lee, 2015</xref>).</p><p>This study explores a fundamentally new line of evidence that can be used to investigate protein function: computational prediction of three-dimensional (3D) structure. The recently developed AlphaFold2 system can automatically predict 3D protein structure with accuracy that often matches far more time-consuming laboratory methods (<xref ref-type="bibr" rid="bib18">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib49">Tunyasuvunakool et al., 2021</xref>), allowing us to generate structure predictions for thousands of gene isoforms. In proteins where a substantial portion folds into an ordered structure, estimated to be 68% of human proteins (<xref ref-type="bibr" rid="bib49">Tunyasuvunakool et al., 2021</xref>; <xref ref-type="bibr" rid="bib7">Deiana et al., 2019</xref>), a well-folded structure within an isoform argues in favor of its functionality. Conversely, a poorly folded isoform may indicate loss of function.</p><p>In a recent effort to create a single consensus annotation of all human protein-coding genes, two of the leading human genome annotation centers created the MANE (Matched Annotation from NCBI and EMBL-EBI) database (<xref ref-type="bibr" rid="bib31">Morales et al., 2022</xref>), a high-quality collection of protein-coding isoforms for which the annotation databases RefSeq (NCBI) and Ensembl-GENCODE (EMBL) match precisely. The goal of MANE is to identify just one isoform for each protein-coding gene that is well supported by experimental data, and to ensure that both databases agree on all exon boundaries as well as the sequence of the associated protein. In addition to the one-isoform-per-gene collection known as MANE Select, a small number of additional transcripts with special clinical significance, known as MANE Plus Clinical, are included in the database. Upon its initial release, MANE included only around 50% of human protein-coding genes. The latest version, v1.0, includes 19,062 genes and 19,120 transcripts, with an additional 58 transcripts included in the MANE Plus Clinical set. These transcripts have been described as a ‘universal standard’ for human gene annotation, and they provide a valuable resource to scientists and clinicians who need a consistent set of functional primary transcripts.</p><p>Here, we describe our use of protein folding predictions from ColabFold (<xref ref-type="bibr" rid="bib29">Mirdita et al., 2022</xref>), an open-source accelerated version of AlphaFold2, alongside experimental RNA-seq expression data from the Genotype-Tissue Expression (GTEx) project (<xref ref-type="bibr" rid="bib12">GTEx Consortium, 2013</xref>), to present substantial evidence for functional isoforms that can be used to improve human gene annotation, including the MANE gene set as well as the comprehensive human annotation databases RefSeq (<xref ref-type="bibr" rid="bib34">O’Leary et al., 2016</xref>), GENCODE (<xref ref-type="bibr" rid="bib14">Harrow et al., 2012</xref>), and the Comprehensive Human Expressed SequenceS (CHESS) database (<xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>). We dive into a few exemplary predictions to explain, biologically and evolutionarily, the 3D structure of our alternate isoforms. We also present an example of a novel protein isoform in mouse to demonstrate the general applicability of this structure-guided approach to improving functional annotation of any genome.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Scoring the transcriptome</title><p>Using CHESS, a large set of transcripts assembled from nearly 10,000 human RNA-seq experiments, we identified all protein-coding gene isoforms that were 1000aa or less in length (see Materials and methods). The 233,973 transcripts at 20,666 gene loci that fit this description encoded 127,398 distinct protein isoforms, and we predicted structures for all of them. Additionally, we included 3302 structure predictions for proteins with length &gt;1000aa from the AlphaFold Protein Structure Database (<xref ref-type="bibr" rid="bib53">Varadi et al., 2022</xref>) for isoforms with an exact protein sequence match in CHESS 3. This resulted in a total of 237,295 transcripts at 20,817 gene loci encoding 130,700 distinct protein isoforms. As shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>, we observe no strong trend in the relationship between predicted local distance difference test (pLDDT) and either protein length or overall transcript expression in the GTEx data. The lack of a clear linear relationship implies protein structure prediction may provide an orthogonal source of useful information for genome annotation efforts.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Predicted local distance difference test (pLDDT) distribution across the human transcriptome.</title><p>Two-dimensional joint histograms comparing pLDDT to protein amino acid length (<bold>a</bold>) and expression (<bold>b</bold>) measured in transcripts per million (TPM). For each protein-coding gene, only the isoform found in the highest number of Genotype-Tissue Expression (GTEx) samples is plotted. No strong trend is visible in the relationship between pLDDT and either protein length (<bold>a</bold>) or transcript expression (<bold>b</bold>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig1-v3.tif"/></fig><p>In total, 22,644 transcripts (9.7% of all examined transcripts) encoded an isoform that scored a higher pLDDT than the isoform encoded by the corresponding MANE transcript. However, many of these higher-scoring transcripts encoded relatively short, often low-expressed, likely non-functional fragments of larger proteins. Therefore, we incorporated filters based on RNA-seq expression data when determining which higher-scoring isoforms appeared superior to MANE isoforms (see ‘Filtering MANE comparisons’). Based on the combination of RNA-seq evidence and protein foldability, we identified 940 unique alternate isoforms at 632 loci (3.4% of all MANE loci) which appeared to have a more stable structure than the annotated primary isoform. Data for these 940 alternate isoforms can be found in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>. Worth noting is that over 96% of the MANE loci evaluated here contained no higher-scoring alternate isoforms that passed our filtering criteria. This is a testament to both the high degree of consistency in MANE and the sensitivity of protein structure prediction for finding instances where alternative isoforms may create functional products. Additionally, for 35% of all human protein-coding gene loci in our analysis, the most commonly observed isoform scored a pLDDT below 70, suggesting that intrinsic disorder may be an important feature of proteins at these loci.</p><p>Gene identifiers for all predicted protein isoforms as well as pLDDT scores and evolutionary conservation data from mouse can be found in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>. Predicted scores and GTEx expression data for all isoforms overlapping a MANE locus can be found in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>. All predicted protein structures as well as data for all tables in this paper are publicly available at our website, <ext-link ext-link-type="uri" xlink:href="https://isoform.io/">isoform.io</ext-link>.</p></sec><sec id="s2-2"><title>Exemplary predictions</title><p>To illustrate the improvements in human gene annotation that can be obtained using accurate structure prediction, we describe a small set of proteins, selected from <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>, where an alternate isoform appears to be superior to the isoform chosen for inclusion in MANE. For these examples, the alternative isoform is clearly functional based on structure as well as evolutionary conservation and, in some cases, additional expression evidence from RNA-seq data. For some of these examples, the MANE isoform is missing critical structural elements and may not be functional at all.</p><sec id="s2-2-1"><title>Acetylserotonin <italic>O</italic>-methyltransferase</title><p>Acetylserotonin <italic>O</italic>-methyltransferase (<italic>ASMT</italic>, alternatively <italic>HIOMT</italic>) is responsible for the final catalytic step in the production of melatonin, a critical hormone in sleep, metabolism, immune response, and neuronal development (<xref ref-type="bibr" rid="bib28">Melke et al., 2008</xref>). Depressed levels of circulating melatonin have been associated with autism spectrum disorder, and clinical studies have classified <italic>ASMT</italic> as a susceptibility gene due to the highly significant association between <italic>ASMT</italic> activity and autism (<xref ref-type="bibr" rid="bib28">Melke et al., 2008</xref>; <xref ref-type="bibr" rid="bib40">Rossignol and Frye, 2011</xref>).</p><p>The CHESS and GENCODE gene databases contain a 345aa isoform of <italic>ASMT</italic> (CHS.57426.4, ENST00000381229.9) while RefSeq is missing this isoform. The MANE version of this gene is 373aa long and appears in the CHESS (CHS.57426.2), GENCODE (ENST00000381241.9), and RefSeq (NM_001171038.2) gene databases. The predicted structures of both isoforms are shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Acetylserotonin <italic>O</italic>-methyltransferase (ASMT) isoform comparison.</title><p>Comparison of predicted structures of ASMT, showing the 373aa isoform from Matched Annotation from NCBI and EMBL-EBI (MANE) (CHS.57426.2, RefSeq NM_001171038.2, GENCODE ENST00000381241.9) on the left, and a 345aa alternate isoform from Comprehensive Human Expressed SequenceS (CHESS) (CHS.57426.4, GENCODE ENST00000381229.9) on the right. The CHESS 345aa isoform closely matches the experimentally determined X-ray crystal structure of the biologically active protein (<xref ref-type="bibr" rid="bib3">Botros et al., 2013</xref>), shown at the bottom.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig2-v3.tif"/></fig><p>We hypothesized that the highest scoring isoform of <italic>ASMT</italic> according to AlphaFold2 corresponds to the biologically active version of the protein. The score we use for these comparisons is the pLDDT score, which has been demonstrated to be a well-calibrated, consistent measure of protein structure prediction accuracy (<xref ref-type="bibr" rid="bib18">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib49">Tunyasuvunakool et al., 2021</xref>). A pLDDT score above 70 (the maximum is 100) indicates that a predicted structure can generally be trusted, while a score below 70 may indicate folding prediction failure or intrinsic disorder within a protein. Scores above 90 imply structure predictions accurate enough for highly shape-sensitive tasks such as chemical binding site characterization. The structure of the 345aa isoform has a very high pLDDT score of 94.7, versus the somewhat lower score of 87.1 for the 373aa MANE isoform.</p><p>Because melatonin is primarily synthesized within the human pineal gland at night, we quantified ASMT isoform expression using RNA-seq data from a previously published experiment that used tissue extracted from the pineal gland of a patient who died at midnight (<xref ref-type="bibr" rid="bib5">Chang et al., 2020</xref>). In this tissue sample, the 345aa isoform of <italic>ASMT</italic> was expressed at a level of 327 transcripts per million (TPM), while the 373aa isoform from MANE was expressed at 34 TPM, nearly 10 times lower, supporting our hypothesis that the higher scoring 345aa isoform is functional.</p><p>Further evidence for the functionality of the 343aa isoform is shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. An ectopic exon in the MANE <italic>ASMT</italic> protein creates an unstructured loop that bulges out from the primary structure. The alternate isoform, missing this ectopic exon, closely matches the experimentally determined <italic>ASMT</italic> X-ray crystal structure of the biologically active protein. Furthermore, as reported in <xref ref-type="bibr" rid="bib3">Botros et al., 2013</xref>, the insertion of exon 6, corresponding to the ectopic exon in the MANE isoform, distorts the structure and destroys its ability to bind <italic>S</italic>-adenosyl-L-methionine and to synthesize melatonin. Thus, the structural comparison, the expression evidence, and a melatonin synthesis activity assay all combine to support our hypothesis that the 345aa isoform represents the primary biologically functional isoform of <italic>ASMT</italic>.</p></sec><sec id="s2-2-2"><title>Gamma-N crystallin</title><p>Gamma-N crystallin (<italic>CRYGN</italic>) is a highly conserved member of the crystallin family of proteins, responsible for the transparency of the lens and cornea in vertebrate eyes (<xref ref-type="bibr" rid="bib1">Andley, 2007</xref>). The intron-exon structure of <italic>CRYGN</italic> has been conserved across at least 400 million years of vertebrate evolution, with close orthologs present in the genomes of chimpanzees, mice, frogs, and the white-rumped snowfinch. Given this extensive evolutionary history, <xref ref-type="bibr" rid="bib55">Wistow et al., 2005</xref>, were surprised to observe that the primate <italic>CRYGN</italic> gene has lost its canonical stop codon, leading them to conclude ‘the human gene has clearly changed its expression and may indeed be heading for extinction’.</p><p>As shown in <xref ref-type="fig" rid="fig3">Figure 3a</xref>, the MANE isoform (CHS.52273.5, RefSeq NM_144727.3, GENCODE ENST00000337323.3) that matches descriptions by <xref ref-type="bibr" rid="bib55">Wistow et al., 2005</xref>, includes sequences that do not fold well, as indicated by its pLDDT score of 67.7. However, we found an alternate <italic>CRYGN</italic> isoform, assembled from <xref ref-type="bibr" rid="bib12">GTEx Consortium, 2013</xref> data, that had a far higher pLDDT score of 92.2, shown in <xref ref-type="fig" rid="fig3">Figure 3b</xref>. Small differences between pLDDT scores may not be meaningful, but large score differences, such as the 24-point gap between the two isoforms of <italic>CRYGN</italic> discussed here, represent a substantial difference in prediction confidence across a large portion of the protein. The higher-scoring <italic>CRYGN</italic> isoform is present in CHESS (CHS.52273.9) and GENCODE (ENST00000644350.1), and it was also present in RefSeq v109 (XM_005249952.4) but was removed in the next release, v110.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>CRYGN isoform comparison.</title><p>(<bold>a</bold>) Predicted protein structure for the Matched Annotation from NCBI and EMBL-EBI (MANE) isoform (CHS.52273.5, RefSeq NM_144727.3, GENCODE ENST00000337323.3) of gamma-N crystallin (CRYGN), colored by predicted local distance difference test (pLDDT) score. (<bold>b</bold>) Predicted protein structure for a CRYGN alternate isoform (CHS.52273.9, GENCODE ENST00000644350.1). (<bold>c</bold>) Ramachandran plot for the MANE (CRYGN) isoform. Dark blue areas represent ‘favored’ regions while light blue represent ‘allowed’ regions (<xref ref-type="bibr" rid="bib26">Lovell et al., 2003</xref>). The 32 red dots represent amino acid residues with secondary structures that fall outside the allowed regions. (<bold>d</bold>) Ramachandran plot for the alternate CRYGN isoform with 4 red dots falling in disallowed regions, compared to 32 disallowed in MANE. All residues associated with the 4 red dots in the alternate isoform are shared with the MANE isoform in the poorly folded N-terminal region.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig3-v3.tif"/></fig><p>Both the MANE and alternate isoforms are exactly the same length despite having different C-terminal sequence content. Visual comparison of the predicted structure for the alternate isoform reveals a marked improvement in the structure in the β domain and a clear recovery of <italic>CRYGN</italic>’s dimer-like characteristic, with two structurally similar domains as shown in <xref ref-type="fig" rid="fig3">Figure 3b</xref> and <xref ref-type="video" rid="video1">Video 1</xref>. Ramachandran plots (<xref ref-type="fig" rid="fig3">Figure 3c and d</xref>) also support the structure of the alternate CHESS isoform.</p><media mimetype="video" mime-subtype="mp4" xlink:href="elife-82556-video1.mp4" id="video1"><label>Video 1.</label><caption><title>CRYGN comparison.</title><p>A three-dimensional (3D) animation comparing the predicted protein structure of the Matched Annotation from NCBI and EMBL-EBI (MANE) isoform (CHS.52273.5, RefSeq NM_144727.3, GENCODE ENST00000337323.3) of gamma-N crystallin (CRYGN) versus the predicted protein structure for the highest-scoring CRYGN alternate isoform (CHS.52273.9, GENCODE ENST00000644350.1).</p></caption></media><p>Encouraged by the substantially improved folding of this alternate isoform, we examined the intron-exon structure of <italic>CRYGN</italic> in human to determine how it recovered its functional shape despite losing its original stop codon. We found that the common vertebrate four-exon structure of <italic>CRYGN</italic> has changed to a five-exon structure in humans, as shown in <xref ref-type="fig" rid="fig4">Figure 4</xref>. In the well-folded alternate isoform, a novel primate-specific splice site removes the last four amino acids as compared to other vertebrates, but the new primate-specific fifth exon contains a downstream stop codon that adds four residues. The poorly folding MANE isoform (<xref ref-type="fig" rid="fig4">Figure 4</xref>, bottom), in contrast, entirely skips the fourth exon, resulting in a frameshift that adds 43 C-terminal amino acids which have no similarity to any <italic>CRYGN</italic> sequence outside of primates.</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>CRYGN intron-exon structure.</title><p>Comparison of gamma-N crystallin (CRYGN) transcript structures in frog, mouse, and human. Exons 1, 2, and 3 are highly conserved across all species. Exon 4 is missing from the poorly folding Matched Annotation from NCBI and EMBL-EBI (MANE) isoform, while exon 5 shows no homology to any species outside of primates. The loss of a stop codon in human exon 4 appears to be balanced by the inclusion of a short novel exon that adds only four amino acids to the final protein. Coding portions of exons are shown with thicker rectangles in teal. Intron lengths are reduced proportionately for the purpose of display.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig4-v3.tif"/></fig></sec><sec id="s2-2-3"><title>Thioredoxin domain-containing protein 8</title><p>The thioredoxin protein family represents an ancient group of highly conserved small globular proteins found in all forms of life (<xref ref-type="bibr" rid="bib30">Modi et al., 2018</xref>). Thioredoxin domain-containing protein 8 (<italic>TXNDC8</italic>, alternatively <italic>PTRX3</italic>) is a testis-specific enzyme responsible for catalyzing redox reactions via the oxidation of cysteine from dithiol to disulfide forms (<xref ref-type="bibr" rid="bib17">Jiménez et al., 2004</xref>).</p><p>As shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>, several canonical protein motifs appear altered or missing in the predicted structure of the human <italic>TXNDC8</italic> MANE transcript, as it lacks a highly conserved sequence that should start only eight residues away from the CGPC dithiol/disulfide active site. The α2 helix is severely truncated, leading directly to the α3 helix, thereby entirely skipping the β3 sheet. The β5 sheet, normally providing an interaction bridge between the α3 and α4 helices, is similarly missing in its entirety. Finally, the α4 helix is present but rotated 140 degrees relative to its canonical position. These large alterations to the fundamental thioredoxin protein structure result in the MANE transcript receiving a pLDDT score of 56.7.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>TXNDC8 isoform comparison.</title><p>Predicted protein structures for seven distinct human isoforms of thioredoxin domain-containing protein 8 (TXNDC8), as well as the primary cattle transcript and a novel mouse transcript. Alternate human isoforms 4, 9, and 12 (right side of figure) lack multiple canonical thioredoxin structures and thus appear non-functional. Several canonical protein motifs are missing or altered in the predicted structure of the Matched Annotation from NCBI and EMBL-EBI (MANE) transcript (top center). In contrast, the alternate human transcript 14 matches cattle and mouse to within 0.8 Å. Human transcript CHS.56446.14 is colored solid dark blue because every amino acid residue scores a predicted local distance difference test (pLDDT) above 90.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig5-v3.tif"/></fig><p>In stark contrast to the poorly folded MANE isoform (CHS.56446.8, RefSeq NM_001286946.2, GENCODE ENST00000423740.7), an alternate isoform assembled as part of the CHESS project (CHS.56446.14) and RefSeq (NM_001364963.2) has a pLDDT score of 96.9, an improvement of 40 points. Inspection of the alternate transcript (<xref ref-type="fig" rid="fig5">Figure 5</xref>) reveals full recovery of the canonical α2, α3, and α4 helices as well as the β3 and β5 sheets. Moreover, 3D alignment of the protein encoded by CHS.56446.14 to the protein from the primary <italic>TXNDC8</italic> isoform in <italic>Bos taurus</italic> (cow) reveals a very close structural correspondence between the two proteins, with a predicted root-mean-square deviation (RMSD) of 0.8 Å. <xref ref-type="fig" rid="fig5">Figure 5</xref> shows multiple alternative isoforms of human <italic>TXNDC8</italic> from the CHESS annotation, as well as the 3D alignment of CHS.56446.14 to its orthologs in cow and mouse. Given its substantially higher pLDDT score and near-perfect structural conservation in other species, the CHESS transcript appears to be a much better candidate for the canonical form of this protein. The MANE isoform, because it is missing multiple key structures, may represent a non-functional product of transcriptional noise.</p><p>All isoforms of <italic>TXNDC8</italic> shown in <xref ref-type="fig" rid="fig5">Figure 5</xref> were assembled from RNA-seq data during the construction of the CHESS database. This figure illustrates another potential use of structure prediction, namely the ability to distinguish among multiple functional and non-functional isoforms when annotating a genome. As discussed above, the MANE <italic>TXNDC8</italic> isoform appears non-functional, lacking several key structures. In addition, isoform 12 in <xref ref-type="fig" rid="fig5">Figure 5</xref> appears clearly non-functional, lacking all four of the β sheets and one of the α helices of isoform 14. Isoforms 4 and 9 also appear likely to be non-functional: both are missing one of the β sheets, and isoform 4 has a low pLDDT score of just 52. Although this example is only one of many, it illustrates how one can employ accurate 3D structure prediction in virtually any species as a powerful new tool to improve gene annotation.</p><p>Assemblies of RNA-seq experiments typically reveal thousands of un-annotated gene isoforms, as was illustrated by the use of GTEx to discover more than 100,000 new isoforms when building the original CHESS gene database (<xref ref-type="bibr" rid="bib37">Pertea et al., 2018</xref>). The approach used here, computationally folding each distinct protein encoded by alternate isoforms, allows us to compare the structure of these predicted proteins to the ‘best’ structure for each protein-coding gene locus. For those proteins with at least one high-confidence structure, this strategy may allow us to identify and remove potentially non-functional isoforms.</p></sec><sec id="s2-2-4"><title>Interleukin 36 beta</title><p>Interleukin 36 beta (<italic>IL36B</italic>, alternatively <italic>IL1F8</italic>, <italic>FIL1-ETA</italic>, or <italic>IL1H2</italic>) mediates inflammation as part of a signaling system in epithelial tissue. The pro-inflammatory properties of <italic>IL36B</italic> have been implicated in the pathogenesis of psoriasis (<xref ref-type="bibr" rid="bib4">Carrier et al., 2011</xref>), a common disease characterized by scaly rashes on the skin. In vitro and in vivo studies have found consistently increased expression of <italic>IL36B</italic> in psoriatic lesions, making the gene a potential target for future anti-psoriatic drugs (<xref ref-type="bibr" rid="bib50">Uppala et al., 2021</xref>).</p><p>The ColabFold structure of the <italic>IL36B</italic> MANE isoform (CHS.30565.1, RefSeq NM_014438.5, GENCODE ENST00000259213.9) averaged a pLDDT of only 50.2, a score indicative of near-complete folding failure, while an alternate isoform in CHESS (CHS.30565.4) and RefSeq (XM_011510962.1), shown in <xref ref-type="fig" rid="fig6">Figure 6</xref>, scored 93.0, the largest relative score increase of any isoform we examined. This alternate isoform contains two C-terminal exons that are not present in the MANE isoform and that contribute nearly half of the total protein sequence. A BLASTP homology search of the 34aa coding sequence of the final C-terminal exon in the MANE transcript yielded no hits beyond primates using default search parameters. In contrast, a BLASTP search of sequence unique to the alternate isoform revealed significant similarity (e-values of 0.001 and smaller) to <italic>IL36B</italic> orthologs in 1517 organisms including mouse, rat, and Hawaiian monk seal. In addition, expression of the MANE isoform was observed in only 1 sample in the GTEx data with an expression level of just 0.01 TPM, while the CHESS isoform was found in 775 samples with a much higher expression level of 8.4 TPM (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>).</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>IL36B isoform comparison.</title><p>Comparison of predicted structures for interleukin 36 beta (IL36B) for the Matched Annotation from NCBI and EMBL-EBI (MANE) isoform (CHS.30565.1, RefSeq NM_014438.5, GENCODE ENST00000259213.9) and an alternate isoform from Comprehensive Human Expressed SequenceS (CHESS) and RefSeq (CHS.30565.4, RefSeq XM_011510962.1). The highly conserved protein sequence of the alternate human isoform achieves a very high predicted local distance difference test (pLDDT) score of 93.0, versus the MANE isoform’s much lower pLDDT of 50.2.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig6-v3.tif"/></fig><p>Further probing the functionality of our high-scoring isoform, we aligned the predicted 3D structures for <italic>IL36B</italic> in human, mouse, and rat. As expected, the mouse and rat proteins aligned to each other remarkably well, with an RMSD of 0.60 Å. We found the low-scoring MANE protein aligned poorly to the structures for mouse and rat, averaging a distance of 2.74 Å, while the alternate isoform aligned far better with an RMSD of 0.76 Å. This close similarity in both sequence and structure to conserved orthologs in distant species strongly reinforces the argument that the alternate isoform represents the functional version of the protein in human.</p></sec><sec id="s2-2-5"><title>Post-GPI attachment to proteins 2</title><p>The protein known as post-GPI attachment to proteins 2 (<italic>PGAP2</italic>, alternatively <italic>FRAG1</italic> or <italic>CWH43N</italic>) is required for stable expression of glycosylphosphatidylinositol (GPI)-anchored proteins (<xref ref-type="bibr" rid="bib46">Tashima et al., 2006</xref>) attached to the external cellular plasma membrane via a post-translational modification system ubiquitous in eukaryotes (<xref ref-type="bibr" rid="bib9">Englund, 1993</xref>). Mutations in GPI pathway proteins have been linked to a wide variety of rare genetic disorders (<xref ref-type="bibr" rid="bib2">Bellai-Dussault et al., 2019</xref>), while mutations in <italic>PGAP2</italic> specifically have been shown to cause to intellectual disability, hyperphosphatasia, and petit mal seizures (<xref ref-type="bibr" rid="bib13">Hansen et al., 2013</xref>).</p><p>Out of 85 GTEx-assembled transcripts for <italic>PGAP2</italic> produced during the latest build of the CHESS database, encoding 33 distinct protein isoforms, the single highest scoring isoform according to ColabFold was CHS.7860.59 (RefSeq NM_001256240.2, GENCODE ENST00000463452.6), with a pLDDT of 87.9. The coding sequence of this isoform exactly matches the sequence of the assumed biologically active protein (<xref ref-type="bibr" rid="bib13">Hansen et al., 2013</xref>; <xref ref-type="bibr" rid="bib22">Krawitz et al., 2013</xref>), and all intron boundaries are conserved in mouse. For comparison, the annotated MANE protein (CHS.7860.58, RefSeq NM_014489.4, GENCODE ENST00000278243.9) has a pLDDT of 78.0 and the intron boundaries are not conserved in mouse. Predicted structures of both proteins are shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>.</p><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>PGAP2 isoform comparison.</title><p>Comparison of the structure of the Matched Annotation from NCBI and EMBL-EBI (MANE) isoform (CHS.7860.58, RefSeq NM_014489.4, GENCODE ENST00000278243.9) versus the highest scoring alternate isoform (CHS.7860.59, RefSeq NM_001256240.2, GENCODE ENST00000463452.6) for PGAP2. Of 33 distinct annotated protein isoforms of PGAP2, the one with the highest predicted local distance difference test (pLDDT) represents the biologically active version (<xref ref-type="bibr" rid="bib13">Hansen et al., 2013</xref>; <xref ref-type="bibr" rid="bib22">Krawitz et al., 2013</xref>) of PGAP2 in humans.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig7-v3.tif"/></fig><p>RNA-seq data from GTEx also showed that the higher scoring isoform, CHS.7860.59, was expressed in 8776 samples with an average expression level of 2.6 TPM, compared to only 4116 samples with a 0.9 TPM average for the MANE isoform. Comparing their intron-exon structure revealed that the MANE transcript has one extra exon (the second exon out of six). On average, across all 31 tissues in the GTEx data, five times more spliced reads supported skipping that exon, as in CHS.7860.59, rather than including it.</p></sec></sec><sec id="s2-3"><title>Functional splice variants may not fold well</title><p>Alternative splicing allows genes to code for multiple functional protein products (<xref ref-type="bibr" rid="bib27">Matlin et al., 2005</xref>). Thus, rejecting all but the top-scoring isoform based on predicted structure may eliminate lower-scoring yet functional proteins. The risk of discarding functional transcripts by relying too heavily on the pLDDT score is well-illustrated by vascular endothelial growth factor B (<italic>VEGFB</italic>), a growth factor implicated in cancer and diabetes-related heart disease (<xref ref-type="bibr" rid="bib23">Lal et al., 2018</xref>). The human <italic>VEGFB</italic> gene encodes two well-characterized protein isoforms: <italic>VEGFB-167</italic> and <italic>VEGFB-186</italic>. Alternative splicing that skips part of the sixth exon in <italic>VEGFB-167</italic> leads to sequestration of the protein to the cell surface due to a highly basic C-terminal heparin binding domain. Full inclusion of exon six in <italic>VEGFB-186</italic> results in a soluble protein freely transported to the blood stream (<xref ref-type="bibr" rid="bib24">Li, 2010</xref>).</p><p>Both isoforms shown in <xref ref-type="fig" rid="fig8">Figure 8</xref> represent highly expressed and similarly functional products, containing a well-conserved cysteine-knot motif (<xref ref-type="bibr" rid="bib16">Iyer and Acharya, 2011</xref>), yet <italic>VEGFB-167</italic> receives a pLDDT score of 81.7 while <italic>VEGFB-186</italic> receives a much lower pLDDT score of 69.5. The MANE isoform (CHS.9039.1, RefSeq NM_003377.5, GENCODE ENST00000309422.7) encodes the freely soluble protein <italic>VEGFB-186</italic>, while the alternate isoform (CHS.9039.2, RefSeq NM_001243733.2, GENCODE ENST00000426086.3) encodes the sequestered protein <italic>VEGFB-167</italic>. Additionally, <italic>VEGFB-186</italic> is present as a full-length cDNA clone (MGC:10373 IMAGE:4053976) in the Mammalian Gene Collection (<xref ref-type="bibr" rid="bib47">Temple et al., 2009</xref>), a fact which strongly supports its functionality. Due to the large pLDDT score difference between the two functional <italic>VEGFB</italic> isoforms, a naïve attempt to use protein folding prediction scores as the sole oracle of protein function might inadvertently discard <italic>VEGFB-186</italic>, a clearly functional transcript. Thus, one must be careful to incorporate multiple sources of information when making decisions about isoform functionality.</p><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Vascular endothelial growth factor B (VEGFB) isoform comparison.</title><p>VEGFB isoforms VEGFB-186 (<bold>a</bold>) and VEGFB-167 (<bold>b</bold>). The inclusion of a heparin binding domain in VEGFB-167 results in sequestration to the cell surface while VEGFB-186 remains freely soluble. Relying solely on predicted local distance difference test (pLDDT) comparisons in this case would be misleading, as both isoforms represent well-understood functional protein products.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig8-v3.tif"/></fig></sec><sec id="s2-4"><title>A novel protein-coding transcript in mouse</title><p>While examining the evolutionary conservation of the 3D structure of <italic>TXNDC8</italic> in human, we noticed that the predicted structure for the same gene in <italic>Mus musculus</italic> (house mouse) seemed to contain a poorly folded region similar to CHS.56446.6, a low-scoring human protein. Further inspection of <italic>TXNDC8</italic> in the mouse genome revealed that the primary RefSeq transcript contains a misfolding fifth exon, while the alternate RefSeq transcript skips the misfolding exon, similar to the functional human isoform, shown in <xref ref-type="fig" rid="fig9">Figure 9</xref>. Interestingly, both mouse transcripts contain a third exon homologous to the sequence missing in the human MANE isoform. A BLASTP search of the misfolding exon resulted in only a single significant hit outside the order <italic>Rodentia</italic> to an unnamed protein product. In contrast, a BLASTP search of the third exon present in all mouse transcripts, homologous to the missing exon in the human MANE transcript, revealed significant hits to at least 31 homologs outside <italic>Rodentia</italic>.</p><fig id="fig9" position="float"><label>Figure 9.</label><caption><title>TXNDC8 human and mouse comparison.</title><p>Intron-exon and predicted protein structure for TXNDC8 in human (<bold>a and b</bold>) and mouse (<bold>c</bold>, <bold>d</bold>, and <bold>e</bold>). Exons are colored according to their average predicted local distance difference test (pLDDT) score. The highest-scoring isoforms in both human (<bold>b</bold>) and mouse (<bold>c</bold>) share conserved intron-exon structure and nearly identical predicted protein structure.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-82556-fig9-v3.tif"/></fig><p>Unlike the functional human isoform, however, the RefSeq alternate mouse transcript is annotated with a start site 26 codons upstream of the translation initiation site annotated in the human orthologs. A BLASTP search of the 26 amino acid additional sequence resulted in zero significant hits outside <italic>Rodentia</italic>. Folding this alternate mouse transcript revealed that these additional N-terminal amino acids fail to form any confident predicted protein structure. As a result, none of the three transcripts annotated in mouse fold into the highly conserved structure of <italic>TXNDC8</italic> in human. Predicted structures and exon alignments for isoforms in human and mouse are shown in <xref ref-type="fig" rid="fig9">Figure 9</xref>.</p><p>We hypothesized that a truly functional isoform of <italic>TXNDC8</italic> in mouse should be similar to the human isoform in 3D structure. Based on our observations, this similarity might be realized if the mouse alternate isoform simply started at the downstream start site that matches human, yielding a 105aa protein rather than the 131aa protein that is currently annotated. Folding the coding sequence of the alternate mouse transcript, minus the 26 N-terminal amino acid residues, revealed a predicted structure remarkably similar to both human and cattle <italic>TXNDC8</italic>. <xref ref-type="fig" rid="fig5">Figure 5</xref> shows the 3D alignment of the human and cow proteins to our predicted (105aa) isoform of mouse <italic>TXNDC8</italic>. Remarkably, the predicted average RMSD between aligned heavy atoms of the putative human and mouse proteins is just 0.83 Å. For reference, the atomic diameter of one carbon atom is 1.4 Å.</p><p>In a further investigation of mouse <italic>TXNDC8</italic> transcription, we aligned 8 gigabases of RNA-seq cDNA from a mouse testis sample (SRR18337982) to the GRCm38 reference genome using HISAT2 (<xref ref-type="bibr" rid="bib20">Kim et al., 2019</xref>) then assembled transcripts using StringTie2 (<xref ref-type="bibr" rid="bib21">Kovaka et al., 2019</xref>). This resulted in two putative transcripts at the mouse <italic>TXNDC8</italic> locus, with neither transcript containing the upstream start site present in the RefSeq annotation. Examination of the read coverage confirmed that both putative <italic>TXNDC8</italic> transcripts appear to use the start site of our proposed shorter protein-coding sequence, with 1060 reads supporting the canonical start site and zero reads supporting the upstream start site. As hypothesized, one of these newly assembled transcripts contained the protein-coding sequence necessary to exactly match our predicted structure-conserved isoform.</p><p>We believe this represents the first experimentally confirmed novel isoform in any organism discovered due to a hypothesis derived from comparison of computationally predicted protein structures. All in all, the structure-guided identification and subsequent experimental confirmation of a novel functional <italic>TXNDC8</italic> isoform in mouse demonstrates the potential of 3D protein structure prediction to enhance functional annotation in any genome.</p></sec><sec id="s2-5"><title>A resource for human annotation</title><p>The examples discussed here are only a small subset of the 130,700 unique protein structures we generated at 20,817 human gene loci. These structures and associated pLDDT scores have already been used to avoid filtering out transcripts encoding functional, clinically relevant isoforms while building the latest version of the CHESS human gene catalog. We provide all of these structures as a searchable and downloadable database, at <ext-link ext-link-type="uri" xlink:href="https://isoform.io/">isoform.io</ext-link>, to create a public resource for improving the annotation of the human genome. The large majority of CHESS isoforms in this collection have direct support from RNA-seq data, as they were assembled from the large GTEx collection, a high-quality set of deep RNA-seq experiments across dozens of human tissues. A very small number of MANE proteins were not assembled from GTEx data, but structures of these too are included in this analysis so that no MANE genes would be omitted.</p><p>In the online resource, we provide for each transcript: (1) the nucleotide and amino acid sequences; (2) the predicted structure, as a file that can be viewed in a standard structure viewer such as PyMOL (<xref ref-type="bibr" rid="bib43">Schrödinger, 2015</xref>) (3) the pLDDT score of that structure; (4) the length of the isoform; (5) the number of GTEx samples in which the isoform was observed; (6) the maximum expression of the isoform in any tissue; (7) an indicator based on alignment of whether all introns are conserved in the mouse genome; (8) an interactive table with functionality to search and sort transcripts to find isoforms of interest; and (9) a Foldseek (<xref ref-type="bibr" rid="bib51">van Kempen et al., 2022</xref>) interface to search any given protein structure against all 237,295 transcripts presented here. These predictions can be mined to discover, for example, cases where a known protein gets a surprisingly low pLDDT score, or where alternative isoforms have structures that get higher scores and appear more stable than previously reported forms of the same protein.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>In this analysis, we demonstrated the ability to improve protein-coding gene annotation by predicting 3D protein structures. We searched tens of thousands of predicted structures of alternate isoforms of human genes and identified a subset that appear to fold more confidently than the isoforms found in MANE, a recently developed ‘universal standard’ for human gene annotation (<xref ref-type="bibr" rid="bib31">Morales et al., 2022</xref>). We found hundreds of gene isoforms, all of which were supported by RNA-seq data, that outscored the corresponding MANE transcript. Inclusion of truly functional protein isoforms in future releases of human gene catalogs, particularly clinically focused catalogs such as MANE, may enable more accurate downstream analyses of these genes.</p><p>In the illustratory examples described here, we provide biological and evolutionary context for cases where an alternate human isoform appears clearly superior in structure to its canonical protein. Given the many additional high-scoring transcripts that we identified (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>), we expect further improvements in human annotation are yet to be discovered. More generally, we followed a structure-guided annotation strategy that may prove useful in refining the annotation of many non-human species as well.</p><p>We expect computational protein structure prediction to become an indispensable tool for future transcriptome annotation efforts. Still, the functionality of many proteins may not be revealed by structure prediction alone. Cases where substantial portions of a protein fail to form a stable structure, such as intrinsically disordered proteins, were not examined here. Additionally, non-functional isoforms may achieve a higher pLDDT simply due to the omission of small, yet functionally important intrinsically disordered regions. In these cases, it is necessary to contextualize results with both expression data, when available, and evolutionary sequence conservation analysis. If a low-scoring region is highly conserved across species or is consistently expressed, this still provides a strong indication of function regardless of foldability.</p><p>Although we restricted our analysis to whole-protein comparisons, comparing local structural portions of a protein, potentially near shape-sensitive ligand binding sites (<xref ref-type="bibr" rid="bib11">Greer et al., 1994</xref>), may enable similar analysis in these proteins. Further advances in predicting structures for multi-chain protein complexes (<xref ref-type="bibr" rid="bib10">Evans et al., 2022</xref>), as well as improvements in prediction efficiency in large proteins, may expand the range of genes that may be analyzed. An important caveat is that in some cases, truly functional isoforms may receive low predicted folding scores relative to well-folded functional alternate isoforms within the same gene. Thus, structure prediction alone is not always sufficient to make functional claims about any individual protein isoform.</p><sec id="s3-1"><title>Ideas and speculation</title><p>Though the complete sequence of the human genome has been revealed (<xref ref-type="bibr" rid="bib33">Nurk et al., 2022</xref>), the annotation of the human genome, by far the most comprehensively studied, remains far from finished. The use of accurate predicted protein structures for gene annotation, as we have done here, represents a new paradigm, not only for human gene discovery but for all other species as well. For decades, the scientific community has relied principally on two methods to discover and validate protein-coding genes at the genome scale: sequencing of transcripts (or cDNAs), and alignment of DNA and protein sequences to detect evolutionary conservation in other species. Protein structure holds valuable information regarding biological functionality, providing an independent and powerful tool to complement these methods. The analysis described here takes a first step toward improving genome annotation of humans using structure prediction, but specific methods deploying this powerful new tool on a broader scale will require fundamentally new computational strategies. As such, development of comprehensive genome annotation protocols incorporating protein structure prediction will remain an area of active investigation for years to come.</p></sec></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Protein structure prediction</title><p>We folded all transcripts in the CHESS annotation less than 1000 amino acids in length. Similar to the initial effort to fold the human proteome (<xref ref-type="bibr" rid="bib49">Tunyasuvunakool et al., 2021</xref>), the length limit was chosen to make the overall computational runtime feasible. This yielded 233,973 transcripts representing 127,398 unique protein-coding sequences at 20,666 loci. Coding sequences for CHESS transcripts were determined with ORFanage (<xref ref-type="bibr" rid="bib52">Varabyou et al., 2021</xref>). For each protein sequence, we generated a multiple sequence alignment by aligning them with ColabFold’s MMseqs2 (<xref ref-type="bibr" rid="bib44">Steinegger and Söding, 2017</xref>) workflow (colabfold_search) against the UniRef100 (<xref ref-type="bibr" rid="bib45">Suzek et al., 2015</xref>) (2021/03) and ColabFoldDB (2021/08) database. Structure predictions were made with ColabFold (commit 3398d3) using AlphaFold2 and MMseqs2 version 13.45111. To speed up the search, we set the sensitivity setting to 7 (-s 7). We predicted each structure using colabfold_batch and stopped the process early if a pLDDT of at least 85 was reached by any model (--stop-at-score 85) or if a model produced a pLDDT less than 70 (--stop-at-score-below 70). All models were ranked by pLDDT in descending order. Runtime was estimated from a sample of 500 proteins randomly selected from the 127,398 structures. Prediction of all structures on 8 × A5000 GPUs required 34 days. Multiple sequence alignment took 34 hr using MMseqs2 on an AMD EPYC 7742 CPU with 64 cores.</p></sec><sec id="s4-2"><title>Filtering MANE comparisons</title><p>To generate the 940 protein isoforms in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>, we used the following filtering criteria and procedures. For each isoform with a distinct coding sequence located at a MANE v1.0 locus and with coding sequence overlapping a MANE annotated protein, we compared the pLDDT score to that of the associated MANE protein. As described previously, pLDDT is a reliable measure of the confidence in a structure, where predictions with 70 ≤ pLDDT ≤ 90 are confident, those with pLDDT &gt;90 are highly confident, and those below 50 represent low-confidence structures and may be disordered proteins (<xref ref-type="bibr" rid="bib53">Varadi et al., 2022</xref>). We only considered alternative isoforms that had a pLDDT score ≥70, indicating a generally well-folded protein, to avoid including any intrinsically disordered proteins (<xref ref-type="bibr" rid="bib41">Ruff and Pappu, 2021</xref>). Filtering and general analysis was performed in Colab (<ext-link ext-link-type="uri" xlink:href="https://colab.research.google.com">https://colab.research.google.com</ext-link>) with Python version 3.8.15.</p><p>We selected isoforms that, when compared to the MANE isoform for the same gene, scored at least 5% higher as measured by pLDDT and were at least 90% as long. Additionally, to capture cases where the MANE transcript might be missing functional sequence elements, we selected alternate isoforms that were at least 5% longer than the MANE isoform, that had equal or higher pLDDT scores, and that were assembled in an equal or higher number of GTEx samples. Finally, to capture cases where an alternate isoform might be functional despite being substantially shorter than the MANE protein, we selected isoforms at least 50% as long as the MANE protein where the alternate isoform scored at least 5% higher and was assembled in an equal or higher number of GTEx samples. After applying these filters, we observed that in some cases, a processed pseudogene (<xref ref-type="bibr" rid="bib56">Zhang et al., 2002</xref>) contained within an intron outscored the associated primary transcript. To eliminate such cases, we used GFFcompare (<xref ref-type="bibr" rid="bib38">Pertea and Pertea, 2020</xref>) to ensure that isoforms overlapped their MANE transcript’s coding sequence. When multiple alternate isoforms contained the same coding sequence, and thus received the same pLDDT score, we selected the isoform assembled in the highest number of GTEx samples.</p></sec><sec id="s4-3"><title>Annotation sources</title><p>Transcript annotations for MANE, CHESS, RefSeq, and GENCODE were retrieved from the following sources. The MANE v1.0 database was downloaded from NCBI at <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/refseq/MANE">https://www.ncbi.nlm.nih.gov/refseq/MANE</ext-link>. Annotations from the CHESS v3.0 database were retrieved from <ext-link ext-link-type="uri" xlink:href="http://ccb.jhu.edu/chess">http://ccb.jhu.edu/chess</ext-link>. Additional CHESS annotations came from an unpublished set of transcript assemblies created as part of the process of building CHESS v3.0; these were assembled from approximately 10,000 GTEx RNA-seq experiments across 31 tissues using StringTie2 (<xref ref-type="bibr" rid="bib21">Kovaka et al., 2019</xref>). Transcripts from this set were given a CHESS ID starting in ‘hypothetical’ if the locus was missing from CHESS v3.0, or else given an ID ending in ‘altN’ if the locus was present in CHESS v3.0 but the exact isoform was not. Note that many of these, particularly those with a poor protein folding score, were not retained in the final CHESS v3.0 database. RefSeq annotations (releases 109 and 110) were downloaded from <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml">https://www.ncbi.nlm.nih.gov/projects/genome/guide/human/index.shtml</ext-link>. GENCODE (v38, v39, v40) annotations were collected from <ext-link ext-link-type="uri" xlink:href="https://www.gencodegenes.org/human/">https://www.gencodegenes.org/human/</ext-link>.</p></sec><sec id="s4-4"><title>Visualization and atomic alignment</title><p>All visualizations and 3D protein structure atomic alignments were performed in PyMol (<xref ref-type="bibr" rid="bib43">Schrödinger, 2015</xref>) version 2.5.2 using non-orthoscopic view, white background, and ray trace 1200,1200. RMSD were calculated without excluding any outliers. Ramachandran plots were created using PyRAMA version 2.0.2 with Richardson (<xref ref-type="bibr" rid="bib26">Lovell et al., 2003</xref>) standard psi and phi values for all amino acids excluding glycine and proline. Intron-exon structure plots were produced with MISO (<xref ref-type="bibr" rid="bib19">Katz et al., 2010</xref>) (commit b714021) and TieBrush (<xref ref-type="bibr" rid="bib52">Varabyou et al., 2021</xref>) (commit e986d64).</p></sec><sec id="s4-5"><title>RNA-seq quantification of the human ASMT gene</title><p>RNA-seq data were downloaded from NCBI for run SRR5756467 from BioSample SAMN07278516, a pineal gland from a patient who died at midnight. A detailed summary of the experimental protocol used to generate these data can be found in NCBI BioProject PRJNA391921. Isoform-level quantification was performed using Salmon (<xref ref-type="bibr" rid="bib36">Patro et al., 2017</xref>) version 1.8.0.</p></sec><sec id="s4-6"><title>RNA-seq assembly of mouse TXNDC8</title><p>RNA-seq data were downloaded for run SRR18337982 from BioSample SAMN26725167, a tissue sample from the testis of a control mouse. A detailed summary of the experimental protocol used to generate these data can be found in NCBI BioProject PRJNA816862. cDNA reads were aligned to the mm39 reference genome using HISAT2 (<xref ref-type="bibr" rid="bib20">Kim et al., 2019</xref>) version 2.1.0 then assembled into transcripts using StringTie2 (<xref ref-type="bibr" rid="bib21">Kovaka et al., 2019</xref>) version 2.2.1.</p></sec><sec id="s4-7"><title>Intron conservation in human and mouse</title><p>We assessed the conservation of GT-AG intron boundaries between CHESS human transcripts and transcripts from the GRCm38 mouse reference genome. Data for the human-mouse alignment was extracted from a 30-species alignment anchored on GRCh38 that was downloaded from the UCSC genome browser (<xref ref-type="bibr" rid="bib32">Navarro Gonzalez et al., 2021</xref>). We used MafIO in BioPython (<xref ref-type="bibr" rid="bib6">Cock et al., 2009</xref>) version 1.71 to check if all intron boundaries were conserved between mouse and human transcripts. In the supplementary files, a value of ‘TRUE’ in the ‘introns in mouse’ column indicates that all boundaries were conserved, while ‘FALSE’ means that at least one splice site (either a GT at a donor site or an AG at an acceptor site) was not conserved in the alignment.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing – original draft, Project administration, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Data curation, Software, Investigation</p></fn><fn fn-type="con" id="con3"><p>Data curation, Software, Visualization</p></fn><fn fn-type="con" id="con4"><p>Visualization</p></fn><fn fn-type="con" id="con5"><p>Data curation, Software, Investigation</p></fn><fn fn-type="con" id="con6"><p>Data curation, Software, Investigation</p></fn><fn fn-type="con" id="con7"><p>Funding acquisition, Validation, Investigation, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con8"><p>Conceptualization, Resources, Supervision, Funding acquisition, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con9"><p>Conceptualization, Resources, Formal analysis, Supervision, Funding acquisition, Validation, Investigation, Methodology, Writing – original draft, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>All isoform summary.</title><p>Folding scores from ColabFold for each transcript from a preliminary new build of the Comprehensive Human Expressed SequenceS (CHESS) database that contained a protein-coding sequence (CDS) that was under 1000aa in length. For transcripts already contained in the released CHESS v3.0 database, the identifier from that database is provided. If the transcript maps to a known gene locus X but is a novel isoform, it is shown with the identifier CHS.X.altY. If a transcript occurs at a novel locus X, the identifier is hypothetical.X.Y, where Y identifies the isoform number. Additional columns show the gene name, the RefSeq ID (release 110), the GENCODE ID (release 40), the predicted local distance difference test (pLDDT) (folding) score, and a flag indicating whether all intron boundaries (for multi-exon genes) are conserved in the mouse genome.</p></caption><media xlink:href="elife-82556-supp1-v3.csv" mimetype="application" mime-subtype="octet-stream"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Matched Annotation from NCBI and EMBL-EBI (MANE) comparison summary.</title><p>Folding scores and additional data for all Comprehensive Human Expressed SequenceS (CHESS) transcripts that match genes in the MANE v1.0 dataset, limited to protein sequences under 1000aa in length. Transcripts must overlap the annotated CDS of the MANE transcript to be included. Columns include: <italic>CHESS_ID_isoform</italic>, the CHESS identifier of the alternate isoform transcript; <italic>CHESS_ID_MANE</italic>, the CHESS identifier of the MANE transcript at the same locus; <italic>gene</italic>, the gene name; <italic>aa_length_isoform</italic>, the amino acid length of the alternate isoform’s CDS; <italic>aa_length_MANE</italic>, the amino acid length of the MANE transcript’s CDS; <italic>length_ratio</italic>, the ratio of the alternate isoform length to the MANE isoform length; <italic>pLDDT_isoform</italic>, the predicted folding score of the alternate isoform; <italic>pLDDT_MANE</italic>, the predicted folding score of the MANE isoform; <italic>pLDDT_ratio</italic>, the ratio of the alternate isoform folding score to the MANE isoform folding score; <italic>GTEx_samples_observed_isoform</italic>, the total number of GTEx samples where the alternate isoform was observed at least once; <italic>GTEx_samples_observed_MANE,</italic> the total number of GTEx samples where the MANE isoform was observed at least once; <italic>GTEx_top_tissue_name_isoform</italic>, the name of the tissue in which the alternate isoform was observed in the highest number of samples; <italic>GTEx_top_tissue_name_MANE</italic>, the name of the tissue in which the MANE isoform was observed in the highest number of samples; <italic>GTEx_top_tissue_TPM_isoform</italic>, the average TPM of the alternate isoform in the named tissue; <italic>GTEx_top_tissue_TPM_MANE</italic>, the observed transcripts per million (TPM) of the MANE isoform in the named tissue; <italic>introns_conserved_in_mouse_isoform</italic>, an indicator of whether introns are conserved between the alternate human isoform and any annotated isoform in the GRCm38 mouse reference genome; <italic>introns_conserved_in_mouse_MANE</italic>, an indicator of whether introns are conserved between the MANE human isoform and any annotated isoform in the GRCm38 mouse reference genome.</p></caption><media xlink:href="elife-82556-supp2-v3.csv" mimetype="application" mime-subtype="octet-stream"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Matched Annotation from NCBI and EMBL-EBI (MANE) comparison summary, filtered subset.</title><p>A filtered set of Comprehensive Human Expressed SequenceS (CHESS) transcripts compared to MANE according to the criteria detailed in the ‘Filtering MANE comparisons’ section of the Materials and methods. Uses the same column names as <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>.</p></caption><media xlink:href="elife-82556-supp3-v3.csv" mimetype="application" mime-subtype="octet-stream"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-82556-mdarchecklist1-v3.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Gene identifiers for all predicted protein isoforms as well as pLDDT scores and evolutionary conservation data from mouse can be found in Supplementary file 1. Predicted scores and GTEx expression data for all isoforms overlapping a MANE locus can be found in Supplementary file 2. Data for the 940 alternate isoforms with evidence of relatively superior structure, and possibly superior function, can be found in Supplementary file 3. Additionally, all data can be downloaded from Figshare (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.21802476.v1">https://doi.org/10.6084/m9.figshare.21802476.v1</ext-link>).</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Sommer</surname><given-names>M</given-names></name><name><surname>Cha</surname><given-names>S</given-names></name><name><surname>Varabyou</surname><given-names>A</given-names></name><name><surname>Rincon</surname><given-names>N</given-names></name><name><surname>Park</surname><given-names>S</given-names></name><name><surname>Minkin</surname><given-names>I</given-names></name><name><surname>Pertea</surname><given-names>M</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Salzberg</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Structure-guided isoform identification for the human transcriptome</data-title><source>Figshare</source><pub-id pub-id-type="doi">10.6084/m9.figshare.21802476.v1</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><name><surname>Pertea</surname><given-names>M</given-names></name><name><surname>Salzberg</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>CHESS 3.0</data-title><source>CHESS</source><pub-id pub-id-type="accession" xlink:href="http://ccb.jhu.edu/chess/">2.2</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset3"><person-group person-group-type="author"><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2013">2013</year><data-title>The Genotype-Tissue Expression (GTEx) project</data-title><source>GTEx</source><pub-id pub-id-type="accession" xlink:href="https://gtexportal.org/home/">V8</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset4"><person-group person-group-type="author"><name><surname>Morales</surname><given-names>J</given-names></name><name><surname>Pujar</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Matched Annotation from NCBI and EMBL-EBI (MANE)</data-title><source>NCBI RefSeq</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/refseq/MANE/">MANE</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset5"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>A multi-species multi-timepoint transcriptome database and webpage for the pineal gland and retina</data-title><source>NCBI Sequence Read Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra/SRR5756467">SRR5756467</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset6"><person-group person-group-type="author"><collab>Wang et al.</collab></person-group><year iso-8601-date="2022">2022</year><data-title>Tes45; <italic>Mus musculus</italic></data-title><source>NCBI Sequence Read Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/sra/20655315/SRR18337982">SRR18337982</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>The authors would like to thank all members of the Salzberg, Pertea, and Steinegger labs, as well as David J Lipman for helpful feedback during project conceptualization, Benjamin Langmead for publicly hosting bulk data files, and Do-Yoon Kim for creating the <ext-link ext-link-type="uri" xlink:href="https://isoform.io/">isoform.io</ext-link> logo.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Andley</surname><given-names>UP</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Crystallins in the eye: function and pathology</article-title><source>Progress in Retinal and Eye Research</source><volume>26</volume><fpage>78</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1016/j.preteyeres.2006.10.003</pub-id><pub-id pub-id-type="pmid">17166758</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bellai-Dussault</surname><given-names>K</given-names></name><name><surname>Nguyen</surname><given-names>TTM</given-names></name><name><surname>Baratang</surname><given-names>NV</given-names></name><name><surname>Jimenez-Cruz</surname><given-names>DA</given-names></name><name><surname>Campeau</surname><given-names>PM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Clinical variability in inherited glycosylphosphatidylinositol deficiency disorders</article-title><source>Clinical Genetics</source><volume>95</volume><fpage>112</fpage><lpage>121</lpage><pub-id pub-id-type="doi">10.1111/cge.13425</pub-id><pub-id pub-id-type="pmid">30054924</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Botros</surname><given-names>HG</given-names></name><name><surname>Legrand</surname><given-names>P</given-names></name><name><surname>Pagan</surname><given-names>C</given-names></name><name><surname>Bondet</surname><given-names>V</given-names></name><name><surname>Weber</surname><given-names>P</given-names></name><name><surname>Ben-Abdallah</surname><given-names>M</given-names></name><name><surname>Lemière</surname><given-names>N</given-names></name><name><surname>Huguet</surname><given-names>G</given-names></name><name><surname>Bellalou</surname><given-names>J</given-names></name><name><surname>Maronde</surname><given-names>E</given-names></name><name><surname>Beguin</surname><given-names>P</given-names></name><name><surname>Haouz</surname><given-names>A</given-names></name><name><surname>Shepard</surname><given-names>W</given-names></name><name><surname>Bourgeron</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Crystal structure and functional mapping of human ASMT, the last enzyme of the melatonin synthesis pathway</article-title><source>Journal of Pineal Research</source><volume>54</volume><fpage>46</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1111/j.1600-079X.2012.01020.x</pub-id><pub-id pub-id-type="pmid">22775292</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Carrier</surname><given-names>Y</given-names></name><name><surname>Ma</surname><given-names>H-L</given-names></name><name><surname>Ramon</surname><given-names>HE</given-names></name><name><surname>Napierata</surname><given-names>L</given-names></name><name><surname>Small</surname><given-names>C</given-names></name><name><surname>O’Toole</surname><given-names>M</given-names></name><name><surname>Young</surname><given-names>DA</given-names></name><name><surname>Fouser</surname><given-names>LA</given-names></name><name><surname>Nickerson-Nutter</surname><given-names>C</given-names></name><name><surname>Collins</surname><given-names>M</given-names></name><name><surname>Dunussi-Joannopoulos</surname><given-names>K</given-names></name><name><surname>Medley</surname><given-names>QG</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Inter-regulation of Th17 cytokines and the IL-36 cytokines in vitro and in vivo: implications in psoriasis pathogenesis</article-title><source>The Journal of Investigative Dermatology</source><volume>131</volume><fpage>2428</fpage><lpage>2437</lpage><pub-id pub-id-type="doi">10.1038/jid.2011.234</pub-id><pub-id pub-id-type="pmid">21881584</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>E</given-names></name><name><surname>Fu</surname><given-names>C</given-names></name><name><surname>Coon</surname><given-names>SL</given-names></name><name><surname>Alon</surname><given-names>S</given-names></name><name><surname>Bozinoski</surname><given-names>M</given-names></name><name><surname>Breymaier</surname><given-names>M</given-names></name><name><surname>Bustos</surname><given-names>DM</given-names></name><name><surname>Clokie</surname><given-names>SJ</given-names></name><name><surname>Gothilf</surname><given-names>Y</given-names></name><name><surname>Esnault</surname><given-names>C</given-names></name><name><surname>Michael Iuvone</surname><given-names>P</given-names></name><name><surname>Mason</surname><given-names>CE</given-names></name><name><surname>Ochocinska</surname><given-names>MJ</given-names></name><name><surname>Tovin</surname><given-names>A</given-names></name><name><surname>Wang</surname><given-names>C</given-names></name><name><surname>Xu</surname><given-names>P</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Dale</surname><given-names>R</given-names></name><name><surname>Klein</surname><given-names>DC</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Resource: a multi-species multi-timepoint transcriptome database and webpage for the pineal gland and retina</article-title><source>Journal of Pineal Research</source><volume>69</volume><elocation-id>e12673</elocation-id><pub-id pub-id-type="doi">10.1111/jpi.12673</pub-id><pub-id pub-id-type="pmid">32533862</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cock</surname><given-names>PJA</given-names></name><name><surname>Antao</surname><given-names>T</given-names></name><name><surname>Chang</surname><given-names>JT</given-names></name><name><surname>Chapman</surname><given-names>BA</given-names></name><name><surname>Cox</surname><given-names>CJ</given-names></name><name><surname>Dalke</surname><given-names>A</given-names></name><name><surname>Friedberg</surname><given-names>I</given-names></name><name><surname>Hamelryck</surname><given-names>T</given-names></name><name><surname>Kauff</surname><given-names>F</given-names></name><name><surname>Wilczynski</surname><given-names>B</given-names></name><name><surname>de Hoon</surname><given-names>MJL</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Biopython: freely available python tools for computational molecular biology and bioinformatics</article-title><source>Bioinformatics</source><volume>25</volume><fpage>1422</fpage><lpage>1423</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp163</pub-id><pub-id pub-id-type="pmid">19304878</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Deiana</surname><given-names>A</given-names></name><name><surname>Forcelloni</surname><given-names>S</given-names></name><name><surname>Porrello</surname><given-names>A</given-names></name><name><surname>Giansanti</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Intrinsically disordered proteins and structured proteins with intrinsically disordered regions have different functional roles in the cell</article-title><source>PLOS ONE</source><volume>14</volume><elocation-id>e0217889</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0217889</pub-id><pub-id pub-id-type="pmid">31425549</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eling</surname><given-names>N</given-names></name><name><surname>Morgan</surname><given-names>MD</given-names></name><name><surname>Marioni</surname><given-names>JC</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Challenges in measuring and understanding biological noise</article-title><source>Nature Reviews. Genetics</source><volume>20</volume><fpage>536</fpage><lpage>548</lpage><pub-id pub-id-type="doi">10.1038/s41576-019-0130-6</pub-id><pub-id pub-id-type="pmid">31114032</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Englund</surname><given-names>PT</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>The structure and biosynthesis of glycosyl phosphatidylinositol protein anchors</article-title><source>Annual Review of Biochemistry</source><volume>62</volume><fpage>121</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.1146/annurev.bi.62.070193.001005</pub-id><pub-id pub-id-type="pmid">8352586</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>O’Neill</surname><given-names>M</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Antropova</surname><given-names>N</given-names></name><name><surname>Senior</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Blackwell</surname><given-names>S</given-names></name><name><surname>Yim</surname><given-names>J</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Protein Complex Prediction with AlphaFold-Multimer</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.10.04.463034</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname><given-names>J</given-names></name><name><surname>Erickson</surname><given-names>JW</given-names></name><name><surname>Baldwin</surname><given-names>JJ</given-names></name><name><surname>Varney</surname><given-names>MD</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Application of the three-dimensional structures of protein target molecules in structure-based drug design</article-title><source>Journal of Medicinal Chemistry</source><volume>37</volume><fpage>1035</fpage><lpage>1054</lpage><pub-id pub-id-type="doi">10.1021/jm00034a001</pub-id><pub-id pub-id-type="pmid">8164249</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2013">2013</year><article-title>The genotype-tissue expression (gtex) project</article-title><source>Nature Genetics</source><volume>45</volume><fpage>580</fpage><lpage>585</lpage><pub-id pub-id-type="doi">10.1038/ng.2653</pub-id><pub-id pub-id-type="pmid">23715323</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hansen</surname><given-names>L</given-names></name><name><surname>Tawamie</surname><given-names>H</given-names></name><name><surname>Murakami</surname><given-names>Y</given-names></name><name><surname>Mang</surname><given-names>Y</given-names></name><name><surname>ur Rehman</surname><given-names>S</given-names></name><name><surname>Buchert</surname><given-names>R</given-names></name><name><surname>Schaffer</surname><given-names>S</given-names></name><name><surname>Muhammad</surname><given-names>S</given-names></name><name><surname>Bak</surname><given-names>M</given-names></name><name><surname>Nöthen</surname><given-names>MM</given-names></name><name><surname>Bennett</surname><given-names>EP</given-names></name><name><surname>Maeda</surname><given-names>Y</given-names></name><name><surname>Aigner</surname><given-names>M</given-names></name><name><surname>Reis</surname><given-names>A</given-names></name><name><surname>Kinoshita</surname><given-names>T</given-names></name><name><surname>Tommerup</surname><given-names>N</given-names></name><name><surname>Baig</surname><given-names>SM</given-names></name><name><surname>Abou Jamra</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Hypomorphic mutations in PGAP2, encoding a GPI-anchor-remodeling protein, cause autosomal-recessive intellectual disability</article-title><source>American Journal of Human Genetics</source><volume>92</volume><fpage>575</fpage><lpage>583</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2013.03.008</pub-id><pub-id pub-id-type="pmid">23561846</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harrow</surname><given-names>J</given-names></name><name><surname>Frankish</surname><given-names>A</given-names></name><name><surname>Gonzalez</surname><given-names>JM</given-names></name><name><surname>Tapanari</surname><given-names>E</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Kokocinski</surname><given-names>F</given-names></name><name><surname>Aken</surname><given-names>BL</given-names></name><name><surname>Barrell</surname><given-names>D</given-names></name><name><surname>Zadissa</surname><given-names>A</given-names></name><name><surname>Searle</surname><given-names>S</given-names></name><name><surname>Barnes</surname><given-names>I</given-names></name><name><surname>Bignell</surname><given-names>A</given-names></name><name><surname>Boychenko</surname><given-names>V</given-names></name><name><surname>Hunt</surname><given-names>T</given-names></name><name><surname>Kay</surname><given-names>M</given-names></name><name><surname>Mukherjee</surname><given-names>G</given-names></name><name><surname>Rajan</surname><given-names>J</given-names></name><name><surname>Despacio-Reyes</surname><given-names>G</given-names></name><name><surname>Saunders</surname><given-names>G</given-names></name><name><surname>Steward</surname><given-names>C</given-names></name><name><surname>Harte</surname><given-names>R</given-names></name><name><surname>Lin</surname><given-names>M</given-names></name><name><surname>Howald</surname><given-names>C</given-names></name><name><surname>Tanzer</surname><given-names>A</given-names></name><name><surname>Derrien</surname><given-names>T</given-names></name><name><surname>Chrast</surname><given-names>J</given-names></name><name><surname>Walters</surname><given-names>N</given-names></name><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Pei</surname><given-names>B</given-names></name><name><surname>Tress</surname><given-names>M</given-names></name><name><surname>Rodriguez</surname><given-names>JM</given-names></name><name><surname>Ezkurdia</surname><given-names>I</given-names></name><name><surname>van Baren</surname><given-names>J</given-names></name><name><surname>Brent</surname><given-names>M</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Valencia</surname><given-names>A</given-names></name><name><surname>Reymond</surname><given-names>A</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name><name><surname>Guigó</surname><given-names>R</given-names></name><name><surname>Hubbard</surname><given-names>TJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>GENCODE: the reference human genome annotation for the ENCODE project</article-title><source>Genome Research</source><volume>22</volume><fpage>1760</fpage><lpage>1774</lpage><pub-id pub-id-type="doi">10.1101/gr.135350.111</pub-id><pub-id pub-id-type="pmid">22955987</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Howe</surname><given-names>KL</given-names></name><name><surname>Achuthan</surname><given-names>P</given-names></name><name><surname>Allen</surname><given-names>J</given-names></name><name><surname>Allen</surname><given-names>J</given-names></name><name><surname>Alvarez-Jarreta</surname><given-names>J</given-names></name><name><surname>Amode</surname><given-names>MR</given-names></name><name><surname>Armean</surname><given-names>IM</given-names></name><name><surname>Azov</surname><given-names>AG</given-names></name><name><surname>Bennett</surname><given-names>R</given-names></name><name><surname>Bhai</surname><given-names>J</given-names></name><name><surname>Billis</surname><given-names>K</given-names></name><name><surname>Boddu</surname><given-names>S</given-names></name><name><surname>Charkhchi</surname><given-names>M</given-names></name><name><surname>Cummins</surname><given-names>C</given-names></name><name><surname>Da Rin Fioretto</surname><given-names>L</given-names></name><name><surname>Davidson</surname><given-names>C</given-names></name><name><surname>Dodiya</surname><given-names>K</given-names></name><name><surname>El Houdaigui</surname><given-names>B</given-names></name><name><surname>Fatima</surname><given-names>R</given-names></name><name><surname>Gall</surname><given-names>A</given-names></name><name><surname>Garcia Giron</surname><given-names>C</given-names></name><name><surname>Grego</surname><given-names>T</given-names></name><name><surname>Guijarro-Clarke</surname><given-names>C</given-names></name><name><surname>Haggerty</surname><given-names>L</given-names></name><name><surname>Hemrom</surname><given-names>A</given-names></name><name><surname>Hourlier</surname><given-names>T</given-names></name><name><surname>Izuogu</surname><given-names>OG</given-names></name><name><surname>Juettemann</surname><given-names>T</given-names></name><name><surname>Kaikala</surname><given-names>V</given-names></name><name><surname>Kay</surname><given-names>M</given-names></name><name><surname>Lavidas</surname><given-names>I</given-names></name><name><surname>Le</surname><given-names>T</given-names></name><name><surname>Lemos</surname><given-names>D</given-names></name><name><surname>Gonzalez Martinez</surname><given-names>J</given-names></name><name><surname>Marugán</surname><given-names>JC</given-names></name><name><surname>Maurel</surname><given-names>T</given-names></name><name><surname>McMahon</surname><given-names>AC</given-names></name><name><surname>Mohanan</surname><given-names>S</given-names></name><name><surname>Moore</surname><given-names>B</given-names></name><name><surname>Muffato</surname><given-names>M</given-names></name><name><surname>Oheh</surname><given-names>DN</given-names></name><name><surname>Paraschas</surname><given-names>D</given-names></name><name><surname>Parker</surname><given-names>A</given-names></name><name><surname>Parton</surname><given-names>A</given-names></name><name><surname>Prosovetskaia</surname><given-names>I</given-names></name><name><surname>Sakthivel</surname><given-names>MP</given-names></name><name><surname>Salam</surname><given-names>AIA</given-names></name><name><surname>Schmitt</surname><given-names>BM</given-names></name><name><surname>Schuilenburg</surname><given-names>H</given-names></name><name><surname>Sheppard</surname><given-names>D</given-names></name><name><surname>Steed</surname><given-names>E</given-names></name><name><surname>Szpak</surname><given-names>M</given-names></name><name><surname>Szuba</surname><given-names>M</given-names></name><name><surname>Taylor</surname><given-names>K</given-names></name><name><surname>Thormann</surname><given-names>A</given-names></name><name><surname>Threadgold</surname><given-names>G</given-names></name><name><surname>Walts</surname><given-names>B</given-names></name><name><surname>Winterbottom</surname><given-names>A</given-names></name><name><surname>Chakiachvili</surname><given-names>M</given-names></name><name><surname>Chaubal</surname><given-names>A</given-names></name><name><surname>De Silva</surname><given-names>N</given-names></name><name><surname>Flint</surname><given-names>B</given-names></name><name><surname>Frankish</surname><given-names>A</given-names></name><name><surname>Hunt</surname><given-names>SE</given-names></name><name><surname>IIsley</surname><given-names>GR</given-names></name><name><surname>Langridge</surname><given-names>N</given-names></name><name><surname>Loveland</surname><given-names>JE</given-names></name><name><surname>Martin</surname><given-names>FJ</given-names></name><name><surname>Mudge</surname><given-names>JM</given-names></name><name><surname>Morales</surname><given-names>J</given-names></name><name><surname>Perry</surname><given-names>E</given-names></name><name><surname>Ruffier</surname><given-names>M</given-names></name><name><surname>Tate</surname><given-names>J</given-names></name><name><surname>Thybert</surname><given-names>D</given-names></name><name><surname>Trevanion</surname><given-names>SJ</given-names></name><name><surname>Cunningham</surname><given-names>F</given-names></name><name><surname>Yates</surname><given-names>AD</given-names></name><name><surname>Zerbino</surname><given-names>DR</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Ensembl 2021</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>D884</fpage><lpage>D891</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa942</pub-id><pub-id pub-id-type="pmid">33137190</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Iyer</surname><given-names>S</given-names></name><name><surname>Acharya</surname><given-names>KR</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Tying the knot: the cystine signature and molecular-recognition processes of the vascular endothelial growth factor family of angiogenic cytokines</article-title><source>The FEBS Journal</source><volume>278</volume><fpage>4304</fpage><lpage>4322</lpage><pub-id pub-id-type="doi">10.1111/j.1742-4658.2011.08350.x</pub-id><pub-id pub-id-type="pmid">21917115</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jiménez</surname><given-names>A</given-names></name><name><surname>Zu</surname><given-names>W</given-names></name><name><surname>Rawe</surname><given-names>VY</given-names></name><name><surname>Pelto-Huikko</surname><given-names>M</given-names></name><name><surname>Flickinger</surname><given-names>CJ</given-names></name><name><surname>Sutovsky</surname><given-names>P</given-names></name><name><surname>Gustafsson</surname><given-names>J-A</given-names></name><name><surname>Oko</surname><given-names>R</given-names></name><name><surname>Miranda-Vizuete</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Spermatocyte/spermatid-specific thioredoxin-3, a novel Golgi apparatus-associated thioredoxin, is a specific marker of aberrant spermatogenesis</article-title><source>The Journal of Biological Chemistry</source><volume>279</volume><fpage>34971</fpage><lpage>34982</lpage><pub-id pub-id-type="doi">10.1074/jbc.M404192200</pub-id><pub-id pub-id-type="pmid">15181017</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Back</surname><given-names>T</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Pacholska</surname><given-names>M</given-names></name><name><surname>Berghammer</surname><given-names>T</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Silver</surname><given-names>D</given-names></name><name><surname>Vinyals</surname><given-names>O</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction with alphafold</article-title><source>Nature</source><volume>596</volume><fpage>583</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id><pub-id pub-id-type="pmid">34265844</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Katz</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>ET</given-names></name><name><surname>Airoldi</surname><given-names>EM</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Analysis and design of RNA sequencing experiments for identifying isoform regulation</article-title><source>Nature Methods</source><volume>7</volume><fpage>1009</fpage><lpage>1015</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1528</pub-id><pub-id pub-id-type="pmid">21057496</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>D</given-names></name><name><surname>Paggi</surname><given-names>JM</given-names></name><name><surname>Park</surname><given-names>C</given-names></name><name><surname>Bennett</surname><given-names>C</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Graph-based genome alignment and genotyping with HISAT2 and HISAT-genotype</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>907</fpage><lpage>915</lpage><pub-id pub-id-type="doi">10.1038/s41587-019-0201-4</pub-id><pub-id pub-id-type="pmid">31375807</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kovaka</surname><given-names>S</given-names></name><name><surname>Zimin</surname><given-names>AV</given-names></name><name><surname>Pertea</surname><given-names>GM</given-names></name><name><surname>Razaghi</surname><given-names>R</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name><name><surname>Pertea</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Transcriptome assembly from long-read RNA-seq alignments with stringtie2</article-title><source>Genome Biology</source><volume>20</volume><elocation-id>278</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-019-1910-1</pub-id><pub-id pub-id-type="pmid">31842956</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krawitz</surname><given-names>PM</given-names></name><name><surname>Murakami</surname><given-names>Y</given-names></name><name><surname>Rieß</surname><given-names>A</given-names></name><name><surname>Hietala</surname><given-names>M</given-names></name><name><surname>Krüger</surname><given-names>U</given-names></name><name><surname>Zhu</surname><given-names>N</given-names></name><name><surname>Kinoshita</surname><given-names>T</given-names></name><name><surname>Mundlos</surname><given-names>S</given-names></name><name><surname>Hecht</surname><given-names>J</given-names></name><name><surname>Robinson</surname><given-names>PN</given-names></name><name><surname>Horn</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>PGAP2 mutations, affecting the GPI-anchor-synthesis pathway, cause hyperphosphatasia with mental retardation syndrome</article-title><source>American Journal of Human Genetics</source><volume>92</volume><fpage>584</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2013.03.011</pub-id><pub-id pub-id-type="pmid">23561847</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lal</surname><given-names>N</given-names></name><name><surname>Puri</surname><given-names>K</given-names></name><name><surname>Rodrigues</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Vascular endothelial growth factor B and its signaling</article-title><source>Frontiers in Cardiovascular Medicine</source><volume>5</volume><elocation-id>39</elocation-id><pub-id pub-id-type="doi">10.3389/fcvm.2018.00039</pub-id><pub-id pub-id-type="pmid">29732375</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Vegf-B: a thing of beauty</article-title><source>Cell Research</source><volume>20</volume><fpage>741</fpage><lpage>744</lpage><pub-id pub-id-type="doi">10.1038/cr.2010.77</pub-id><pub-id pub-id-type="pmid">20531376</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lindblad-Toh</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Broad institute sequencing platform and whole genome assembly team</article-title><source>Nature</source><volume>478</volume><fpage>476</fpage><lpage>482</lpage><pub-id pub-id-type="doi">10.1038/nature10530</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lovell</surname><given-names>SC</given-names></name><name><surname>Davis</surname><given-names>IW</given-names></name><name><surname>Arendall</surname><given-names>WB</given-names></name><name><surname>de Bakker</surname><given-names>PIW</given-names></name><name><surname>Word</surname><given-names>JM</given-names></name><name><surname>Prisant</surname><given-names>MG</given-names></name><name><surname>Richardson</surname><given-names>JS</given-names></name><name><surname>Richardson</surname><given-names>DC</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Structure validation by calpha geometry: phi,psi and cbeta deviation</article-title><source>Proteins</source><volume>50</volume><fpage>437</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1002/prot.10286</pub-id><pub-id pub-id-type="pmid">12557186</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matlin</surname><given-names>AJ</given-names></name><name><surname>Clark</surname><given-names>F</given-names></name><name><surname>Smith</surname><given-names>CWJ</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Understanding alternative splicing: towards a cellular code</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>6</volume><fpage>386</fpage><lpage>398</lpage><pub-id pub-id-type="doi">10.1038/nrm1645</pub-id><pub-id pub-id-type="pmid">15956978</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Melke</surname><given-names>J</given-names></name><name><surname>Goubran Botros</surname><given-names>H</given-names></name><name><surname>Chaste</surname><given-names>P</given-names></name><name><surname>Betancur</surname><given-names>C</given-names></name><name><surname>Nygren</surname><given-names>G</given-names></name><name><surname>Anckarsäter</surname><given-names>H</given-names></name><name><surname>Rastam</surname><given-names>M</given-names></name><name><surname>Ståhlberg</surname><given-names>O</given-names></name><name><surname>Gillberg</surname><given-names>IC</given-names></name><name><surname>Delorme</surname><given-names>R</given-names></name><name><surname>Chabane</surname><given-names>N</given-names></name><name><surname>Mouren-Simeoni</surname><given-names>M-C</given-names></name><name><surname>Fauchereau</surname><given-names>F</given-names></name><name><surname>Durand</surname><given-names>CM</given-names></name><name><surname>Chevalier</surname><given-names>F</given-names></name><name><surname>Drouot</surname><given-names>X</given-names></name><name><surname>Collet</surname><given-names>C</given-names></name><name><surname>Launay</surname><given-names>J-M</given-names></name><name><surname>Leboyer</surname><given-names>M</given-names></name><name><surname>Gillberg</surname><given-names>C</given-names></name><name><surname>Bourgeron</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Abnormal melatonin synthesis in autism spectrum disorders</article-title><source>Molecular Psychiatry</source><volume>13</volume><fpage>90</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1038/sj.mp.4002016</pub-id><pub-id pub-id-type="pmid">17505466</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mirdita</surname><given-names>M</given-names></name><name><surname>Schütze</surname><given-names>K</given-names></name><name><surname>Moriwaki</surname><given-names>Y</given-names></name><name><surname>Heo</surname><given-names>L</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ColabFold: making protein folding accessible to all</article-title><source>Nature Methods</source><volume>19</volume><fpage>679</fpage><lpage>682</lpage><pub-id pub-id-type="doi">10.1038/s41592-022-01488-1</pub-id><pub-id pub-id-type="pmid">35637307</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Modi</surname><given-names>T</given-names></name><name><surname>Huihui</surname><given-names>J</given-names></name><name><surname>Ghosh</surname><given-names>K</given-names></name><name><surname>Ozkan</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Ancient thioredoxins evolved to modern-day stability-function requirement by altering native state ensemble</article-title><source>Philosophical Transactions of the Royal Society of London. Series B, Biological Sciences</source><volume>373</volume><elocation-id>20170184</elocation-id><pub-id pub-id-type="doi">10.1098/rstb.2017.0184</pub-id><pub-id pub-id-type="pmid">29735738</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morales</surname><given-names>J</given-names></name><name><surname>Pujar</surname><given-names>S</given-names></name><name><surname>Loveland</surname><given-names>JE</given-names></name><name><surname>Astashyn</surname><given-names>A</given-names></name><name><surname>Bennett</surname><given-names>R</given-names></name><name><surname>Berry</surname><given-names>A</given-names></name><name><surname>Cox</surname><given-names>E</given-names></name><name><surname>Davidson</surname><given-names>C</given-names></name><name><surname>Ermolaeva</surname><given-names>O</given-names></name><name><surname>Farrell</surname><given-names>CM</given-names></name><name><surname>Fatima</surname><given-names>R</given-names></name><name><surname>Gil</surname><given-names>L</given-names></name><name><surname>Goldfarb</surname><given-names>T</given-names></name><name><surname>Gonzalez</surname><given-names>JM</given-names></name><name><surname>Haddad</surname><given-names>D</given-names></name><name><surname>Hardy</surname><given-names>M</given-names></name><name><surname>Hunt</surname><given-names>T</given-names></name><name><surname>Jackson</surname><given-names>J</given-names></name><name><surname>Joardar</surname><given-names>VS</given-names></name><name><surname>Kay</surname><given-names>M</given-names></name><name><surname>Kodali</surname><given-names>VK</given-names></name><name><surname>McGarvey</surname><given-names>KM</given-names></name><name><surname>McMahon</surname><given-names>A</given-names></name><name><surname>Mudge</surname><given-names>JM</given-names></name><name><surname>Murphy</surname><given-names>DN</given-names></name><name><surname>Murphy</surname><given-names>MR</given-names></name><name><surname>Rajput</surname><given-names>B</given-names></name><name><surname>Rangwala</surname><given-names>SH</given-names></name><name><surname>Riddick</surname><given-names>LD</given-names></name><name><surname>Thibaud-Nissen</surname><given-names>F</given-names></name><name><surname>Threadgold</surname><given-names>G</given-names></name><name><surname>Vatsan</surname><given-names>AR</given-names></name><name><surname>Wallin</surname><given-names>C</given-names></name><name><surname>Webb</surname><given-names>D</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Pruitt</surname><given-names>KD</given-names></name><name><surname>Frankish</surname><given-names>A</given-names></name><name><surname>Cunningham</surname><given-names>F</given-names></name><name><surname>Murphy</surname><given-names>TD</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>A joint NCBI and EMBL-EBI transcript set for clinical genomics and research</article-title><source>Nature</source><volume>604</volume><fpage>310</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.1038/s41586-022-04558-8</pub-id><pub-id pub-id-type="pmid">35388217</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Navarro Gonzalez</surname><given-names>J</given-names></name><name><surname>Zweig</surname><given-names>AS</given-names></name><name><surname>Speir</surname><given-names>ML</given-names></name><name><surname>Schmelter</surname><given-names>D</given-names></name><name><surname>Rosenbloom</surname><given-names>KR</given-names></name><name><surname>Raney</surname><given-names>BJ</given-names></name><name><surname>Powell</surname><given-names>CC</given-names></name><name><surname>Nassar</surname><given-names>LR</given-names></name><name><surname>Maulding</surname><given-names>ND</given-names></name><name><surname>Lee</surname><given-names>CM</given-names></name><name><surname>Lee</surname><given-names>BT</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Fyfe</surname><given-names>AC</given-names></name><name><surname>Fernandes</surname><given-names>JD</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Casper</surname><given-names>J</given-names></name><name><surname>Benet-Pagès</surname><given-names>A</given-names></name><name><surname>Barber</surname><given-names>GP</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Kuhn</surname><given-names>RM</given-names></name><name><surname>Haeussler</surname><given-names>M</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The UCSC genome browser database: 2021 update</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>D1046</fpage><lpage>D1057</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa1070</pub-id><pub-id pub-id-type="pmid">33221922</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nurk</surname><given-names>S</given-names></name><name><surname>Koren</surname><given-names>S</given-names></name><name><surname>Rhie</surname><given-names>A</given-names></name><name><surname>Rautiainen</surname><given-names>M</given-names></name><name><surname>Bzikadze</surname><given-names>AV</given-names></name><name><surname>Mikheenko</surname><given-names>A</given-names></name><name><surname>Vollger</surname><given-names>MR</given-names></name><name><surname>Altemose</surname><given-names>N</given-names></name><name><surname>Uralsky</surname><given-names>L</given-names></name><name><surname>Gershman</surname><given-names>A</given-names></name><name><surname>Aganezov</surname><given-names>S</given-names></name><name><surname>Hoyt</surname><given-names>SJ</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Logsdon</surname><given-names>GA</given-names></name><name><surname>Alonge</surname><given-names>M</given-names></name><name><surname>Antonarakis</surname><given-names>SE</given-names></name><name><surname>Borchers</surname><given-names>M</given-names></name><name><surname>Bouffard</surname><given-names>GG</given-names></name><name><surname>Brooks</surname><given-names>SY</given-names></name><name><surname>Caldas</surname><given-names>GV</given-names></name><name><surname>Chen</surname><given-names>N-C</given-names></name><name><surname>Cheng</surname><given-names>H</given-names></name><name><surname>Chin</surname><given-names>C-S</given-names></name><name><surname>Chow</surname><given-names>W</given-names></name><name><surname>de Lima</surname><given-names>LG</given-names></name><name><surname>Dishuck</surname><given-names>PC</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name><name><surname>Dvorkina</surname><given-names>T</given-names></name><name><surname>Fiddes</surname><given-names>IT</given-names></name><name><surname>Formenti</surname><given-names>G</given-names></name><name><surname>Fulton</surname><given-names>RS</given-names></name><name><surname>Fungtammasan</surname><given-names>A</given-names></name><name><surname>Garrison</surname><given-names>E</given-names></name><name><surname>Grady</surname><given-names>PGS</given-names></name><name><surname>Graves-Lindsay</surname><given-names>TA</given-names></name><name><surname>Hall</surname><given-names>IM</given-names></name><name><surname>Hansen</surname><given-names>NF</given-names></name><name><surname>Hartley</surname><given-names>GA</given-names></name><name><surname>Haukness</surname><given-names>M</given-names></name><name><surname>Howe</surname><given-names>K</given-names></name><name><surname>Hunkapiller</surname><given-names>MW</given-names></name><name><surname>Jain</surname><given-names>C</given-names></name><name><surname>Jain</surname><given-names>M</given-names></name><name><surname>Jarvis</surname><given-names>ED</given-names></name><name><surname>Kerpedjiev</surname><given-names>P</given-names></name><name><surname>Kirsche</surname><given-names>M</given-names></name><name><surname>Kolmogorov</surname><given-names>M</given-names></name><name><surname>Korlach</surname><given-names>J</given-names></name><name><surname>Kremitzki</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Maduro</surname><given-names>VV</given-names></name><name><surname>Marschall</surname><given-names>T</given-names></name><name><surname>McCartney</surname><given-names>AM</given-names></name><name><surname>McDaniel</surname><given-names>J</given-names></name><name><surname>Miller</surname><given-names>DE</given-names></name><name><surname>Mullikin</surname><given-names>JC</given-names></name><name><surname>Myers</surname><given-names>EW</given-names></name><name><surname>Olson</surname><given-names>ND</given-names></name><name><surname>Paten</surname><given-names>B</given-names></name><name><surname>Peluso</surname><given-names>P</given-names></name><name><surname>Pevzner</surname><given-names>PA</given-names></name><name><surname>Porubsky</surname><given-names>D</given-names></name><name><surname>Potapova</surname><given-names>T</given-names></name><name><surname>Rogaev</surname><given-names>EI</given-names></name><name><surname>Rosenfeld</surname><given-names>JA</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name><name><surname>Schneider</surname><given-names>VA</given-names></name><name><surname>Sedlazeck</surname><given-names>FJ</given-names></name><name><surname>Shafin</surname><given-names>K</given-names></name><name><surname>Shew</surname><given-names>CJ</given-names></name><name><surname>Shumate</surname><given-names>A</given-names></name><name><surname>Sims</surname><given-names>Y</given-names></name><name><surname>Smit</surname><given-names>AFA</given-names></name><name><surname>Soto</surname><given-names>DC</given-names></name><name><surname>Sović</surname><given-names>I</given-names></name><name><surname>Storer</surname><given-names>JM</given-names></name><name><surname>Streets</surname><given-names>A</given-names></name><name><surname>Sullivan</surname><given-names>BA</given-names></name><name><surname>Thibaud-Nissen</surname><given-names>F</given-names></name><name><surname>Torrance</surname><given-names>J</given-names></name><name><surname>Wagner</surname><given-names>J</given-names></name><name><surname>Walenz</surname><given-names>BP</given-names></name><name><surname>Wenger</surname><given-names>A</given-names></name><name><surname>Wood</surname><given-names>JMD</given-names></name><name><surname>Xiao</surname><given-names>C</given-names></name><name><surname>Yan</surname><given-names>SM</given-names></name><name><surname>Young</surname><given-names>AC</given-names></name><name><surname>Zarate</surname><given-names>S</given-names></name><name><surname>Surti</surname><given-names>U</given-names></name><name><surname>McCoy</surname><given-names>RC</given-names></name><name><surname>Dennis</surname><given-names>MY</given-names></name><name><surname>Alexandrov</surname><given-names>IA</given-names></name><name><surname>Gerton</surname><given-names>JL</given-names></name><name><surname>O’Neill</surname><given-names>RJ</given-names></name><name><surname>Timp</surname><given-names>W</given-names></name><name><surname>Zook</surname><given-names>JM</given-names></name><name><surname>Schatz</surname><given-names>MC</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Miga</surname><given-names>KH</given-names></name><name><surname>Phillippy</surname><given-names>AM</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The complete sequence of a human genome</article-title><source>Science</source><volume>376</volume><fpage>44</fpage><lpage>53</lpage><pub-id pub-id-type="doi">10.1126/science.abj6987</pub-id><pub-id pub-id-type="pmid">35357919</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>O’Leary</surname><given-names>NA</given-names></name><name><surname>Wright</surname><given-names>MW</given-names></name><name><surname>Brister</surname><given-names>JR</given-names></name><name><surname>Ciufo</surname><given-names>S</given-names></name><name><surname>Haddad</surname><given-names>D</given-names></name><name><surname>McVeigh</surname><given-names>R</given-names></name><name><surname>Rajput</surname><given-names>B</given-names></name><name><surname>Robbertse</surname><given-names>B</given-names></name><name><surname>Smith-White</surname><given-names>B</given-names></name><name><surname>Ako-Adjei</surname><given-names>D</given-names></name><name><surname>Astashyn</surname><given-names>A</given-names></name><name><surname>Badretdin</surname><given-names>A</given-names></name><name><surname>Bao</surname><given-names>Y</given-names></name><name><surname>Blinkova</surname><given-names>O</given-names></name><name><surname>Brover</surname><given-names>V</given-names></name><name><surname>Chetvernin</surname><given-names>V</given-names></name><name><surname>Choi</surname><given-names>J</given-names></name><name><surname>Cox</surname><given-names>E</given-names></name><name><surname>Ermolaeva</surname><given-names>O</given-names></name><name><surname>Farrell</surname><given-names>CM</given-names></name><name><surname>Goldfarb</surname><given-names>T</given-names></name><name><surname>Gupta</surname><given-names>T</given-names></name><name><surname>Haft</surname><given-names>D</given-names></name><name><surname>Hatcher</surname><given-names>E</given-names></name><name><surname>Hlavina</surname><given-names>W</given-names></name><name><surname>Joardar</surname><given-names>VS</given-names></name><name><surname>Kodali</surname><given-names>VK</given-names></name><name><surname>Li</surname><given-names>W</given-names></name><name><surname>Maglott</surname><given-names>D</given-names></name><name><surname>Masterson</surname><given-names>P</given-names></name><name><surname>McGarvey</surname><given-names>KM</given-names></name><name><surname>Murphy</surname><given-names>MR</given-names></name><name><surname>O’Neill</surname><given-names>K</given-names></name><name><surname>Pujar</surname><given-names>S</given-names></name><name><surname>Rangwala</surname><given-names>SH</given-names></name><name><surname>Rausch</surname><given-names>D</given-names></name><name><surname>Riddick</surname><given-names>LD</given-names></name><name><surname>Schoch</surname><given-names>C</given-names></name><name><surname>Shkeda</surname><given-names>A</given-names></name><name><surname>Storz</surname><given-names>SS</given-names></name><name><surname>Sun</surname><given-names>H</given-names></name><name><surname>Thibaud-Nissen</surname><given-names>F</given-names></name><name><surname>Tolstoy</surname><given-names>I</given-names></name><name><surname>Tully</surname><given-names>RE</given-names></name><name><surname>Vatsan</surname><given-names>AR</given-names></name><name><surname>Wallin</surname><given-names>C</given-names></name><name><surname>Webb</surname><given-names>D</given-names></name><name><surname>Wu</surname><given-names>W</given-names></name><name><surname>Landrum</surname><given-names>MJ</given-names></name><name><surname>Kimchi</surname><given-names>A</given-names></name><name><surname>Tatusova</surname><given-names>T</given-names></name><name><surname>DiCuccio</surname><given-names>M</given-names></name><name><surname>Kitts</surname><given-names>P</given-names></name><name><surname>Murphy</surname><given-names>TD</given-names></name><name><surname>Pruitt</surname><given-names>KD</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Reference sequence (refseq) database at NCBI: current status, taxonomic expansion, and functional annotation</article-title><source>Nucleic Acids Research</source><volume>44</volume><fpage>D733</fpage><lpage>D745</lpage><pub-id pub-id-type="doi">10.1093/nar/gkv1189</pub-id><pub-id pub-id-type="pmid">26553804</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palazzo</surname><given-names>AF</given-names></name><name><surname>Lee</surname><given-names>ES</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Non-coding RNA: what is functional and what is junk?</article-title><source>Frontiers in Genetics</source><volume>6</volume><elocation-id>2</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2015.00002</pub-id><pub-id pub-id-type="pmid">25674102</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patro</surname><given-names>R</given-names></name><name><surname>Duggal</surname><given-names>G</given-names></name><name><surname>Love</surname><given-names>MI</given-names></name><name><surname>Irizarry</surname><given-names>RA</given-names></name><name><surname>Kingsford</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Salmon provides fast and bias-aware quantification of transcript expression</article-title><source>Nature Methods</source><volume>14</volume><fpage>417</fpage><lpage>419</lpage><pub-id pub-id-type="doi">10.1038/nmeth.4197</pub-id><pub-id pub-id-type="pmid">28263959</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pertea</surname><given-names>M</given-names></name><name><surname>Shumate</surname><given-names>A</given-names></name><name><surname>Pertea</surname><given-names>G</given-names></name><name><surname>Varabyou</surname><given-names>A</given-names></name><name><surname>Breitwieser</surname><given-names>FP</given-names></name><name><surname>Chang</surname><given-names>YC</given-names></name><name><surname>Madugundu</surname><given-names>AK</given-names></name><name><surname>Pandey</surname><given-names>A</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Chess: a new human gene catalog curated from thousands of large-scale RNA sequencing experiments reveals extensive transcriptional noise</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>208</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1590-2</pub-id><pub-id pub-id-type="pmid">30486838</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pertea</surname><given-names>G</given-names></name><name><surname>Pertea</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>GFF utilities: gffread and gffcompare</article-title><source>F1000Research</source><volume>9</volume><elocation-id>ISCB Comm J-304</elocation-id><pub-id pub-id-type="doi">10.12688/f1000research.23297.2</pub-id><pub-id pub-id-type="pmid">32489650</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ponting</surname><given-names>CP</given-names></name><name><surname>Haerty</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Genome-Wide analysis of human long noncoding RNAs: a provocative review</article-title><source>Annual Review of Genomics and Human Genetics</source><volume>23</volume><fpage>153</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-112921-123710</pub-id><pub-id pub-id-type="pmid">35395170</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rossignol</surname><given-names>DA</given-names></name><name><surname>Frye</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Melatonin in autism spectrum disorders: a systematic review and meta-analysis</article-title><source>Developmental Medicine and Child Neurology</source><volume>53</volume><fpage>783</fpage><lpage>792</lpage><pub-id pub-id-type="doi">10.1111/j.1469-8749.2011.03980.x</pub-id><pub-id pub-id-type="pmid">21518346</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ruff</surname><given-names>KM</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>AlphaFold and implications for intrinsically disordered proteins</article-title><source>Journal of Molecular Biology</source><volume>433</volume><elocation-id>167208</elocation-id><pub-id pub-id-type="doi">10.1016/j.jmb.2021.167208</pub-id><pub-id pub-id-type="pmid">34418423</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Open questions: how many genes do we have?</article-title><source>BMC Biology</source><volume>16</volume><elocation-id>94</elocation-id><pub-id pub-id-type="doi">10.1186/s12915-018-0564-x</pub-id><pub-id pub-id-type="pmid">30124169</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Schrödinger</surname><given-names>LLC</given-names></name></person-group><year iso-8601-date="2015">2015</year><data-title>The pymol molecular graphics system</data-title><version designator="Version 1.8">Version 1.8</version><source>Pymol</source><ext-link ext-link-type="uri" xlink:href="https://pymol.org/2/">https://pymol.org/2/</ext-link></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets</article-title><source>Nature Biotechnology</source><volume>35</volume><fpage>1026</fpage><lpage>1028</lpage><pub-id pub-id-type="doi">10.1038/nbt.3988</pub-id><pub-id pub-id-type="pmid">29035372</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Suzek</surname><given-names>BE</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name><name><surname>McGarvey</surname><given-names>PB</given-names></name><name><surname>Wu</surname><given-names>CH</given-names></name><collab>UniProt Consortium</collab></person-group><year iso-8601-date="2015">2015</year><article-title>UniRef clusters: a comprehensive and scalable alternative for improving sequence similarity searches</article-title><source>Bioinformatics</source><volume>31</volume><fpage>926</fpage><lpage>932</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btu739</pub-id><pub-id pub-id-type="pmid">25398609</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tashima</surname><given-names>Y</given-names></name><name><surname>Taguchi</surname><given-names>R</given-names></name><name><surname>Murata</surname><given-names>C</given-names></name><name><surname>Ashida</surname><given-names>H</given-names></name><name><surname>Kinoshita</surname><given-names>T</given-names></name><name><surname>Maeda</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>PGAP2 is essential for correct processing and stable expression of GPI-anchored proteins</article-title><source>Molecular Biology of the Cell</source><volume>17</volume><fpage>1410</fpage><lpage>1420</lpage><pub-id pub-id-type="doi">10.1091/mbc.e05-11-1005</pub-id><pub-id pub-id-type="pmid">16407401</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Temple</surname><given-names>G</given-names></name><name><surname>Gerhard</surname><given-names>DS</given-names></name><name><surname>Rasooly</surname><given-names>R</given-names></name><name><surname>Feingold</surname><given-names>EA</given-names></name><name><surname>Good</surname><given-names>PJ</given-names></name><name><surname>Robinson</surname><given-names>C</given-names></name><name><surname>Mandich</surname><given-names>A</given-names></name><name><surname>Derge</surname><given-names>JG</given-names></name><name><surname>Lewis</surname><given-names>J</given-names></name><name><surname>Shoaf</surname><given-names>D</given-names></name><name><surname>Collins</surname><given-names>FS</given-names></name><name><surname>Jang</surname><given-names>W</given-names></name><name><surname>Wagner</surname><given-names>L</given-names></name><name><surname>Shenmen</surname><given-names>CM</given-names></name><name><surname>Misquitta</surname><given-names>L</given-names></name><name><surname>Schaefer</surname><given-names>CF</given-names></name><name><surname>Buetow</surname><given-names>KH</given-names></name><name><surname>Bonner</surname><given-names>TI</given-names></name><name><surname>Yankie</surname><given-names>L</given-names></name><name><surname>Ward</surname><given-names>M</given-names></name><name><surname>Phan</surname><given-names>L</given-names></name><name><surname>Astashyn</surname><given-names>A</given-names></name><name><surname>Brown</surname><given-names>G</given-names></name><name><surname>Farrell</surname><given-names>C</given-names></name><name><surname>Hart</surname><given-names>J</given-names></name><name><surname>Landrum</surname><given-names>M</given-names></name><name><surname>Maidak</surname><given-names>BL</given-names></name><name><surname>Murphy</surname><given-names>M</given-names></name><name><surname>Murphy</surname><given-names>T</given-names></name><name><surname>Rajput</surname><given-names>B</given-names></name><name><surname>Riddick</surname><given-names>L</given-names></name><name><surname>Webb</surname><given-names>D</given-names></name><name><surname>Weber</surname><given-names>J</given-names></name><name><surname>Wu</surname><given-names>W</given-names></name><name><surname>Pruitt</surname><given-names>KD</given-names></name><name><surname>Maglott</surname><given-names>D</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name><name><surname>Brejova</surname><given-names>B</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Harte</surname><given-names>R</given-names></name><name><surname>Baertsch</surname><given-names>R</given-names></name><name><surname>Kent</surname><given-names>J</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Brent</surname><given-names>M</given-names></name><name><surname>Langton</surname><given-names>L</given-names></name><name><surname>Comstock</surname><given-names>CLG</given-names></name><name><surname>Stevens</surname><given-names>M</given-names></name><name><surname>Wei</surname><given-names>C</given-names></name><name><surname>van Baren</surname><given-names>MJ</given-names></name><name><surname>Salehi-Ashtiani</surname><given-names>K</given-names></name><name><surname>Murray</surname><given-names>RR</given-names></name><name><surname>Ghamsari</surname><given-names>L</given-names></name><name><surname>Mello</surname><given-names>E</given-names></name><name><surname>Lin</surname><given-names>C</given-names></name><name><surname>Pennacchio</surname><given-names>C</given-names></name><name><surname>Schreiber</surname><given-names>K</given-names></name><name><surname>Shapiro</surname><given-names>N</given-names></name><name><surname>Marsh</surname><given-names>A</given-names></name><name><surname>Pardes</surname><given-names>E</given-names></name><name><surname>Moore</surname><given-names>T</given-names></name><name><surname>Lebeau</surname><given-names>A</given-names></name><name><surname>Muratet</surname><given-names>M</given-names></name><name><surname>Simmons</surname><given-names>B</given-names></name><name><surname>Kloske</surname><given-names>D</given-names></name><name><surname>Sieja</surname><given-names>S</given-names></name><name><surname>Hudson</surname><given-names>J</given-names></name><name><surname>Sethupathy</surname><given-names>P</given-names></name><name><surname>Brownstein</surname><given-names>M</given-names></name><name><surname>Bhat</surname><given-names>N</given-names></name><name><surname>Lazar</surname><given-names>J</given-names></name><name><surname>Jacob</surname><given-names>H</given-names></name><name><surname>Gruber</surname><given-names>CE</given-names></name><name><surname>Smith</surname><given-names>MR</given-names></name><name><surname>McPherson</surname><given-names>J</given-names></name><name><surname>Garcia</surname><given-names>AM</given-names></name><name><surname>Gunaratne</surname><given-names>PH</given-names></name><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Muzny</surname><given-names>D</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Young</surname><given-names>AC</given-names></name><name><surname>Bouffard</surname><given-names>GG</given-names></name><name><surname>Blakesley</surname><given-names>RW</given-names></name><name><surname>Mullikin</surname><given-names>J</given-names></name><name><surname>Green</surname><given-names>ED</given-names></name><name><surname>Dickson</surname><given-names>MC</given-names></name><name><surname>Rodriguez</surname><given-names>AC</given-names></name><name><surname>Grimwood</surname><given-names>J</given-names></name><name><surname>Schmutz</surname><given-names>J</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Hirst</surname><given-names>M</given-names></name><name><surname>Zeng</surname><given-names>T</given-names></name><name><surname>Tse</surname><given-names>K</given-names></name><name><surname>Moksa</surname><given-names>M</given-names></name><name><surname>Deng</surname><given-names>M</given-names></name><name><surname>Ma</surname><given-names>K</given-names></name><name><surname>Mah</surname><given-names>D</given-names></name><name><surname>Pang</surname><given-names>J</given-names></name><name><surname>Taylor</surname><given-names>G</given-names></name><name><surname>Chuah</surname><given-names>E</given-names></name><name><surname>Deng</surname><given-names>A</given-names></name><name><surname>Fichter</surname><given-names>K</given-names></name><name><surname>Go</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>S</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Griffith</surname><given-names>M</given-names></name><name><surname>Morin</surname><given-names>R</given-names></name><name><surname>Moore</surname><given-names>RA</given-names></name><name><surname>Mayo</surname><given-names>M</given-names></name><name><surname>Munro</surname><given-names>S</given-names></name><name><surname>Wagner</surname><given-names>S</given-names></name><name><surname>Jones</surname><given-names>SJM</given-names></name><name><surname>Holt</surname><given-names>RA</given-names></name><name><surname>Marra</surname><given-names>MA</given-names></name><name><surname>Lu</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>S</given-names></name><name><surname>Hartigan</surname><given-names>J</given-names></name><name><surname>Graf</surname><given-names>M</given-names></name><name><surname>Wagner</surname><given-names>R</given-names></name><name><surname>Letovksy</surname><given-names>S</given-names></name><name><surname>Pulido</surname><given-names>JC</given-names></name><name><surname>Robison</surname><given-names>K</given-names></name><name><surname>Esposito</surname><given-names>D</given-names></name><name><surname>Hartley</surname><given-names>J</given-names></name><name><surname>Wall</surname><given-names>VE</given-names></name><name><surname>Hopkins</surname><given-names>RF</given-names></name><name><surname>Ohara</surname><given-names>O</given-names></name><name><surname>Wiemann</surname><given-names>S</given-names></name><collab>MGC Project Team</collab></person-group><year iso-8601-date="2009">2009</year><article-title>The completion of the mammalian gene collection (mgc)</article-title><source>Genome Research</source><volume>19</volume><fpage>2324</fpage><lpage>2333</lpage><pub-id pub-id-type="doi">10.1101/gr.095976.109</pub-id><pub-id pub-id-type="pmid">19767417</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tung</surname><given-names>KF</given-names></name><name><surname>Pan</surname><given-names>CY</given-names></name><name><surname>Chen</surname><given-names>CH</given-names></name><name><surname>Lin</surname><given-names>WC</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Top-ranked expressed gene transcripts of human protein-coding genes investigated with gtex dataset</article-title><source>Scientific Reports</source><volume>10</volume><elocation-id>16245</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-020-73081-5</pub-id><pub-id pub-id-type="pmid">33004865</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Wu</surname><given-names>Z</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Laydon</surname><given-names>A</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name><name><surname>Kleywegt</surname><given-names>GJ</given-names></name><name><surname>Bateman</surname><given-names>A</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction for the human proteome</article-title><source>Nature</source><volume>596</volume><fpage>590</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03828-1</pub-id><pub-id pub-id-type="pmid">34293799</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Uppala</surname><given-names>R</given-names></name><name><surname>Tsoi</surname><given-names>LC</given-names></name><name><surname>Harms</surname><given-names>PW</given-names></name><name><surname>Wang</surname><given-names>B</given-names></name><name><surname>Billi</surname><given-names>AC</given-names></name><name><surname>Maverakis</surname><given-names>E</given-names></name><name><surname>Michelle Kahlenberg</surname><given-names>J</given-names></name><name><surname>Ward</surname><given-names>NL</given-names></name><name><surname>Gudjonsson</surname><given-names>JE</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>“ autoinflammatory psoriasis ” -genetics and biology of pustular psoriasis</article-title><source>Cellular &amp; Molecular Immunology</source><volume>18</volume><fpage>307</fpage><lpage>317</lpage><pub-id pub-id-type="doi">10.1038/s41423-020-0519-3</pub-id><pub-id pub-id-type="pmid">32814870</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>van Kempen</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>SS</given-names></name><name><surname>Tumescheit</surname><given-names>C</given-names></name><name><surname>Mirdita</surname><given-names>M</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Foldseek: Fast and Accurate Protein Structure Search</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.02.07.479398</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Varabyou</surname><given-names>A</given-names></name><name><surname>Pertea</surname><given-names>G</given-names></name><name><surname>Pockrandt</surname><given-names>C</given-names></name><name><surname>Pertea</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>TieBrush: an efficient method for aggregating and summarizing mapped reads across large datasets</article-title><source>Bioinformatics</source><volume>37</volume><fpage>3650</fpage><lpage>3651</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btab342</pub-id><pub-id pub-id-type="pmid">33964128</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Varadi</surname><given-names>M</given-names></name><name><surname>Anyango</surname><given-names>S</given-names></name><name><surname>Deshpande</surname><given-names>M</given-names></name><name><surname>Nair</surname><given-names>S</given-names></name><name><surname>Natassia</surname><given-names>C</given-names></name><name><surname>Yordanova</surname><given-names>G</given-names></name><name><surname>Yuan</surname><given-names>D</given-names></name><name><surname>Stroe</surname><given-names>O</given-names></name><name><surname>Wood</surname><given-names>G</given-names></name><name><surname>Laydon</surname><given-names>A</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Green</surname><given-names>R</given-names></name><name><surname>Vora</surname><given-names>A</given-names></name><name><surname>Lutfi</surname><given-names>M</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Hobbs</surname><given-names>N</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Kleywegt</surname><given-names>G</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>AlphaFold protein structure database: massively expanding the structural coverage of protein-sequence space with high-accuracy models</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D439</fpage><lpage>D444</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1061</pub-id><pub-id pub-id-type="pmid">34791371</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>ET</given-names></name><name><surname>Sandberg</surname><given-names>R</given-names></name><name><surname>Luo</surname><given-names>S</given-names></name><name><surname>Khrebtukova</surname><given-names>I</given-names></name><name><surname>Zhang</surname><given-names>L</given-names></name><name><surname>Mayr</surname><given-names>C</given-names></name><name><surname>Kingsmore</surname><given-names>SF</given-names></name><name><surname>Schroth</surname><given-names>GP</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Alternative isoform regulation in human tissue transcriptomes</article-title><source>Nature</source><volume>456</volume><fpage>470</fpage><lpage>476</lpage><pub-id pub-id-type="doi">10.1038/nature07509</pub-id><pub-id pub-id-type="pmid">18978772</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wistow</surname><given-names>G</given-names></name><name><surname>Wyatt</surname><given-names>K</given-names></name><name><surname>David</surname><given-names>L</given-names></name><name><surname>Gao</surname><given-names>C</given-names></name><name><surname>Bateman</surname><given-names>O</given-names></name><name><surname>Bernstein</surname><given-names>S</given-names></name><name><surname>Tomarev</surname><given-names>S</given-names></name><name><surname>Segovia</surname><given-names>L</given-names></name><name><surname>Slingsby</surname><given-names>C</given-names></name><name><surname>Vihtelic</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>GammaN-crystallin and the evolution of the betagamma-crystallin superfamily in vertebrates</article-title><source>The FEBS Journal</source><volume>272</volume><fpage>2276</fpage><lpage>2291</lpage><pub-id pub-id-type="doi">10.1111/j.1742-4658.2005.04655.x</pub-id><pub-id pub-id-type="pmid">15853812</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Harrison</surname><given-names>P</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Identification and analysis of over 2000 ribosomal protein pseudogenes in the human genome</article-title><source>Genome Research</source><volume>12</volume><fpage>1466</fpage><lpage>1482</lpage><pub-id pub-id-type="doi">10.1101/gr.331902</pub-id><pub-id pub-id-type="pmid">12368239</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82556.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Dötsch</surname><given-names>Volker</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04cvxnb49</institution-id><institution>Goethe University</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/2022.06.08.495354" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/2022.06.08.495354"/></front-stub><body><p>This study applies AlphaFold to the CHESS selection of transcripts with the goal of generating predicted 3D protein structures and a quality measure of folding, the pLDDT score. From these data, the authors build a database for result exploration, documented by several examples, including proteins, where the authors propose the pLDDT score as a measure of presumed superior biological functionality over other isoforms. These results will be highly relevant for anyone working with proteins that occur in different isoforms.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82556.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Dötsch</surname><given-names>Volker</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04cvxnb49</institution-id><institution>Goethe University</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Poetsch</surname><given-names>Anna</given-names></name><role>Reviewer</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/042aqky30</institution-id><institution>TU Dresden</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2022.06.08.495354">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2022.06.08.495354v1">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Structure-guided isoform identification for the human transcriptome&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 2 peer reviewers, and the evaluation has been overseen by a Reviewing Editor and Volker Dötsch as the Senior Editor. The following individual involved in review of your submission has agreed to reveal their identity: Anna Poetsch (Reviewer #2).</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>1) Using the representative isoform from the MANE database the authors provide 5 exemplar genes whose MANE representative isoform when translated yields a protein structure with a lower pLDDT score than a non-representative isoform, suggesting that the protein stability of the representative isoform is suspect. The authors clearly explain that not all isoforms encode a well folded functional protein. Thus, it would be helpful for the authors to provide a clearer indication of how common the non-exemplar representative isoforms in the MANE database may have a lower pLDDT score. Armed with this a false positive and negative estimate could be provided.</p><p>2) The authors evaluate the single representative isoform selected for each human protein coding gene in the MANE database. In addition, the authors provide a list of isoforms with the highest pLDDT score for each human gene. However, it would be useful for the authors to provide a section that speaks to the confounding conditions and sequence features are understood to be confounding in evaluating the use of pLDDT scores. For example, since evolutionary conservation and synteny considerations of isoform sequences are an important consideration in evaluating the potential utility of an isoform, what would the authors recommend if the pLDDT scores are elevated but one or more of the other considerations are less than ideal?</p><p>3) The study is based on the elegant idea to aid genome annotation through 3D structure prediction. This is a very powerful approach that allows large-scale data generation for functional interpretation. This approach appears technically sound and well executed (although I may miss details not being a protein expert). However, in my opinion, the authors could make more use of the potential of their approach. From the big-data start, they seem to directly restrict themselves to interesting examples. I am missing a global analysis that shows the bigger picture of their results. Given that they have generated structures from 90,415 isoforms, each associated with a pLDDT score, conservation scores, length, expression levels and other quantifiable data listed on page 18. I would wish for a comprehensive analysis of these data and their potential before applying the focus on a few (admittedly very nice) examples.</p><p>4) One of the weak spots of such an analysis is the relationship between foldability and functional relevance. Disordered regions would imply reduced relevance due to poor pLDDT scores, which may be a misleading conclusion. While this may be a problem difficult to solve with this approach, it still needs to be addressed and discussed throughout the paper and particularly as part of the global analysis, not just in the context of examples.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>The manuscript is well written and speaks to an important and timely issue concerning the number how to annotate and evaluate isoforms for each gene. As the number of isoforms for a genome continues to increase it will be helpful to provide some logic to distinguish among the isoforms.</p><p>Using the representative isoform from the MANE database the authors provide 5 exemplar genes whose MANE representative isoform when translated yields a protein structure with a lower pLDDT score than a non-representative isoform, suggesting that the protein stability of the representative isoform is suspect. The authors clearly explain that not all isoforms encode a well folded functional protein. Thus, it would be helpful for the authors to provide a clearer indication of how common the non-exemplar representative isoforms in the MANE database may have a lower pLDDT score. Armed with this a false positive and negative estimate could be provided.</p><p>The authors evaluate the single representative isoform selected for each human protein coding gene in the MANE database. In addition, the authors provide a list of isoforms with the highest pLDDT score for each human gene. However, it would be useful for the authors to provide a section that speaks to the confounding conditions and sequence features are understood to be confounding in evaluating the use of pLDDT scores. For example, since evolutionary conservation and synteny considerations of isoform sequences are important consideration in evaluating the potential utility of an isoform, what would the authors recommend if the pLDDT scores are elevated but one or more of the other considerations are less than ideal?</p><p>Finally, the authors are encouraged to provide readers with a reasoned argument that the addition of structure-guided isoform considerations is more than an incremental advancement in the annotation of the human transcriptome.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.82556.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>1) Using the representative isoform from the MANE database the authors provide 5 exemplar genes whose MANE representative isoform when translated yields a protein structure with a lower pLDDT score than a non-representative isoform, suggesting that the protein stability of the representative isoform is suspect. The authors clearly explain that not all isoforms encode a well folded functional protein. Thus, it would be helpful for the authors to provide a clearer indication of how common the non-exemplar representative isoforms in the MANE database may have a lower pLDDT score. Armed with this a false positive and negative estimate could be provided.</p></disp-quote><p>We have updated our analysis to include all CHESS 3 isoforms &lt;= 1000aa (rather than the previous limit of 500aa) and have lifted over all CHESS 3 matching predictions from the AlphaFold Protein Structure Database, which increases our maximum length limit to 2699aa. This resulted in &gt;98% of all human loci containing at least one protein structure prediction, which we hope makes the following additions to the global analysis more truly global.</p><p>We have rewritten the “Scoring the transcriptome” section of the results to include a more comprehensive explanation of the comparison between MANE and the higher-scoring isoforms. To clarify how commonly the non-exemplar isoforms (i.e., the ones not included in MANE) outscore the canonical isoform, we added a count of the number of isoforms that score a higher pLDDT vs. their associated MANE isoform. However, even with RNA-seq, protein structure prediction, and evolutionary conservation evidence, we do not believe we can create a gold standard for which transcripts are truly functional, and thus we cannot provide an unbiased estimate of false positives and negatives for MANE. Many of the hundreds of isoforms identified in Table S3 represent poorly-studied proteins where the additional experimental evidence used to further analyze our Exemplary Predictions simply does not exist. We believe that our addition of more explicit numbers, as now provided in the updated Results section, provides a fair comparison to MANE despite lacking formal false positive and negative estimates.</p><disp-quote content-type="editor-comment"><p>2) The authors evaluate the single representative isoform selected for each human protein coding gene in the MANE database. In addition, the authors provide a list of isoforms with the highest pLDDT score for each human gene. However, it would be useful for the authors to provide a section that speaks to the confounding conditions and sequence features are understood to be confounding in evaluating the use of pLDDT scores. For example, since evolutionary conservation and synteny considerations of isoform sequences are an important consideration in evaluating the potential utility of an isoform, what would the authors recommend if the pLDDT scores are elevated but one or more of the other considerations are less than ideal?</p></disp-quote><p>We have added a more thorough explanation to the “Scoring the transcriptome” section on why certain isoforms may achieve a high pLDDT but still be non-functional. Due in part to the problem of short protein fragments (now mentioned in the results), which sometimes get misleadingly high scores, our set of filtered transcripts is based on a combination of foldability and RNA-seq expression evidence rather than foldability alone. We hope this clearer description of the reasoning behind our methods, alongside the examples which incorporate additional experimental evidence, may show future researchers how to handle cases where considerations are less than ideal, e.g. ASMT lacking expression data in GTEx because it is expressed primarily in the pineal gland which was not sampled. Our overall goal here is not to provide a definitive solution to the problem of determining which isoforms are functional (as we believe any attempt to do so would be suboptimal and incomplete at this time), but rather to provide a type of field-guide and associated genome-wide database which can be used as resource in future efforts.</p><disp-quote content-type="editor-comment"><p>3) The study is based on the elegant idea to aid genome annotation through 3D structure prediction. This is a very powerful approach that allows large-scale data generation for functional interpretation. This approach appears technically sound and well executed (although I may miss details not being a protein expert). However, in my opinion, the authors could make more use of the potential of their approach. From the big-data start, they seem to directly restrict themselves to interesting examples. I am missing a global analysis that shows the bigger picture of their results. Given that they have generated structures from 90,415 isoforms, each associated with a pLDDT score, conservation scores, length, expression levels and other quantifiable data listed on page 18. I would wish for a comprehensive analysis of these data and their potential before applying the focus on a few (admittedly very nice) examples.</p></disp-quote><p>We have added Figure 1, a transcriptome-wide view of the relationship between pLDDT vs. protein length and GTEx expression to the results. The figures shows a lack of a clear linear relationship between pLDDT and length or expression, which implies that protein structure prediction provides an orthogonal source of useful information for annotation. This is also noted in the updated manuscript.</p><disp-quote content-type="editor-comment"><p>4) One of the weak spots of such an analysis is the relationship between foldability and functional relevance. Disordered regions would imply reduced relevance due to poor pLDDT scores, which may be a misleading conclusion. While this may be a problem difficult to solve with this approach, it still needs to be addressed and discussed throughout the paper and particularly as part of the global analysis, not just in the context of examples.</p></disp-quote><p>This is a great point. Many functional proteins are either entirely disordered or contain intrinsically disordered regions, and protein folding will inherently fail to provide useful information in these cases. We have added a section as part of the global analysis in the results which quantifies how often we observe this in our data. Additionally, we have expanded our Discussion section to include a more thorough description of how we believe this problem should be approached by researchers looking at protein foldability in context with expression and evolutionary evidence.</p></body></sub-article></article>