<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article article-type="research-article" dtd-version="1.2" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">70692</article-id><article-id pub-id-type="doi">10.7554/eLife.70692</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>RNA splicing programs define tissue compartments and cell types at single-cell resolution</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes" id="author-239939"><name><surname>Olivieri</surname><given-names>Julia Eve</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0850-5498</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes" id="author-127927"><name><surname>Dehghannasiri</surname><given-names>Roozbeh</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7413-3437</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund4"/><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-31538"><name><surname>Wang</surname><given-names>Peter L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9651-3860</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-141507"><name><surname>Jang</surname><given-names>SoRi</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-48103"><name><surname>de Morree</surname><given-names>Antoine</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8316-4531</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-240364"><name><surname>Tan</surname><given-names>Serena Y</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-240365"><name><surname>Ming</surname><given-names>Jingsi</given-names></name><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-150753"><name><surname>Ruohao Wu</surname><given-names>Angela</given-names></name><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author"><collab>Tabula Sapiens Consortium<contrib-group><contrib><name><surname>Jones</surname><given-names>Robert C</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Karkanias</surname><given-names>Jim</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Krasnow</surname><given-names>Mark</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Howard Hughes Medical Institute</institution><addr-line><named-content content-type="city">Chevy Chase</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Pisco</surname><given-names>Angela Oliveira</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Quake</surname><given-names>Stephen</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Applied Physics, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Salzman</surname><given-names>Julia</given-names></name><aff><institution>Howard Hughes Medical Institute</institution><addr-line><named-content content-type="city">Chevy Chase</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Yosef</surname><given-names>Nir</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff><institution>Center for Computational Biology, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Electrical Engineering and Computer Sciences, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff><aff><institution>Ragon Institute of MGH, MIT and Harvard</institution><addr-line><named-content content-type="city">Cambridge</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Bulthaup</surname><given-names>Bryan</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Brown</surname><given-names>Phillip</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Harper</surname><given-names>Will</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Hemenez</surname><given-names>Marisa</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Ponnusamy</surname><given-names>Ravikumar</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Salehi</surname><given-names>Ahmad</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sanagavarapu</surname><given-names>Bhavani</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Spallino</surname><given-names>Eileen</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Aaron</surname><given-names>Ksenia A</given-names></name><aff><institution>Department of Otolaryngology-Head and Neck Surgery, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Concepcion</surname><given-names>Waldo</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Gardner</surname><given-names>James M</given-names></name><aff><institution>Department of Surgery, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff><institution>Diabetes Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kelly</surname><given-names>Burnett</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff><aff><institution>DCI Donor Services</institution><addr-line><named-content content-type="city">Sacramento</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Neidlinger</surname><given-names>Nikole</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wang</surname><given-names>Zifa</given-names></name><aff><institution>Donor Network West</institution><addr-line><named-content content-type="city">San Ramon</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Crasta</surname><given-names>Sheela</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kolluru</surname><given-names>Saroja</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Morri</surname><given-names>Maurizio</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Tan</surname><given-names>Serena Y</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Travaglini</surname><given-names>Kyle J</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Xu</surname><given-names>Chenling</given-names></name><aff><institution>Center for Computational Biology, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Alcántara-Hernández</surname><given-names>Marcela</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Almanzar</surname><given-names>Nicole</given-names></name><aff><institution>Department of Pediatrics - Pulmonary Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Antony</surname><given-names>Jane</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Beyersdorf</surname><given-names>Benjamin</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Burhan</surname><given-names>Deviana</given-names></name><aff><institution>Department of Medicine and Liver Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Calcuttawala</surname><given-names>Kruti</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Carter</surname><given-names>Mathew</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Chan</surname><given-names>Charles KF</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Surgery - Plastic and Reconstructive Surgery, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Chang</surname><given-names>Charles A</given-names></name><aff><institution>Department of Developmental Biology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Colville</surname><given-names>Alex</given-names></name><aff><institution>Department of Neurology and Neurological Sciences, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Paul F. Glenn Center for the Biology of Aging, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Culver</surname><given-names>Rebecca</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Cvijović</surname><given-names>Ivana</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Applied Physics, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>D'Amato</surname><given-names>Gaetano</given-names></name><aff><institution>Department of Biology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Ezran</surname><given-names>Camille</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Galdos</surname><given-names>Francisco</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Gillich</surname><given-names>Astrid</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Goodyer</surname><given-names>William R</given-names></name><aff><institution>Department of Pediatrics, Division of Cardiology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Hang</surname><given-names>Yan</given-names></name><aff><institution>Department of Developmental Biology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Hayashi</surname><given-names>Alyssa</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Houshdaran</surname><given-names>Sahar</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Huang</surname><given-names>Xianxi</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Critical Care Medicine, The First Affiliated Hospital of Shantou University Medical College</institution><addr-line><named-content content-type="city">Shantou</named-content></addr-line><country>China</country></aff></contrib><contrib><name><surname>Irwin</surname><given-names>Juan</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Jang</surname><given-names>SoRi</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Juanico</surname><given-names>Julia Vallve</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kershner</surname><given-names>Aaron M</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kim</surname><given-names>Soochi</given-names></name><aff><institution>Department of Neurology and Neurological Sciences, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Paul F. Glenn Center for the Biology of Aging, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kiss</surname><given-names>Bernhard</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kong</surname><given-names>William</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kumar</surname><given-names>Maya E</given-names></name><aff><institution>Sean N. Parker Center for Asthma and Allergy Research, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Leylek</surname><given-names>Rebecca</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Li</surname><given-names>Baoxiang</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Liu</surname><given-names>Shixuan</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Loeb</surname><given-names>Gabriel</given-names></name><aff><institution>Division of Nephrology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Lu</surname><given-names>Wan-Jin</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Maltzman</surname><given-names>Jonathan</given-names></name><aff><institution>Division of Nephrology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Veterans Administration Palo Alto Health Care System and Department of Medicine</institution><addr-line><named-content content-type="city">Palo Alto</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Mantri</surname><given-names>Sruthi</given-names></name><aff><institution>Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Markovic</surname><given-names>Maxim</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>McAlpine</surname><given-names>Patrick L</given-names></name><aff><institution>Mass Spectrometry Platform, Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Metzger</surname><given-names>Ross</given-names></name><aff><institution>Department of Pediatrics, Division of Cardiology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Vera Moulton Wall Center for Pulmonary and Vascular Disease, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>de Morree</surname><given-names>Antoine</given-names></name><aff><institution>Department of Neurology and Neurological Sciences, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Paul F. Glenn Center for the Biology of Aging, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Mrouj</surname><given-names>Karim</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Mukherjee</surname><given-names>Shravani</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Muser</surname><given-names>Tyler</given-names></name><aff><institution>Department of Pediatrics - Pulmonary Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Neuhöfer</surname><given-names>Patrick</given-names></name><aff><institution>Stanford Cancer Institute, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Nguyen</surname><given-names>Thi</given-names></name><aff><institution>Division of Nephrology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Perez</surname><given-names>Kimberly</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Phansalkar</surname><given-names>Ragini</given-names></name><aff><institution>Department of Biology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Puluca</surname><given-names>Nazan</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Qi</surname><given-names>Zhen</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Rao</surname><given-names>Poorvi</given-names></name><aff><institution>Department of Medicine and Liver Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Raquer-McKay</surname><given-names>Hayley</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sasagawa</surname><given-names>Koki</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Schaum</surname><given-names>Nicholas</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Neurology and Neurological Sciences, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Scott</surname><given-names>Bronwyn Lane</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Seddighzadeh</surname><given-names>Bobak</given-names></name><aff><institution>Division of Hematology and Oncology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Segal</surname><given-names>Joe</given-names></name><aff><institution>Department of Medicine and Liver Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sen</surname><given-names>Sushmita</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Spencer</surname><given-names>Sean</given-names></name><aff><institution>Department of Medicine - Med/Gastroenterology and Hepatology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Steffes</surname><given-names>Lea</given-names></name><aff><institution>Department of Pediatrics - Pulmonary Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Subramaniam</surname><given-names>Varun R</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Swarup</surname><given-names>Aditi</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Swift</surname><given-names>Michael</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Van Treuren</surname><given-names>Will</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Trimm</surname><given-names>Emily</given-names></name><aff><institution>Department of Biology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Tsui</surname><given-names>Maggie</given-names></name><aff><institution>Department of Medicine and Liver Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Veizades</surname><given-names>Stefan</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Stanford Cardiovascular Institute</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>College of Medicine and Veterinary Medicine, University of Edinburgh</institution><addr-line><named-content content-type="city">Edinburgh</named-content></addr-line><country>United Kingdom</country></aff></contrib><contrib><name><surname>Vijayakumar</surname><given-names>Sivakamasundari</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Vo</surname><given-names>Kim Chi</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Vorperian</surname><given-names>Sevahn K</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Weinstein</surname><given-names>Hannah</given-names></name><aff><institution>Division of Hematology and Oncology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Winkler</surname><given-names>Juliane</given-names></name><aff><institution>Department of Cell &amp; Tissue Biology, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wu</surname><given-names>Timothy TH</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Xie</surname><given-names>Jamie</given-names></name><aff><institution>Division of Hematology and Oncology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Yung</surname><given-names>Andrea R</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Zhang</surname><given-names>Yue</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Detweiler</surname><given-names>Angela M</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Mekonen</surname><given-names>Honey</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Neff</surname><given-names>Norma</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sit</surname><given-names>Rene V</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Tan</surname><given-names>Michelle</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Yan</surname><given-names>Jia</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Bean</surname><given-names>Gregory R</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Berry</surname><given-names>Gerald J</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Charu</surname><given-names>Vivek</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Forgó</surname><given-names>Erna</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Martin</surname><given-names>Brock A</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Ozawa</surname><given-names>Michael G</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Silva</surname><given-names>Oscar</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Toland</surname><given-names>Angus</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Vemuri</surname><given-names>Venkata NP</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Afik</surname><given-names>Shaked</given-names></name><aff><institution>Center for Computational Biology, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Bierman</surname><given-names>Rob</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Botvinnik</surname><given-names>Olga Borisovna</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Byrne</surname><given-names>Ashley</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Chen</surname><given-names>Michelle</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Dehghannasiri</surname><given-names>Roozbeh</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Gayoso</surname><given-names>Adam</given-names></name><aff><institution>Center for Computational Biology, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Granados</surname><given-names>Alejandro A</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Li</surname><given-names>Qiqing</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Mahmoudabadi</surname><given-names>Gita</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>McGeever</surname><given-names>Aaron</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Olivieri</surname><given-names>Julia Eve</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Institute for Computational and Mathematical Engineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Park</surname><given-names>Madeline</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Ravikumar</surname><given-names>Neha</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Salzman</surname><given-names>Julia</given-names></name><aff><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Schmid</surname><given-names>Sandra L</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Stanley</surname><given-names>Geoff</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Tan</surname><given-names>Weilun</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Tarashansky</surname><given-names>Alexander J</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Vanheusden</surname><given-names>Rohan</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wang</surname><given-names>Sheng</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Xing</surname><given-names>Galen</given-names></name><aff><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Yosef</surname><given-names>Nir</given-names></name><aff><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Department of Electrical Engineering and Computer Sciences, University of California Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Dethlefsen</surname><given-names>Les</given-names></name><aff><institution>Division of Infectious Diseases &amp; Geographic Medicine, Department of Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Ho</surname><given-names>Po-Yi</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Irwin</surname><given-names>Juan C</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kumar</surname><given-names>Maya E</given-names></name><aff><institution>Department of Pediatrics - Pulmonary Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kuo</surname><given-names>Angera H</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Neuhöfer</surname><given-names>Patrick</given-names></name><aff><institution>Stanford Cancer Institute, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Perez</surname><given-names>Kimberly</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Raquer-McKay</surname><given-names>Hayley</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sinha</surname><given-names>Rahul</given-names></name><aff><institution>Department of Pathology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Stanford Cancer Institute, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Song</surname><given-names>Hanbing</given-names></name><aff><institution>Division of Hematology and Oncology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Spencer</surname><given-names>Sean</given-names></name><aff><institution>Department of Medicine - Med/Gastroenterology and Hepatology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wang</surname><given-names>Bruce</given-names></name><aff><institution>Department of Medicine and Liver Center, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Winkler</surname><given-names>Juliane</given-names></name><aff><institution>Department of Cell &amp; Tissue Biology, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Artandi</surname><given-names>Steven E</given-names></name><aff><institution>Department of Biochemistry, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Stanford Cancer Institute, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Beachy</surname><given-names>Philip</given-names></name><aff><institution>Department of Developmental Biology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Clarke</surname><given-names>Michael F</given-names></name><aff><institution>Institute for Stem Cell Biology and Regenerative Medicine, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Giudice</surname><given-names>Linda</given-names></name><aff><institution>Center for Gynecology and Reproductive Sciences, Department of Obstetrics, Gynecology and Reproductive Sciences, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Huang</surname><given-names>Franklin</given-names></name><aff><institution>Division of Hematology and Oncology, Department of Medicine, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Huang</surname><given-names>Kerwyn Casey</given-names></name><aff><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Idoyaga</surname><given-names>Juliana</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kim</surname><given-names>Seung K</given-names></name><aff><institution>Department of Developmental Biology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Krasnow</surname><given-names>Mark</given-names></name><aff><institution>Howard Hughes Medical Institute</institution><addr-line><named-content content-type="city">Chevy Chase</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Kuo</surname><given-names>Christin</given-names></name><aff><institution>Department of Pediatrics - Pulmonary Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Nguyen</surname><given-names>Patricia</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Veterans Administration Palo Alto Health Care System and Department of Medicine</institution><addr-line><named-content content-type="city">Palo Alto</named-content></addr-line><country>United States</country></aff><aff><institution>Stanford Cardiovascular Institute</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Rando</surname><given-names>Thomas A</given-names></name><aff><institution>Paul F. Glenn Center for the Biology of Aging, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Red-Horse</surname><given-names>Kristy</given-names></name><aff><institution>Department of Biology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Reiter</surname><given-names>Jeremy</given-names></name><aff><institution>Department of Biochemistry, University of California San Francisco</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Sonnenburg</surname><given-names>Justin</given-names></name><aff><institution>Department of Microbiology and Immunology, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wu</surname><given-names>Albert</given-names></name><aff><institution>Department of Ophthalmology, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wu</surname><given-names>Sean</given-names></name><aff><institution>Department of Medicine, Division of Cardiovascular Medicine, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff><institution>Stanford Cardiovascular Institute</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib><name><surname>Wyss-Coray</surname><given-names>Tony</given-names></name><aff><institution>Paul F. Glenn Center for the Biology of Aging, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></collab><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-169216"><name><surname>Quake</surname><given-names>Stephen R</given-names></name><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="aff" rid="aff10">10</xref><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" id="author-34531"><name><surname>Krasnow</surname><given-names>Mark A</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con10"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes" id="author-88696"><name><surname>Salzman</surname><given-names>Julia</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7630-3436</contrib-id><email>julia.salzman@stanford.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con11"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Institute for Computational and Mathematical Engineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Biomedical Data Science, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>Department of Biochemistry, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution>Department of Neurology and Neurological Sciences, Stanford University School of Medicine</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution>Department of Pathology, Stanford University Medical Center</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff6"><label>6</label><institution>Academy for Statistics and Interdisciplinary Sciences, Faculty of Economics and Management,East China Normal University</institution><addr-line><named-content content-type="city">Shanghai</named-content></addr-line><country>China</country></aff><aff id="aff7"><label>7</label><institution>Department of Mathematics, The Hong Kong University of Science and Technology</institution><addr-line><named-content content-type="city">Hong Kong</named-content></addr-line><country>China</country></aff><aff id="aff8"><label>8</label><institution>Department of Chemical and Biological Engineering, The Hong Kong University of Science and Technology</institution><addr-line><named-content content-type="city">Hong Kong</named-content></addr-line><country>China</country></aff><aff id="aff9"><label>9</label><institution>Chan Zuckerberg Biohub</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff id="aff10"><label>10</label><institution>Department of Bioengineering, Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Yeo</surname><given-names>Gene W</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, San Diego</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Senior Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date date-type="publication" publication-format="electronic"><day>13</day><month>09</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e70692</elocation-id><history><date date-type="received" iso-8601-date="2021-05-26"><day>26</day><month>05</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2021-09-10"><day>10</day><month>09</month><year>2021</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2021-05-02"><day>02</day><month>05</month><year>2021</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2021.05.01.442281"/></event></pub-history><permissions><copyright-statement>© 2021, Olivieri et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Olivieri et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-70692-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-70692-figures-v2.pdf"/><abstract><p>The extent splicing is regulated at single-cell resolution has remained controversial due to both available data and methods to interpret it. We apply the SpliZ, a new statistical approach, to detect cell-type-specific splicing in &gt;110K cells from 12 human tissues. Using 10X Chromium data for discovery, 9.1% of genes with computable SpliZ scores are cell-type-specifically spliced, including ubiquitously expressed genes <italic>MYL6</italic> and <italic>RPS24</italic>. These results are validated with RNA FISH, single-cell PCR, and Smart-seq2. SpliZ analysis reveals 170 genes with regulated splicing during human spermatogenesis, including examples conserved in mouse and mouse lemur. The SpliZ allows model-based identification of subpopulations indistinguishable based on gene expression, illustrated by subpopulation-specific splicing of classical monocytes involving an ultraconserved exon in <italic>SAT1</italic>. Together, this analysis of differential splicing across multiple organs establishes that splicing is regulated cell-type-specifically.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>scRNA-seq</kwd><kwd>splicing</kwd><kwd>statistics</kwd><kwd>computational biology</kwd><kwd>RNA</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd><kwd>Mouse</kwd><kwd>Mouse lemur</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>DGE-1656518</award-id><principal-award-recipient><name><surname>Olivieri</surname><given-names>Julia Eve</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000057</institution-id><institution>National Institute of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R01 GM116847</award-id><principal-award-recipient><name><surname>Salzman</surname><given-names>Julia</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>MCB1552196</award-id><principal-award-recipient><name><surname>Salzman</surname><given-names>Julia</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T15 LM7033-36</award-id><principal-award-recipient><name><surname>Dehghannasiri</surname><given-names>Roozbeh</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>R25 CA180993</award-id><principal-award-recipient><name><surname>Dehghannasiri</surname><given-names>Roozbeh</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Comprehensive analysis of alternative splicing from human droplet-based scRNA-seq data identifies genes with regulated splicing conserved in mouse and mouse lemur.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Isoform-specific RNA expression is conserved in higher eukaryotes (<xref ref-type="bibr" rid="bib37">Merkin et al., 2012</xref>), tissue-specific, and controls developmental (<xref ref-type="bibr" rid="bib4">Baralle and Giudice, 2017</xref>; <xref ref-type="bibr" rid="bib28">Keren et al., 2010</xref>; <xref ref-type="bibr" rid="bib52">Ule and Blencowe, 2019</xref>; <xref ref-type="bibr" rid="bib57">Zhang et al., 2016</xref>) and myriad cell signaling pathways (<xref ref-type="bibr" rid="bib22">Hartmann et al., 2009</xref>; <xref ref-type="bibr" rid="bib35">Martinez et al., 2012</xref>). Alternative splicing also plays a major functional role as it expands proteomic complexity and rewires protein interaction networks (<xref ref-type="bibr" rid="bib8">Buljan et al., 2012</xref>; <xref ref-type="bibr" rid="bib11">Ellis et al., 2012</xref>). Alternative RNA isoforms of the same gene can even be translated into proteins with opposite functions (<xref ref-type="bibr" rid="bib56">Yang et al., 2016</xref>). Splicing is dysregulated in many diseases from neurological disorders to cancers (<xref ref-type="bibr" rid="bib2">Anczuków and Krainer, 2016</xref>). Alternative splicing studies have been mostly limited to bulk-level analysis, and they have shown evidence that as many as one-third of all human genes express tissue-dependent dominant isoforms, while most highly expressed human genes express a single dominant isoform in different tissues (<xref ref-type="bibr" rid="bib12">Ezkurdia et al., 2015</xref>; <xref ref-type="bibr" rid="bib18">Gonzàlez-Porta et al., 2013</xref>). It has been known for decades that genes can have cell-type-specific splicing patterns, best characterized in the immune, muscle, and nervous systems (<xref ref-type="bibr" rid="bib13">Florea et al., 2013</xref>; <xref ref-type="bibr" rid="bib17">Giudice et al., 2016</xref>; <xref ref-type="bibr" rid="bib32">Li et al., 2007</xref>; <xref ref-type="bibr" rid="bib35">Martinez et al., 2012</xref>; <xref ref-type="bibr" rid="bib59">Zipursky and Sanes, 2010</xref>). But the extent of cell-type-specific splicing is still controversial, partly because it has only been studied indirectly through profiling tissues, which is confounded by differential cell type composition. Many other questions remain such as whether cells of the same type in different tissues have shared splicing programs.</p><p>Determining how splicing is regulated in single cells could improve predictive models of splice isoform expression and move toward systems-level prediction of function. Furthermore, single-cell RNA splicing analysis has tremendous implications for biomedicine. Drugs targeting ‘genes’ may actually target only a subset of isoforms of the gene, and it is critically important to know which cells express these isoforms to predict on- and off-target drug interactions.</p><p>Genome-wide characterization of cell-type-specific splicing is still lacking mainly due to inherent challenges in scRNA-seq such as data sparsity. The field still debates whether single-cell splicing heterogeneity constitutes another layer of splicing regulation or is dominated by stochastic but stereotyped ‘binary’ exon inclusion (<xref ref-type="bibr" rid="bib3">Arzalluz-Luque and Conesa, 2018</xref>; <xref ref-type="bibr" rid="bib7">Buen Abad Najar et al., 2020</xref>), and whether cells’ spliced RNA is sequenced deeply enough in scRNA-seq for biologically meaningful inference. Most differential splicing analysis requires isoform estimation, which is unreliable with low or biased counts (<xref ref-type="bibr" rid="bib55">Westoby et al., 2020</xref>), or ‘percent spliced in’ (PSI) point estimates, which suffer from high variance at low read depth and amplify the multiple hypothesis testing problem (<xref ref-type="bibr" rid="bib3">Arzalluz-Luque and Conesa, 2018</xref>; <xref ref-type="bibr" rid="bib7">Buen Abad Najar et al., 2020</xref>). Most methods for splicing analysis from scRNA-seq data are not designed for droplet-based data (<xref ref-type="bibr" rid="bib25">Huang and Sanguinetti, 2017</xref>; <xref ref-type="bibr" rid="bib45">Song et al., 2017</xref>). Studies of splicing in scRNA-seq data have mostly focused on just a single cell type or organ and used pseudo-bulked data before differential splicing is analyzed, thus do not provide the potential to discover new subclusters or provide bona fide quantification of splicing at single-cell resolution. Further, studies have almost exclusively used full-length data such as Smart-seq2 (SS2) (<xref ref-type="bibr" rid="bib3">Arzalluz-Luque and Conesa, 2018</xref>; <xref ref-type="bibr" rid="bib7">Buen Abad Najar et al., 2020</xref>). Without genome-wide resolution, global splicing trends are missed and the focus on full-length sequencing data means that single cells sequenced with droplet-based technology, the majority of sequenced single cells including many cell types that are not captured by SS2 (<xref ref-type="bibr" rid="bib47">Svensson et al., 2020</xref>; <xref ref-type="bibr" rid="bib51">Travaglini et al., 2020</xref>), have been neglected (<xref ref-type="bibr" rid="bib39">Patrick et al., 2020</xref>).</p><p>To overcome statistical challenges that have prevented analysis of cell-type-specific alternative splicing, we used the SpliZ (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>), a statistical approach that generalizes PSI (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, Materials and methods) and increases the power to detect cell-type-specific alternative splicing in single cells. As detailed in <xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>, for each gene, the SpliZ quantifies splicing deviation in each cell from the population average. A large negative (resp. positive) SpliZ score for a gene in a cell means that the cell has shorter (resp. longer) introns than the average for that gene. Highlighting its disciplined statistical nature, the SpliZ reduces to PSI in the simplest exon skipping case (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Analysis of alternative splicing in single-cell RNA-seq.</title><p>(<bold>A</bold>) Human, mouse lemur, and mouse single-cell RNA-seq from 10X and SS2 were run through the SpliZ pipeline for differential splicing discovery. (<bold>B</bold>) 10X data from the first human individual contains 82 cell types with variable sequencing depth. (<bold>C</bold>) Given cell type annotation, SpliZ scores can be aggregated for each cell type, allowing identification of cell types with statistically deviant splicing. (<bold>D</bold>) Cell-wise SpliZ values can be correlated with pseudotime to identify developmentally regulated alternative splicing. (<bold>E</bold>) The fraction of genes called as having significant differential alternative splicing by cell type is higher at higher sequencing depths, plateauing at around 20,000 spliced reads in the dataset, at which point around 15%of genes were called as significant. (<bold>F</bold>) The SpliZ is calculated independently for SS2 data restricted to junctions found in 10Xdata, and used to validate results from 10X data.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig1-v2.tif"/><permissions><copyright-statement>© 2021, Olivieri et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Olivieri et al</copyright-holder><license><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>Panels C and F were also included as Figures 1C and 2D, respectively, in <xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>, published under the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution Non-Commercial No Derivatives 4.0 International License</ext-link>.</license-p></license></permissions></fig><p>When cell type annotations are available, the SpliZ statistically identifies genes with cell-type-specific splicing patterns. The SpliZ is an unbiased and annotation-free algorithm and is applicable to both droplet-based and full-length scRNA-seq technologies. The SpliZ attains high power to detect differential alternative splicing in scRNA-seq when genes are variably and sparsely sampled (<xref ref-type="fig" rid="fig1">Figure 1B</xref>) by controlling for sparsely sampled counts and technological biases such as those introduced by 10X Chromium (10X). Because the SpliZ gives a single value for each gene and each cell, it enables analyses beyond differential splicing between cell types, including correlation of splicing changes with developmental trajectories and subcluster discovery within cell types based on splicing differences (<xref ref-type="fig" rid="fig1">Figure 1C and D</xref>). It also provides a statistical, completely annotation-free approach that identifies splice sites called SpliZsites that contribute most variation to cell-type-specific splicing as measured by analysis of SpliZ components through the singular value decomposition (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>).</p><p>Here, we used the SpliZ to analyze 75,789 cells profiled with 10X across 12 tissues and 82 cell types from one human individual through the <italic>Tabula Sapiens</italic> project (<xref ref-type="bibr" rid="bib50">Tabula Sapiens Consortium, 2021</xref>). We also performed SpliZ analysis on a second human and two mouse lemur and two mouse individuals: together we analyzed 109,981 human (<xref ref-type="bibr" rid="bib50">Tabula Sapiens Consortium, 2021</xref>), 165,200 mouse lemur (<xref ref-type="bibr" rid="bib48">Tabula Microcebus Consortium, 2021</xref>), and 14,700 mouse cells (<xref ref-type="bibr" rid="bib49">Tabula Muris Consortium, 2018</xref>) sequenced with 10X (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). Additionally, we analyzed spermatogenesis trajectories across 4490, 4342, and 5601 10X sperm cells from human (<xref ref-type="bibr" rid="bib23">Hermann et al., 2018</xref>), mouse (<xref ref-type="bibr" rid="bib23">Hermann et al., 2018</xref>), and mouse lemur (<xref ref-type="bibr" rid="bib48">Tabula Microcebus Consortium, 2021</xref>), respectively. The SpliZ has higher power to detect differential alternative splicing between cell types at higher sequencing depths, plateauing at around 20,000 spliced reads measured for the gene, at which point around 15% of genes were called as significant (<xref ref-type="fig" rid="fig1">Figure 1E</xref>).</p><p>We performed high-throughput computational validation with the SS2 cells (<xref ref-type="fig" rid="fig1">Figure 1F</xref>) along with experimental and in situ validations including Sanger sequencing and RNA FISH on cells from the lung and muscle. Mouse and mouse lemur data was used to assess evolutionary conservation of the discoveries in human. Examples found by this analysis include differential cell-type-specific and compartment-specific alternative splicing in a subset of ubiquitously expressed genes including <italic>MYL6</italic>, an actin light chain subunit, <italic>RPS24</italic>, a core ribosomal subunit associated with Diamond-Blackfan Anemia (<xref ref-type="bibr" rid="bib21">Gupta and Warner, 2014</xref>), and <italic>TPM1</italic>, a tumor suppressor tropomyosin. Knockout studies of RNA-binding proteins have implied the importance of alternative splicing in spermatogenesis; however, comprehensive profiling of alternative splicing in normal spermatogenesis has not been possible. In this study, for the first time we identify regulated splicing changes in 170 genes during normal human spermatogenesis using rigorous statistical methodology for automatic computational single-cell splicing profiling, including conserved regulated splicing in centrosomal protein domain and lncRNA.</p><p>To our knowledge, this work provides the first unbiased and systematic screen for cell-type-specific splicing regulation in highly resolved human cells, predicting functionally significant alternative splicing, and calls for more attention to the potential of scRNA-seq for discovering regulated splicing in single cells.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Conserved splicing in ubiquitously expressed genes, including <italic>ATP5FC1</italic> and <italic>RPS24,</italic> predicts cellular compartment at single-cell resolution</title><p>We applied the SpliZ (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>), a recently developed method to identify cell-type-specific splicing, to ~75k 10X cells in 12 tissues from one human donor (<xref ref-type="bibr" rid="bib50">Tabula Sapiens Consortium, 2021</xref>), beginning by testing for splicing regulation differing by tissue compartment (immune, epithelial, endothelial, and stromal) regardless of the tissue of origin (SpliZ scores available for download at the following FigShare repository: DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.14531721">10.6084</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.14531721">/m9</ext-link><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.14531721">.figshare.14531721</ext-link>). This analysis identified 1.6% (22 of 1353) of genes with computable SpliZ scores as having consistent compartment-specific splicing effects (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>, Materials and methods). <italic>ATP5F1C</italic>, <italic>RPS24</italic>, and <italic>MYL6</italic> have the highest compartment-specific splicing effects, defined as the largest magnitude median SpliZ in any compartment, and their compartment-specific splicing was conserved in mouse and mouse lemur. <italic>ATP5F1C</italic> is the gamma subunit of mitochondrial ATP synthase, a multi-subunit molecular motor that converts the energy of the proton potential across the mitochondrial membrane into ATP. <italic>MYL6</italic> is an actin light chain subunit known to have cell-type-specific splicing differences in the muscle (<xref ref-type="bibr" rid="bib6">Brozovich et al., 2016</xref>). <italic>RPS24</italic> is an essential ribosomal protein for ribosome small subunit 40S discussed in detail later. Among the examples of genes demonstrating compartment-specific splicing is <italic>LIMCH1</italic> (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). <italic>LIMCH1</italic> has been reported as a non-muscle myosin regulator (<xref ref-type="bibr" rid="bib34">Lin et al., 2017</xref>) and has been associated with Huntington’s disease (<xref ref-type="bibr" rid="bib33">Lin et al., 2016</xref>) with little other characterization, including, to our knowledge, no reports of regulated splicing. The SpliZ values for <italic>MYL6, RPS24,</italic> and <italic>ATP5F1C</italic> are not correlated with gene expression (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplements 1</xref>–<xref ref-type="fig" rid="fig2s3">3</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Compartment-specific alternative splicing revealed by applying the SpliZ to scRNA-seq data.</title><p>(<bold>A</bold>) Dot and sashimi plots showing <italic>LIMCH1</italic> compartment-specific exon skipping (involving 5′ splice site [5′ SS] 41,619,440 and two 3′ splice sites [3′ SS] 41,638,932 and 41,644,500) impacting a protein domain of unknown function DUF4757 (shown by the purple color on the gene structure) across cell types and 10X and SS2 data from both human individuals. Each dot shows junction expression for the splice junction from the 5′ SS to one of the 3′ SS’s, with dot size proportional to the fraction of junctional reads supporting the splice junction in that cell type and dataset. Columns of dots are biological replicates; the first column is the individual 1 10X dataset (circles) and the next two columns are SS2 datasets from individuals 1 and 2 (squares). Cell types are grouped in two sets depending on the sign of the median SpliZ score in 10X data from human individual 1. The thickness of the sashimi arcs represents the fraction of the reads mapping to each 3′ SS when all datasets and corresponding cell types for the sashimi arc are grouped together. The box plot for each cell type shows the distribution of the weighted average 3′ SS (weights being the number of reads aligning to each 3′ SS in the cell) for each cell and the reads are assigned 1 (for those aligning to the closer 3′ SS) and 2 (for those aligning to the farther 3′ SS). Stromal cells including vasculature smooth muscle cells and fibroblasts always include the exon (higher fraction of reads aligning to the splice site at 41,638,932), while epithelial cells including bladder urothelial cells skip with &gt;50% rate. (<bold>B</bold>) Unsupervised k means clustering results in 78, 84, and 95% accuracy of compartment classification for the stromal, epithelial, and immune compartments, respectively, for individual 1, and 70, 100, and 49%, respectively, for individual 2. (<bold>C</bold>) The SpliZ scores of the genes <italic>ATP5F1C</italic> and <italic>RPS24</italic> together separate compartments in both human individuals. Each dot represents the SpliZ score in a single cell and is color coded by the compartment. (<bold>D</bold>) Using the spliced read counts for each gene rather than the SpliZ does not separate the compartments, showing that this separation is not driven by gene expression differences. Each dot represents the number of spliced reads in a single cell and is color coded by the compartment.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig2-v2.tif"/><permissions><copyright-statement>© 2021, Olivieri et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Olivieri et al</copyright-holder><license><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This figure was also included as Figure 2D in <xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>, published under the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution Non-Commercial No Derivatives 4.0 International License</ext-link>.</license-p></license></permissions></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Gene expression and the SpliZ value of <italic>MYL6</italic> across single cells in the <italic>Tabula Sapiens</italic> dataset.</title><p>Coloring <italic>Tabula Sapiens</italic> cells by both gene expression and SpliZ value shows that <italic>MYL6</italic> is ubiquitously expressed and that the SpliZ is independent of gene expression for these cases. Plots are obtained by using the cellxgene (<xref ref-type="bibr" rid="bib36">Megill et al., 2021</xref>) visualization platform.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Gene expression and the SpliZ value of <italic>RPS24</italic> across single cells in the <italic>Tabula Sapiens</italic> dataset.</title><p>Coloring <italic>Tabula Sapiens</italic> cells by both gene expression and SpliZ value shows that <italic>RPS24</italic> is ubiquitously expressed and that the SpliZ is independent of gene expression for these cases. Plots are obtained by using the cellxgene (<xref ref-type="bibr" rid="bib36">Megill et al., 2021</xref>) visualization platform.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig2-figsupp2-v2.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Gene expression and the SpliZ value of <italic>ATP5F1C</italic> across single cells in the <italic>Tabula Sapiens</italic> dataset.</title><p>Coloring <italic>Tabula Sapiens</italic> cells by both gene expression and SpliZ value shows that <italic>ATP5F1C</italic> is ubiquitously expressed and that the SpliZ is independent of gene expression for these cases. Plots are obtained by using the cellxgene (<xref ref-type="bibr" rid="bib36">Megill et al., 2021</xref>) visualization platform.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig2-figsupp3-v2.tif"/></fig></fig-group><p>To test the predictive power of compartment-specific genes at single-cell resolution, we performed unsupervised k-means clustering on the SpliZ scores of <italic>RPS24</italic> and <italic>ATP5F1C</italic> alone. Setting <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, cells from stromal, epithelial, and immune compartments in the first human individual were classified with accuracies of 78, 84, and 95%, respectively, independent of gene expression (70, 100, and 49% in the second individual) (<xref ref-type="fig" rid="fig2">Figure 2B–D</xref>, Materials and methods). The lower accuracy for individual 2 may be caused by individual 2 having only a third as many cells. The endothelial compartment was not included because it had a small proportion of cells in both datasets (3.7% in individual 1, 4.5% in individual 2). This establishes that splicing of a minimal set of genes, in this case only two, has high predictive power of the compartmental origin of each single cell. Underlining tight biological regulation of splicing in these genes, parallel analysis in the 10X scRNA-seq data from mouse lemur and mouse shows compartment-specific splicing is conserved for <italic>RPS24</italic> and <italic>MYL6</italic> (<xref ref-type="fig" rid="fig3">Figures 3</xref> and <xref ref-type="fig" rid="fig4">4</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Compartment-specific alternative splicing in <italic>MYL6</italic>.</title><p>Differential alternative splicing between compartments for <italic>MYL6</italic> is driven by an exon skipping event with orthologous splice sites (SS) in (<bold>A</bold>) human (5′ SS: 56,160,320 and two 3′ SSs: 56,161,387 and 56,160,626), (<bold>B</bold>) mouse lemur, and (<bold>C</bold>) mouse. Each dot shows the expression for the splicing to one of the 3′ SSs marked by vertical red lines on the gene annotation in (<bold>D</bold>) in a 10X (circles) or SS2 (squares) dataset from individuals 1 and 2. Columns of dots are biological replicates; for human data, the first two columns are 10X and the second two columns are SS2. Dots are colored by compartment. For mouse and mouse lemur, the two columns are 10X samples. The box plot is obtained by assigning 1 and 2 to the closer and farther 3′ SS and then computing their weighted average for each cell according to their corresponding fraction of junctional reads in the cell. Cells in the immune compartment have higher exon skipping rates than cells in the stromal compartment in all three organisms. Smooth muscle cell types are boxed within the stromal compartment. Mouse cells have the same relative proportions of exon inclusion between compartments, but express higher levels of the exon included isoform overall. The SpliZ scores (and also gene expression values) for <italic>MYL6</italic> across all 10X cells in human individual 1 are shown in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>. (<bold>D</bold>) Gene structures showing <italic>MYL6</italic> annotation in human, mouse, and mouse lemur. The gray arrows between different organisms show LiftOver mapping between human, mouse, and mouse lemur, indicating that orthologous splice sites are involved in alternative splicing in different organisms. (<bold>E</bold>) Protein domains in MYL6 and how they are organized in the two <italic>MYL6</italic> isoforms. The exon skipping leads to the deletion in the EF_hand_8 domain (shown by the red color). (<bold>F</bold>) RNA FISH validation in human lung: each slide is stained simultaneously with probes in red (specific to exon inclusion) and brown (specific to exon exclusion). As found from the scRNA-seq data, smooth muscle cells have a higher proportion of the included exon than the other compartments and immune cells have the lowest proportion.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Uncropped plots of <italic>MYL6</italic> expression.</title><p>(<bold>A</bold>) Human, (<bold>B</bold>) lemur, and (<bold>C</bold>) mouse for the splice sites shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>FISH validation for <italic>MYL6</italic> alternative splicing in cells isolated from human muscle.</title><p>Indicated cell types were isolated from human muscle and stained by RNA FISH. Example images are shown on the left, with the exon 6 isoform shown in red, the 5–7 isoform shown in gray in the DIC channel, and DAPI shown in blue. Graph depicting the quantifications is shown on the right.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig3-figsupp2-v2.tif"/></fig></fig-group><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>RPS24 has striking compartment-specific alternative splicing.</title><p>(<bold>A</bold>) Each colored circle in the plot represents one <italic>RPS24</italic> junction that uniquely identifies an isoform (junction with green circle represents two isoforms). For each cell type (y axis), the median of all single-cell point estimates of junction fraction in the cell type is plotted on the x axis, with bars representing the 25th and 75th quantiles of single-cell junction fractions. Cell types are sorted by compartment. Black arrows show the splice junctions that can be used for identifying each isoform. Within the immune compartment, the fraction of the blue junction increases from classical monocytes to intermediate monocytes to non-classical monocytes. (<bold>B</bold>) The isoform with epithelial-specific splicing in human is not expressed in mouse lemur. However, the same isoform is expressed in smooth muscle as in human. Retinal cells are the only cells to express the +<italic>a</italic>+<italic>b</italic>+<italic>c</italic> isoform. (<bold>C</bold>) <italic>RPS24</italic> isoform structure in human shows alternative inclusion of three cassette exons <italic>a</italic>, <italic>b</italic>, and <italic>c</italic> create five annotated isoforms. (<bold>D</bold>) Single-cell PCR validates the prediction that the +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is epithelial-specific. All the epithelial cells contain the isoform with the 3-nt exon <italic>a</italic>, while none of the cells from other compartments do. PCR products from the cells prefixed by asterisks were Sanger-sequenced and matched the expected isoform without evidence of mixture. (<bold>E</bold>) <italic>RPS24</italic> FISH in human lung validates scRNA-seq computational predictions. Slides were simultaneously stained with probes in red and brown, specific for alternative splice junctions. As found from the scRNA-seq data, respiratory epithelium and bronchiole smooth muscle in the epithelial and stromal compartments, respectively, have a low proportion of the -<italic>a</italic>-<italic>b</italic>-<italic>c</italic> isoform compared to alveolar macrophages and lymphocytes, both of which are in the immune compartment.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>FISH validation for <italic>RPS24</italic> alternative splicing in cells isolated from human muscle.</title><p>Indicated cell types were isolated from human muscle and stained by RNA FISH. Example images are shown on the left, with the -<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform shown in red, the -<italic>a</italic>-<italic>b</italic>-<italic>c</italic> isoform shown in gray in the DIC channel, and DAPI shown in blue. Graphs depicting the quantifications are shown on the right. Graph depicting the quantifications is shown on the right.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title><italic>RPS24</italic> isoforms quantified by the Bowtie2 aligner validate STAR-based discoveries.</title><p>Bar plots for the (<bold>A</bold>) endothelial, (<bold>B</bold>) epithelial, (<bold>C</bold>) immune, and (<bold>D</bold>) stromal compartments show proportions of each isoform for each species, with error bars corresponding to 95% binomial confidence intervals. Within epithelial cells the +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform was expressed at the highest fraction in human, while only at a small fraction in mouse and mouse lemur, confirming that this isoform’s epithelial specificity is not conserved in mouse and mouse lemur.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig4-figsupp2-v2.tif"/></fig></fig-group></sec><sec id="s2-2"><title>The splicing of actin regulator <italic>MYL6</italic> is compartment-specifically regulated</title><p>We identified <italic>MYL6</italic> as both cell-type-specifically and compartment-specifically spliced in humans and its splicing pattern is conserved (<xref ref-type="fig" rid="fig3">Figure 3</xref>). <italic>MYL6</italic> is a ubiquitously expressed myosin light chain subunit and is known to have a lower level of exon skipping in muscle than non-muscle tissue (<xref ref-type="bibr" rid="bib6">Brozovich et al., 2016</xref>), but differential exon skipping at a single-cell level has only been characterized in smooth muscle cells. We find in human, mouse, and mouse lemur that the stromal compartment, which includes smooth muscle, as well as the endothelial compartment have a relatively higher proportion of exon inclusion, while the epithelial compartment has a lower level of exon inclusion and the immune compartment has the lowest level of exon inclusion (<xref ref-type="fig" rid="fig3">Figure 3A–E</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). Despite these trends being the same for all three species, mouse has higher levels of exon inclusion in all compartments than in the other two species.</p><p>We validated compartment-specific differential alternative splicing in <italic>MYL6</italic> using RNA FISH with isoform-specific probes on human adult lung tissue obtained from the Stanford Tissue Bank (<xref ref-type="fig" rid="fig3">Figure 3F</xref>, Materials and methods). In human lung, this confirmed that bronchiole smooth muscle cells have the highest fraction of the exon inclusion isoform (57%), while the respiratory epithelium has a lower fraction of this isoform (16%) and the two profiled immune cell types (macrophages and lymphocytes) have the lowest fractions of the exon inclusion isoform (10% and 8%, respectively). We separately performed RNA FISH on human cells isolated from the muscle, which showed that mesenchymal stem cells and muscle stem cells have a higher proportion of the exon inclusion isoform than endothelial cells (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>, Materials and methods).</p></sec><sec id="s2-3"><title><italic>RPS24</italic> has compartmentally regulated alternative splicing and expresses a microexon in human epithelial cells</title><p><italic>RPS24</italic> is a highly expressed and essential ribosomal protein. Our analysis revealed that <italic>RPS24</italic> has the most significant cell-type-specific and compartment-specific alternative splicing patterns at its C terminus in human and mouse lemur. The significance of the alternative splicing patterns of <italic>RPS24</italic> is underscored by recent findings that ribosome composition is more modular than previously appreciated in a cell- and tissue-specific manner (<xref ref-type="bibr" rid="bib16">Genuth and Barna, 2018</xref>). There has been a partial study of <italic>RPS24</italic> splicing treating two isoforms (<xref ref-type="bibr" rid="bib45">Song et al., 2017</xref>), and another study reported modest differential splicing at the tissue level (<xref ref-type="bibr" rid="bib21">Gupta and Warner, 2014</xref>; <xref ref-type="bibr" rid="bib45">Song et al., 2017</xref>) for <italic>RPS24</italic> involving three isoforms. However, here, we show that splicing of <italic>RPS24</italic> is more complex and highly regulated at a single-cell level (<xref ref-type="fig" rid="fig4">Figure 4A and B</xref>).</p><p>Differential alternative splicing of <italic>RPS24</italic> involves alternate inclusion of three short exons, <italic>a</italic>, <italic>b</italic>, and <italic>c</italic>, each only 3, 18, and 22 nucleotides long, respectively (<xref ref-type="fig" rid="fig4">Figure 4C</xref>); regions of genomic sequence around exon <italic>a</italic> are ultraconserved. Splicing of the <italic>a</italic>, <italic>b</italic>, and <italic>c</italic> exons results in isoforms whose protein domains differ by the presence of a single lysine at the solvent-exposed site of the ribosome, and some isoforms (e.g., +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> and -<italic>a</italic>+<italic>b</italic>+<italic>c</italic>) have no change of amino acids. The -<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is dominant in all endothelial cell types in human, as well as most stromal cell types and half of the immune cell types (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). Within the immune compartment, our global analysis reveals differential alternative splicing of <italic>RPS24</italic> in monocytes, where the -<italic>a</italic>-<italic>b</italic>-<italic>c</italic> isoform is dominant in classical monocytes residing in multiple tissues and the -<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is dominant in non-classical monocytes. Intermediate monocytes have equal proportions of each.</p><p>Epithelial cell types in human are marked by the dominance of the +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform (as shown by the fraction of the pink splice junction in <xref ref-type="fig" rid="fig4">Figure 4A</xref>), which is barely present in any non-epithelial cell types and is not dominant in any of them, and only differs by three nucleotides from the -<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform. The +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is only found at very low levels in mouse and mouse lemur. The human epithelial specificity of the +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is further supported by single-cell RT-PCR (<xref ref-type="fig" rid="fig4">Figure 4D</xref>).</p><p>Other cell types have distinct isoform expression as well: the -<italic>a</italic>+<italic>b</italic>+<italic>c</italic> isoform is specific to fast and smooth muscle cells in both human and mouse lemur, such as thymus fast muscle and bladder smooth muscle in human, as well as vascular-associated bladder smooth muscle in mouse lemur, though some smooth muscle cell types in human do not express it, specifically thymus vascular-associated smooth muscle and vasculature smooth muscle. Among profiled cell types, the +<italic>a</italic>+<italic>b</italic>+<italic>c</italic> isoform is found only in neural retinal cells in the mouse lemur, the only dominant isoform including the microexon <italic>a</italic> in the mouse lemur (retina data not available for human).</p><p>In addition to using RNA FISH to independently validate cell-type-specific splicing in a subset of lung and muscle cells (<xref ref-type="fig" rid="fig4">Figure 4E</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>), we performed high-throughput validation using Bowtie2 alignment (<xref ref-type="bibr" rid="bib31">Langmead and Salzberg, 2012</xref>; <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>). The Bowtie2 alignment data confirms that the +<italic>a</italic>-<italic>b</italic>+<italic>c</italic> isoform is present at low levels in the epithelium of mouse and mouse lemur compared to high levels in human epithelium. The RNA FISH data confirms that the -<italic>a</italic>-<italic>b</italic>-<italic>c</italic> isoform composes just ~1% to 2% in the respiratory epithelium and bronchiole smooth muscle, while alveolar macrophages and lymphocytes have about 34–41% -<italic>a</italic>-<italic>b</italic>-<italic>c</italic> (<xref ref-type="fig" rid="fig4">Figure 4E</xref>).</p><p>Together, the subtle changes in protein sequence from alternative splicing of <italic>RPS24</italic> prompt two hypotheses: one is that splicing affects post-translational modifications (<xref ref-type="bibr" rid="bib29">Kondrashov et al., 2011</xref>). However, the fact that some splice variants have subtle or no variation in the encoded protein suggests an alternative that, like isoforms of <italic>Actin</italic> in mouse, <italic>RPS24</italic> splicing could function at the nucleotide rather than protein level (<xref ref-type="bibr" rid="bib53">Vedula et al., 2017</xref>).</p></sec><sec id="s2-4"><title>Approximately 9% of measured genes have cell-type-specific splicing regulation</title><p>Splicing regulation in the vast majority of human cell types has not been characterized. We used the 82 annotated cell types in the <italic>Tabula Sapiens</italic> cell atlas to identify genes with statistical support for having differential alternative splicing patterns using the same SpliZ procedure for identifying compartment-specific genes (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>; <xref ref-type="fig" rid="fig1">Figure 1C</xref>, Materials and methods, <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Among genes called significant, the Pearson correlation between the median SpliZ in individuals 1 and 2 (10X) was 0.77 and it was 0.44 between 10X and SS2 within individual 1 (p-value &lt; 10e-50, Materials and methods, <xref ref-type="fig" rid="fig5">Figure 5</xref>). 129 out of 1416 genes (9%) had significant cell-type-specific splicing profiles based on discovery with 10X data from individual 1 (p-value &lt; 0.05, effect size &gt;0.5) (Materials and methods, <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). Genes with cell-type-specific splicing regulation include <italic>TPM1</italic> (<xref ref-type="fig" rid="fig6">Figure 6</xref>), <italic>PNRC1</italic> (<xref ref-type="fig" rid="fig7">Figure 7A</xref>), and <italic>FYB</italic> (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>), among others.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Correlations show high concordance of the SpliZ values for significant genes across biological replicates.</title><p>(<bold>A</bold>) When subsetted to only shared junctions and shared cell types, the SpliZ values for significant genes for both 10X datasets are highly concordant (Pearson correlation of 0.776). (<bold>B</bold>) Comparing datasets from the 10X and SS2 technologies.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Choosing effect size filters.</title><p>Effect size filters were chosen based on correlation analysis of (<bold>A</bold>) Both human 10X datasets and (<bold>B</bold>) separately for SpliZVD by correlation of both human SS2 datasets.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig5-figsupp1-v2.tif"/></fig></fig-group><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Cell-type-specific and conserved alternative splicing in <italic>TPM1</italic>.</title><p>Conserved splicing in <italic>TPM1</italic> is recovered in (<bold>A</bold>) human, (<bold>B</bold>) mouse lemur, and (<bold>C</bold>) mouse. <italic>TPM1</italic> has a pattern of differential splicing involving two cassette exons and an alternate 5′ end (as shown by the gene structures at the bottom of the figure). Capillary endothelial cells mostly express the isoform with the alternate 5′ end (5′ splice site [5’ SS] 1), while smooth muscle almost exclusively expresses the isoform with the 5′-most domain (boxed in the figure). The box plot shows the distribution of the average 5′ SS (obtained as the weighted average of 5′ SS when ranked from 1 to 3 from the closest to the farthest according to their fraction of junctional reads) for the cells within a cell type (see <xref ref-type="fig" rid="fig2">Figures 2</xref> and <xref ref-type="fig" rid="fig3">3</xref> for more explanation of dot and box plots). There is differential isoform usage within the stromal compartment as well, for example, human bladder stromal fibroblasts and bladder stromal pericytes each express a different dominant cassette exon. Both lemur and mouse similarly express cell-type-specific differences in <italic>TPM1</italic> isoform usage. Orthologous SpliZsites in human, mouse, and mouse lemur are involved in alternative splicing based on the LiftOver mapping, as shown by gray arrows on the gene structures.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig6-v2.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Cell-type-specific splicing in other members of the tropomyosin family.</title><p>Both <italic>TPM2</italic> (<bold>A</bold>) and <italic>TPM3</italic> (<bold>B</bold>) exhibit cell-type-specific splicing patterns in human.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig6-figsupp1-v2.tif"/></fig></fig-group><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Cell-type-specific alternative splicing is prevalent in human genes and can reveal novel cell subpopulations.</title><p>(<bold>A</bold>) Cell-type-specific exon skipping in <italic>PNRC1</italic> (involving one 3′ SS and two 5′ SS’s) is replicated across the four human datasets. Skeletal muscle satellite stem cells include the exon about 50% of the time, whereas vein endothelial cells in the thymus never include the exon. Cell types with negative median SpliZ values are on the top panel, and those with positive median SpliZ values are on the bottom panel. Each of the four human datasets is plotted, with circles representing 10X data and squares representing SS2 data. The gene annotation is shown above the dots, with sashimi arcs indicating the mean expression of each junction for the given cell types and datasets in the corresponding panel (similar to <xref ref-type="fig" rid="fig2">Figure 2</xref>). The known protein domain is marked on the gene structure. Box plots for each cell type are based on weighted 3′ usage (based on the fraction of junctional reads to each 3′ SS) of the 5′ SS for each cell in the cell type in individual 1 10X data (<bold>L</bold>). Each box shows 25–75% quantiles of average 5′ SS per cell. (<bold>B</bold>) Unsupervised clustering analysis with the SpliZ identified clusters of cell types and compartments independent of tissue. Dots show the median SpliZ (effect size) for genes found to be significantly regulated across cell types. Only 50 significant genes with the highest effect size and cell types with &gt;25 significant genes are shown. Hierarchical clustering was performed on both genes and cell types based on median SpliZ values. Cell type names are color-coded based on their tissue (same tissue colors as in <bold>A</bold>) and the side bar shows the compartment for each cell type. (<bold>C</bold>) Alternative splicing of gene <italic>SAT1</italic> distinguishes two populations of cells within blood classical monocytes and involves an ultraconserved exon. The dot plot shows the differential inclusion of the 5′ SSs for the 3′ SS at 23,785,300 for cells grouped based on their assigned subclusters. The number of reads (X) and cells (Y) containing the splice junctions involving the 3′ SS in each individual are shown at right. Clustering based on gene expression as shown by cellxgene visualization (middle panel) and scatter plot (right panel) does not distinguish cell populations with distinct splice profiles. In the scatter plot, the x and y axes represent the gene expression and SpliZ values for <italic>SAT1</italic> in each cell, respectively. Cells are colored according to their human individual number. Visualization of the gene expression value for <italic>SAT1</italic> does not distinguish the populations; both subclusters contain classical monocytes from both human individuals (right scatter plot).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig7-v2.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>Differential alternative splicing of <italic>FYB1</italic> within the immune compartment.</title><p>The pattern is replicated in 10X and SS2 data from both human individuals and technologies, with monocytes and macrophages tending to include an exon (at position 39125998) that is usually skipped in T cells. (See <xref ref-type="fig" rid="fig3">Figure 3C</xref> for top-panel plot explanation.) At the bottom are diagrams of <italic>FYB1</italic> genomic organization and protein features. The exon structure is completely conserved between mouse and human, including the cassette exon; the conserved mapping of exons to protein sequence is shown for the C-terminal portion. Most of the protein is predicted to be unstructured, except for the distal SH3 and hSH3 domains. Numerous tyrosine (Y) phosphorylation sites are conserved and annotated in the expanded view of the protein. The cassette exon contains two predicted short alpha-helical stretches; it may allow additional protein interactions and/or may modulate accessibility of the two important phosphorylation sites that flank it.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig7-figsupp1-v2.tif"/></fig></fig-group><p>Tropomyosin 1, <italic>TPM1,</italic> which has three isoforms each impacting the tropomyosin domain at the 3′ end of the transcript, served as a positive control in this analysis as it is known to undergo cell-type-specific splicing. It is ranked as the 27th most significant effect size. Unbiased SpliZ analysis finds that capillary endothelial cells express about equal levels of three isoforms, while smooth muscle cells almost exclusively express the isoform with the 3′-most domain (<xref ref-type="fig" rid="fig6">Figure 6A–C</xref>). This trend, among others, is consistent with knowledge of <italic>TPM1</italic> splicing from other studies (<xref ref-type="bibr" rid="bib19">Gooding and Smith, 2008</xref>), though it extends its splicing profile to cell types where it has never been characterized. Among the comprehensive catalog of differences, smooth muscle and pericyte cells consistently include different cassette exons at the 3′ end of the transcript, affecting the tropomyosin protein domain (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). Cell types outside of muscle such as bladder pericytes and bladder fibroblasts have similar splicing profiles as smooth muscle and muscle mesenchymal stem cells, respectively. Splicing biology of <italic>TPM2</italic> and <italic>TPM3,</italic> two other genes from the <italic>TPM</italic> family where partial characterization has suggested cell-type-specific splicing, is similarly rediscovered in our analysis and significantly extended: slow muscle cells (and fast muscle cells for <italic>TPM2</italic>) have different splicing patterns than other cell types for both genes (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>).</p><p><italic>PNRC1</italic>, a nuclear receptor coregulator that functions as a tumor suppressor, has the fifth highest effect size (<xref ref-type="fig" rid="fig7">Figure 7A</xref>; <xref ref-type="bibr" rid="bib15">Gaviraghi et al., 2018</xref>); limited in vitro studies have found evidence that splice variants of <italic>PNRC1</italic> modify its interaction domains and nuclear functions (<xref ref-type="bibr" rid="bib54">Wang et al., 2008</xref>). The largest magnitude median SpliZ score for <italic>PNRC1</italic> is found in muscle stromal mesenchymal stem cells, revealing new splicing regulation. Other cell types, including immune and stromal types in the bladder, have markedly distinct splicing programs (<xref ref-type="fig" rid="fig7">Figure 7A</xref>).</p><p>The high dimensionality of SpliZ scores enabled us to test if unsupervised clustering on the median SpliZ scores could recapitulate relationships between cell types (<xref ref-type="fig" rid="fig7">Figure 7B</xref>). We found that the same cell types from different tissues are generally clustered together, including macrophages and T cells from different tissues and also intestinal cell types from large and small intestine (<xref ref-type="fig" rid="fig7">Figure 7B</xref>, Materials and methods). This clustering also reveals that the splicing programs of cell types from the same compartment are highly similar and automatically clustered together independent of their tissues (<xref ref-type="fig" rid="fig7">Figure 7B</xref>).</p></sec><sec id="s2-5"><title>The most statistically variable splice sites with cell-type-specific regulated splicing are annotated splice sites involved in unannotated alternative splicing</title><p>The biological importance of splicing detected by the SpliZ and the fact that it is completely agnostic to isoform annotation led us to test whether cell-type-specific splicing variation is (a) focused at exons that are annotated to undergo alternative splicing and (b) conserved. The SpliZ method uses a statistical, annotation-free approach to identify SpliZsites: variable splice sites that contribute most to the cell-type-specific splicing of a gene agnostic to gene annotation. SpliZsites blindly reidentify known alternative splice sites in <italic>ATP5F1C</italic>, <italic>MYL6</italic>, <italic>TPM1</italic>, and <italic>RPS24</italic> (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). In <italic>TPM2</italic>, the SpliZ reidentifies two known alternative splicing sites but also predicts a cell-type-specific unannotated alternative splicing event in stromal cell types involving 5′ splice site 35,684,315 affecting Tropomyosin and Tropomyosin 1 protein domains.</p><p>Genome-wide, the vast majority of SpliZsites (93%) in significant genes are at boundaries of annotated exons. However, only 38.5% are annotated as alternatively spliced, suggesting that unannotated – rather than annotated – exon skipping accounts for underappreciated splicing variation in single cells. Further, exon skipping has a global effect on single-cell proteomes: more than half of SpliZsites impact protein coding domains; 34% impact the 3′ UTR and 16% impact the 5′ UTR, consistent with a bias in 10X technology towards the 3′ gene end. Supporting the idea that SpliZsites discover a real biological signal, 15.5% of LiftOver human SpliZsites were also SpliZsites in the mouse lemur compared to 7% expected under the null (Materials and methods). Only 8.0% of LiftOver SpliZsites in human were called as SpliZsites in mouse compared to 8.8% expected under the null. This could be due to many factors including a larger evolutionary distance between mouse and human, smaller number of analyzed mouse cells, or lower sequencing depth (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>).</p></sec><sec id="s2-6"><title>The SpliZ identifies subpopulations of classical monocytes with distinct splicing of an ultraconserved exon of <italic>SAT1</italic></title><p>The SpliZ has a theoretical normal distribution under the assumption that cells within a cell type all have the same propensity to express each splice isoform (the ‘null hypothesis’). This property allows us to statistically test whether cell types subcluster on the basis of splice isoform, as quantified by the SpliZ, using an integrated complete-data likelihood (ICL) model selection framework. This is based on Gaussian mixture modeling (GMM), and includes a measure of ‘effect size’ differences between clusters via the Bhattacharyya distance, a measure of the distance between probability distributions (Materials and methods). Importantly, this approach avoids false-positive calls of apparent binary exon inclusion (<xref ref-type="bibr" rid="bib7">Buen Abad Najar et al., 2020</xref>; Materials and methods, manuscript in preparation). We applied the ICL analysis of the SpliZ to immune cell types in individual 2 to illustrate the power of single-cell splicing quantification by the SpliZ for defining subsets of cells within annotated cell types defined by gene expression. SpliZ values for <italic>SAT1</italic> in blood classical monocytes had the largest Bhattacharyya distance among identified subpopulations (Materials and methods).</p><p>Junctional reads (defined by SpliZsites) driving the separation of the two subpopulations show distinct isoform expression profiles (<xref ref-type="fig" rid="fig7">Figure 7C</xref>): cells in cluster 1 splice to a 5′ splice site that includes an ultraconserved genomic sequence, whereas those in cluster 2 contain splice to a different 5′ splice site (<xref ref-type="fig" rid="fig7">Figure 7C</xref>). These clusters are not driven by <italic>SAT1</italic> gene expression and are not detected by gene-based clustering of monocytes as shown by visualization in cellxgene (<xref ref-type="bibr" rid="bib36">Megill et al., 2021</xref>; <xref ref-type="fig" rid="fig7">Figure 7C</xref>). We used predictions of subpopulations of cells’ splicing profiles in <italic>SAT1</italic> in individual 2 to blindly test whether classical monocytes in individual 1 also exhibited evidence of subpopulations based on their splicing profile. The number of cells with reads from the same junction – supporting subpopulations – is significantly greater than the number expected under the null assumption of randomly sampling two reads per cell regardless of cluster (p-value &lt; 0.05, exact binomial test in individual 1, Materials and methods). Further supporting a biological role of <italic>SAT1</italic> splicing in the immune system, the same GMM-based approach identified two subpopulations of cells based on the SpliZ values for <italic>SAT1</italic> in both lung macrophages and thymus monocytes. Together with statistical support and blinded validation, this supports that <italic>SAT1</italic> exhibits splicing programs that define two splicing states within classical monocytes and calls for further investigations for up- and downstream regulation. Other genes including <italic>PTP4A2</italic>, <italic>RABAC1</italic>, and <italic>MAGOH</italic> have similar evidence of subpopulation structure and warrant further investigation.</p></sec><sec id="s2-7"><title>The SpliZ discovers conserved alternative splicing in mammalian spermatogenesis</title><p>The SpliZ provides a systematic framework to discover how splicing is regulated at a single-cell level along developmental trajectories (i.e., pseudotime). Previous studies have shown that testis is among the tissues with the highest isoform complexity and that even the isoform diversity in spermatogenic cells (spermatogonia, spermatocytes, round spermatids, and spermatozoa) is higher than that of many tissues (<xref ref-type="bibr" rid="bib46">Soumillon et al., 2013</xref>). Also, RNA processing has been known to be important in spermatogenesis (<xref ref-type="bibr" rid="bib20">Green et al., 2018</xref>). However, alternative splicing in single cells during spermatogenesis at the resolution of developmental time enabled by single-cell trajectory inference has not been studied. To systematically identify cells whose splicing is regulated during spermatogenesis, we applied the SpliZ to 4490 human sperm cells (<xref ref-type="bibr" rid="bib23">Hermann et al., 2018</xref>) and compared findings to mouse (<xref ref-type="bibr" rid="bib23">Hermann et al., 2018</xref>) and mouse lemur (<xref ref-type="bibr" rid="bib48">Tabula Microcebus Consortium, 2021</xref>) sperm cells to test for conservation of regulated splicing changes.</p><p>170 genes out of 1757 genes with computable SpliZ in &gt;100 human cells have splicing patterns that are significantly correlated with the pseudotime (|Spearman’s correlation| &gt; 0.1, Bonferroni-corrected p-value &lt; 0.05, <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). The highest correlated genes included <italic>TPPP2</italic>, a gene regulating tubulin polymerization implicated in male infertility (<xref ref-type="bibr" rid="bib58">Zhu et al., 2019</xref>), <italic>FAM71E1</italic>, being predominantly expressed in testes (<xref ref-type="bibr" rid="bib30">Kwon et al., 2017</xref>), <italic>SPATA42</italic>, a long non-coding RNA implicated in azoospermia (<xref ref-type="bibr" rid="bib5">Bo et al., 2020</xref>), <italic>MTFR1</italic>, a gene regulating mitochondrial fission, and <italic>MLF1</italic>, an oncogene regulated in <italic>Drosophila</italic> testes (<xref ref-type="bibr" rid="bib44">Singh et al., 2016</xref>; <xref ref-type="fig" rid="fig8">Figure 8A</xref>, <xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>). In <italic>MTFR1,</italic> SpliZsites identify an unannotated 3' splice site in immature sperm showing evidence of novel transcripts (<xref ref-type="fig" rid="fig8">Figure 8A</xref>).</p><fig-group><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Developmentally regulated alternative splicing in mammalian spermatogenesis.</title><p>(<bold>A</bold>) Regulated alternative splicing of <italic>MTFR1</italic> during sperm development. Significant negative correlation (Spearman’s correlation = –0.27, p-value = 1.23e-7) between the SpliZ score for gene <italic>MTFR1</italic> and pseudotime in human sperm cells (top left). Dot plot and box plot show increasing use of a downstream 3′ splice site driving the <italic>MTFR1</italic> SpliZ in equal pseudotime quantiles (top right) with the same trend in immature (spermatocyte) and mature (spermatid) cells (bottom left). The gene structure for <italic>MTFR1</italic> according to the human RefSeq annotation database is shown in the bottom-right panel. The orange box on the gene structure represents the PDEase_I domain and how it is affected by the alternative splicing. (<bold>B</bold>) Dot plots showing the developmentally regulated alternative splicing of gene <italic>CEP112</italic> in testis cells from human, mouse, and mouse lemur. Cells are grouped according to pseudotime quantiles. The alternative splicing is conserved (i.e., involves the same set of 5′ and 3′ splice sites in human, mouse, and mouse lemur data as shown by the gray arrows on the gene structures) and involves 5′ splice sites 65,826,138 and 65,851,804 in human, 5′ splice sites 108,664,726 and 108,682,875 in mouse, and 5′ splice sites 38,798,359 and 38,809,336 in mouse lemur. The 3′ splice site and the two 5′ splice sites involved in alternative splicing are shown by black and red vertical lines, respectively, on the gene structures. <italic>CEP112</italic> is on the minus strand in the human genome but is on the plus strand in mouse and mouse lemur genomes, leading to a negative correlation in splice site usage. Gray arrows show the LiftOver mapping between the 3′ splice site and two 5′ splice sites of the exon skipping event (indicating that the alternative splicing is conserved) and gray dashed lines for the human plot show the location of the 5′ splice sites and how splicing changes the apolipoprotein protein domain.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig8-v2.tif"/></fig><fig id="fig8s1" position="float" specific-use="child-fig"><label>Figure 8—figure supplement 1.</label><caption><title>Regulated alternative splicing of <italic>MLF1</italic> during sperm development in human, mouse, and mouse lemur.</title><p>The same 5′ splice site drives the alternative splicing in human and mouse. The gene structures for <italic>MLF1</italic> according to the RefSeq database for human, mouse, and mouse lemur are shown.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig8-figsupp1-v2.tif"/></fig><fig id="fig8s2" position="float" specific-use="child-fig"><label>Figure 8—figure supplement 2.</label><caption><title>Regulated alternative splicing of <italic>SPTY2D1OS</italic> during sperm development in human and mouse lemur.</title><p>Regulated splicing in both human and mouse lemur involves one unannotated 5′ splice site. The gene structures according to the RefSeq database for human and mouse lemur are shown.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-70692-fig8-figsupp2-v2.tif"/></fig></fig-group><p>Among significantly correlated genes in human cells, splicing in 10 of these genes is also developmentally regulated in mouse and mouse lemur (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). Centrosomal protein 112 (<italic>CEP112</italic>), a coiled-coil domain containing centrosomal protein and member of the cell division control protein, had the highest SpliZ-pseudotime correlation. It is highly expressed in testes and is essential for maintaining sperm function: loss-of-function mutations in <italic>CEP112</italic> have recently been associated with male infertility (<xref ref-type="bibr" rid="bib43">Sha et al., 2020</xref>). Strikingly, the same 3' splice site and 5' splice sites identified by SpliZsites are orthologous and affect the apolipoprotein domain, a protein involved in the delivery of lipid between cell membranes and which is critical for the sperm development and fertility (<xref ref-type="bibr" rid="bib42">Setarehbadi et al., 2012</xref>; <xref ref-type="fig" rid="fig8">Figure 8B</xref>).</p><p><italic>SPTY2D1OS</italic> is another gene with conserved regulated splicing in spermatogenesis (<xref ref-type="fig" rid="fig8s2">Figure 8—figure supplement 2</xref>). Though highly expressed in human testes, it has unknown function in sperm development. <italic>SPTY2D1OS</italic> is located between <italic>Uveld</italic> and <italic>SPTY2D</italic> in the human genome; in mouse, <italic>SPTY2D1OS</italic> corresponds to a lncRNA named <italic>Sirena1</italic>, which has been recently shown to have function in mouse oocyte development (<xref ref-type="bibr" rid="bib14">Ganesh et al., 2020</xref>) but has not previously been implicated in spermatogenesis. Together, our results suggest transcriptome-wide regulation of splicing in spermatogenic cells and call for more investigation into the function of splicing regulation not only in sperm development but also in other developmental trajectories.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Cell-type-specific splicing has been known to have functional effects in some cell types and in some genes for decades. However, technological limitations in measurement technology and methods to analyze resulting data have prevented high-throughput studies that profile the extent to which cell type can be predicted from splicing information alone. Full-coverage technologies such as SS2 have been the primary technologies for analyzing splicing in single cells thus far. However, SS2 is very difficult to scale: sequencing of 5000 cells that would take 2–3 days using 10X is estimated to take ~26 weeks using SS2 (<xref ref-type="bibr" rid="bib41">See et al., 2019</xref>; <xref ref-type="bibr" rid="bib47">Svensson et al., 2020</xref>). The lack of analysis of splicing in droplet data prevents discovery of regulated splicing in cell types that cannot be adequately profiled by plate-based approaches, and therefore causes biologically regulated splicing in these cell types to be missed (<xref ref-type="bibr" rid="bib51">Travaglini et al., 2020</xref>). Here, we apply new analytic methodologies to find highly regulated splicing patterns from ubiquitous droplet-based sequencing platforms. These results reveal deeply conserved splicing programs that define tissue compartment and cell type in vivo.</p><p>Although the SpliZ method enables biological discovery of splicing differences based on droplet-based sequencing data, droplet-based data still presents major challenges for splicing analysis compared to full-length data. In this study, droplet-based sequencing has much lower sequencing coverage than full-length data, resulting in only 1416 genes with computable SpliZ values in the first human individual based on 10X data compared to 9802 genes with computable SpliZ values in SS2 data. Additionally, current droplet-based data is 3′-biased, meaning that some splicing events will never be sequenced by the technology and therefore cannot be analyzed. Despite these challenges, the ubiquity of droplet-based data, its utility for profiling rare cell types, and its unprecedented scale make it a powerful approach to discover regulated splicing.</p><p>The reproducibility of models in independently generated datasets suggests that the SpliZ can be applied globally to larger numbers of cell types to further identify splicing regulation at a single-cell level. We predict that as the number of cells profiled and the cell types grow, and global analyses are performed on data that is not 3′ biased, the fraction of genes with evidence of cell-type-specific splicing will increase substantially beyond our current estimate of around 10% (<xref ref-type="fig" rid="fig1">Figure 1F</xref>). The results presented here lay the foundation for comprehensive splicing analysis in any scRNA-seq dataset and a reference to which future studies can be compared. Together, this work provides strong evidence for the hypothesis that alternative splicing in a large fraction of human genes is cell-type-specifically regulated and supports the idea that splicing is central to functional specialization of cell types.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Reagent type (species) or resource</th><th align="left" valign="bottom">Designation</th><th align="left" valign="bottom">Source or reference</th><th align="left" valign="bottom">Identifiers</th><th align="left" valign="bottom">Additional information</th></tr></thead><tbody><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">SICILIAN</td><td align="left" valign="bottom"><xref ref-type="bibr" rid="bib9">Dehghannasiri et al., 2021</xref></td><td align="left" valign="bottom"/><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/salzmanlab/SICILIAN">https://github.com/salzmanlab/SICILIAN</ext-link>, <xref ref-type="bibr" rid="bib40">Roozbeh, 2021</xref></td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">SpliZ Pipeline</td><td align="left" valign="bottom"><xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref></td><td align="left" valign="bottom"/><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/juliaolivieri/SpliZ_pipeline">https://github.com/juliaolivieri/SpliZ_pipeline</ext-link>, <xref ref-type="bibr" rid="bib27">Julia, 2021b</xref></td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">STAR</td><td align="left" valign="bottom"><xref ref-type="bibr" rid="bib10">Dobin et al., 2013</xref></td><td align="left" valign="bottom"/><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://github.com/alexdobin/STAR">https://github.com/alexdobin/STAR</ext-link>, <xref ref-type="bibr" rid="bib1">Alexander, 2021</xref></td></tr><tr><td align="left" valign="bottom">Commercial assay or kit</td><td align="left" valign="bottom">BaseScope Duplex Reagent Kit - Hs</td><td align="left" valign="bottom">ACD (Bio-Techne)</td><td align="left" valign="bottom">cat. no 323,870</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Commercial assay or kit</td><td align="left" valign="bottom">BaseScope Probes for <italic>MYL6</italic></td><td align="left" valign="bottom">ACD (Bio-Techne)</td><td align="left" valign="bottom">BA-Hs-MYL6-tv1-1zz-st-C2 and BA-Hs-MYL6-tv2-1zz-st</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Commercial assay or kit</td><td align="left" valign="bottom">BaseScope Probes for <italic>RPS24</italic></td><td align="left" valign="bottom">ACD (Bio-Techne)</td><td align="left" valign="bottom">BA-Hs-RPS24-tva-1zz-st-C2 and BA-Hs-RPS24-tvc-1zz-st</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Sequence-based reagent</td><td align="left" valign="bottom">FL-RPS24ex4F1</td><td align="left" valign="bottom">This paper</td><td align="left" valign="bottom">PCR primer</td><td align="left" valign="bottom"><named-content content-type="sequence">/6FAM/CAATGTTGGTGCTGGCAAAA</named-content></td></tr><tr><td align="left" valign="bottom">Sequence-based reagent</td><td align="left" valign="bottom">RPS24ex6R2</td><td align="left" valign="bottom">This paper</td><td align="left" valign="bottom">PCR primer</td><td align="left" valign="bottom"><named-content content-type="sequence">GCAGCACCTTTACTCCTTCGG</named-content></td></tr></tbody></table></table-wrap><sec id="s4-1"><title>File downloads</title><list list-type="bullet"><list-item><p>Human RefSeq hg38 annotation file was downloaded from <ext-link ext-link-type="uri" xlink:href="ftp://ftp.ncbi.nlm.nih.gov/refseq/H_sapiens/annotation/GRCh38_latest/refseq_identifiers/GRCh38_latest_genomic.gff.gz">ftp://ftp.ncbi.nlm.nih.gov/refseq/H_sapiens/annotation/GRCh38_latest/refseq_identifiers/GRCh38_latest_genomic.gff.gz</ext-link></p></list-item><list-item><p>Mouse lemur RefSeq Micmur3 annotation file was downloaded from <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/assembly/GCF_000165445.2/">https://www.ncbi.nlm.nih.gov/assembly/GCF_000165445.2/</ext-link></p></list-item><list-item><p>Mouse RefSeq GRCm38.p6 annotation file was downloaded from <ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/genomes/all/GCF/000/001/635/GCF_000001635.26_GRCm38.p6/GCF_000001635.26_GRCm38.p6_genomic.gtf.gz">https://ftp.ncbi.nlm.nih.gov/genomes/all/GCF/000/001/635/GCF_000001635.26_GRCm38.p6/GCF_000001635.26_GRCm38.p6_genomic.gtf.gz</ext-link></p></list-item><list-item><p>The UCSC Pfam database for the human hg38 genome assembly was downloaded from <ext-link ext-link-type="uri" xlink:href="http://hgdownload.soe.ucsc.edu/goldenPath/hg38/database/ucscGenePfam.txt.gz">http://hgdownload.soe.ucsc.edu/goldenPath/hg38/database/ucscGenePfam.txt.gz</ext-link></p></list-item><list-item><p>The Gene name mapping file for orthologous genes between human, mouse, and mouse lemur was downloaded from the Ensembl BioMart search tool (<ext-link ext-link-type="uri" xlink:href="http://www.ensembl.org/biomart/martview/">http://www.ensembl.org/biomart/martview/</ext-link>) on 12/11/2020.</p></list-item></list></sec><sec id="s4-2"><title>Code availability</title><p>Code to reproduce analysis and create figures is available through this GitHub repository: <ext-link ext-link-type="uri" xlink:href="https://github.com/juliaolivieri/DiffSplice">https://github.com/juliaolivieri/DiffSplice</ext-link> (copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8ce4ed66a931f705fc924095639b3c02efe9a944;origin=https://github.com/juliaolivieri/DiffSplice;visit=swh:1:snp:3c0aaffa3eedcba8221494094dc49ecdbf13cb7a;anchor=swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830">swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830</ext-link>, <xref ref-type="bibr" rid="bib26">Julia, 2021a</xref>).</p></sec><sec id="s4-3"><title>Explanation of the SpliZ method</title><p>The SpliZ is a scalar score assigned to each cell-gene pair in a single-cell dataset. It is calculated using a three-step procedure (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>). First, for every splice site with multiple partners in a dataset, those partners are assigned ranks according to their distance from the splice site. Next, each of these ranks is converted to a mean-zero, variance-one residual that quantifies the statistical deviation of that rank compared to the overall population. Finally, for a given cell and gene, these residuals are summed for each spliced read mapping to the corresponding splice site, and then scaled. Intuitively, the SpliZ for a particular gene has a large negative value if the introns for the gene in a given cell are smaller than average, and has a large positive value if the introns for the gene in a given cell are larger than average.</p><p>The SpliZVD is a modification of the SpliZ, in which rather than simply summing the splicing residuals for a given cell the residuals are scaled based on the eigenvector loadings of the first eigenvector of the residual matrix. The SVD decomposition of the residual matrix is also used to determine the SpliZsites for a given gene, which are defined as the three largest-magnitude components of the first eigenvector.</p><p>After SpliZ values are computed, if annotations are provided the SpliZ pipeline calculates which genes are differentially spliced between groups in the annotation. To calculate a p-value for whether the median SpliZ values by annotation are different for a given gene, the distribution of medians is first referred to as a null distribution. For p-values passing a nominal 0.05 level, permutations are performed to estimate the p-value with higher precision, and then adjusted using the Benjamini–Hochberg correction (<xref ref-type="bibr" rid="bib24">Hochberg and Benjamini, 1990</xref>; <xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>).</p></sec><sec id="s4-4"><title>SpliZ pipeline</title><p>Data from each individual was preprocessed from fastqs using SICILIAN with default parameters (<xref ref-type="bibr" rid="bib9">Dehghannasiri et al., 2021</xref>). SICILIAN is a statistical method that can be applied to the BAM files by spliced aligners such as STAR (<xref ref-type="bibr" rid="bib10">Dobin et al., 2013</xref>) to remove false-positive junction calls, enabling unbiased discovery of unannotated junctions that can contribute to alternative splicing. The scRNA-seq datasets were mapped using STAR version 2.7.5a in two-pass mode with default parameters. Also, SICILIAN performs UMI deduplication to remove PCR duplicates. SpliZ scores were calculated using the SpliZ pipeline with default parameters (<xref ref-type="bibr" rid="bib38">Olivieri et al., 2021</xref>). A SpliZ score was assigned to a gene-cell pair if there were at least five spliced reads from that gene aligned in that cell. Differential analysis was performed both based on tissue compartment (endothelial, epithelial, immune, and stromal) and independently based on cell type (defined by the tissue, compartment, and individual cell type, e.g., ‘lung immune macrophage’). For a given gene, only cell types with at least 10 cells with computable SpliZ values for that gene were used. The SpliZ was used to call genes as significant for all datasets except the full SS2 datasets, for which the SpliZVD was used because of the increased complexity of full-length transcript data. We used a p-value cutoff of 0.05 after Benjamini–Hochberg correction. We define ‘effect size’ for a gene to be the largest magnitude median SpliZ (or SpliZVD) value out of all cell types with computable SpliZ for the gene. For between-cell-type-analysis, we use an effect size threshold of 0.5 (3.5 for SpliZVD) (Supplement) and require a difference of at least 0.5 within a single tissue and compartment for the gene to be called.</p></sec><sec id="s4-5"><title>FISH methods</title><p>Human rectus abdominis muscle biopsies from two donors were processed to single-cell suspensions by a combination of manual and enzymatical dissociation (<xref ref-type="bibr" rid="bib50">Tabula Sapiens Consortium, 2021</xref>). Single-cell suspensions were stained with a combination of antibodies against CD45, CD31, THY1, and CD82, allowing for the isolation of immune cells (CD45+), endothelial cells (CD31+), mesenchymal cells (THY1+), and skeletal muscle satellite stem cells (CD82+). Due to the low number of immune cells present in the tissues, only the latter three cell types were stained. Cells were cytospun onto ECM-coated 8-well chamber slides and fixed in 4% PFA. Cells were washed in PBS and prepared for RNA FISH by replacing the PBS to 100% ethanol. Cells were stained with custom probes according to the manufacturer’s protocol (BaseScope Duplex Detection Reagent Kit [Advanced Cell Diagnostics, ACD]). Briefly, cells were rehydrated and treated with Protease IV solution (1;15 dilution) and were subsequently stained with indicated BaseScope probes for 2 hr in a hybridization oven set to 40°C. Cells were then treated with amplification steps and imaged immediately after completion of the staining. As a control, human primary myoblasts were stained with the BaseScope probes and a negative control probe. Images were captured with a Zeiss Axiofluor microscope with collected CCD camera and a 40× objective lens. The red dye fluoresces in the 555 channel, whereas the green dye shows as gray in the DIC channel. Images were quantified with Volocity software. One muscle sample was independently fixed in 10% neutral buffered formalin for 24 hr in preparation for BaseScope staining in cryosections. Tissue was dehydrated in 20% sucrose for 24 hr, washed in PBS, dried, embedded in OCT, and frozen in cooled isopentane. Sections of 10 µm were cut and dried in –20°C for 1 hr and stored –80°C until use. Tissue slides were removed from –80°C and immediately washed with PBS to remove OCT, dried, and baked in 60°C for 30 min. Tissue slides were post-fixed in 4% PFA for 15 min and dehydrated by immersing slides in 50, 70, and 100% ethanol for 5 min each. Tissue slides were then treated with hydrogen peroxide for 10 min and washed briefly with distilled water and subjected to target retrieval for 5 min, washed briefly with distilled water and in 100% ethanol. Tissue slides were treated with Protease IV solution in 40°C for 10 min and washed twice with distilled water and hybridized with indicated BaseScope probes for 2 hr in 40°C. Tissue slides were then treated with amplification steps. For dual FISH and IHC staining, tissue slides were immediately blocked in blocking buffer for 30 min (5% FBS, 1% BSA, 0.1% Triton-X100, 0.01% sodium azide in PBS) and stained with Pax7 antibody (1:100) in blocking buffer overnight in 4°C. Tissue slides were washed with 0.1% Tween-20 in PBS three times and then fluorescently conjugated secondary antibodies were added for an hour in room temperature. After three additional washes, tissue slides were dried, mounted, and imaged immediately.</p><p>Deidentified human adult lung tissue was obtained from the Stanford Tissue Bank. The tissue was fixed in 10% neutral buffered formalin and embedded in paraffin. For the single-molecule in situ hybridization, 6-µm-thick paraffin sections were prepared and processed following the BaseScope Duplex Detection Reagent Kit (ACD) protocol, modified to use brown DAB chromogen in place of the usual green chromogen as the second color (custom protocol from ACD). Stained slides were visualized using an Olympus upright bright field microscope at 20× and 40× magnification. Cell types were identified by a pathologist based on cell morphology highlighted by the hematoxylin counterstain. Representative images of each cell type of interest were captured using an Olympus digital microscope color camera. Quantification was done by demarcating a polygonal image region containing multiple cells of homogeneous type and manually counting all the dots of each color within the region.</p><p>Proprietary probes (ACD) used for both human lung and muscle: BA-Hs-RPS24-tvc-1zz-st (targets 400–437 of NM_001026.5), BA-Hs-RPS24-tva-1zz-st (targets 399–437 of NM_033022.4); BA-Hs-MYL6-tv1-1zz-st (targets 469–505 of NM_021019.5), BA-Hs-MYL6-tv2-1zz-st (targets 436–480 of NM_079423.4).</p></sec><sec id="s4-6"><title>Single-cell RT-PCR</title><p>SS2 preamplified cDNA of single cells from the Human Lung Cell Atlas project (<xref ref-type="bibr" rid="bib51">Travaglini et al., 2020</xref>) was used as starting templates. The cells correspond to wells N14, A16, H14, B6, A3, A7, A13, A11, A8, D1, A12, A17, B12, J16, A21, P22, D23, A22, B22 of plate B002014; cell type metadata was taken from <ext-link ext-link-type="uri" xlink:href="https://www.synapse.org/#!Synapse:syn21041850/wiki/60086">https://www.synapse.org/#!Synapse:syn21041850/wiki/60086</ext-link>. 1 µl of primary preamp was further preamplified in a 20 µl reaction (100 nM ISPCR primer = <named-content content-type="sequence">AAGCAGTGGTATCAACGCAGAGT</named-content>, KAPA HiFi Fidelity mix; program: 95° 3'; 9 × [98° 20&quot;; 67° 15&quot;; 72° 4']; 72° 5'), then diluted eightfold with water. 2 µl of this secondary preamp was used as template in a 40 µl reaction (500 nM each of primers FL-RPS24ex4F1 = /6FAM/<named-content content-type="sequence">CAATGTTGGTGCTGGCAAAA</named-content> and RPS24ex6R2 = <named-content content-type="sequence">GCAGCACCTTTACTCCTTCGG</named-content>, New England Biolabs Phusion HF buffer, 200 nM dNTPs, 0.4 units Phusion DNA Polymerase; program: 98° 30&quot;; 24 × [98° 10&quot;; 60° 15&quot;; 72° 20&quot;]; 72° 5'). PCRs were diluted 1:100 and run on an ABI 3130xl Genetic Analyzer with GS500ROX standard; peaks were called by the Thermo Fisher Cloud Peak Scanner app, and presented as a pseudo-gel image using a custom Python script. For Sanger sequencing, secondary preamps were used in a similar PCR but with primers RPS24ex4F4 = <named-content content-type="sequence">AAGCAACGAAAGGAACGCAA</named-content> and RPS24ex6R4 = <named-content content-type="sequence">CCACAGCTAACATCATTGCAG</named-content>; the cleaned-up products were sequenced with the same primers. Oligonucleotide synthesis and capillary electrophoresis were done by Stanford PAN (Protein and Nucleic Acid Facility).</p></sec><sec id="s4-7"><title>Concordance analysis between technologies and donors</title><p>Concordance with SS2 was used as an extra test of the reproducibility of the method. SS2 and 10X datasets were subset to include only junctions and cell types shared in both to make the datasets as comparable as possible, and remove RNA measurements that could only be detected by SS2. Next, the SpliZ was calculated independently for both datasets as described for 10X. We then correlated the median SpliZ scores for matched genes and ontologies for genes called as significant by both technologies in the same individual. This resulted in a Pearson correlation of 0.439 between the two technologies for individual 1 and 0.769 for individual 2. We similarly subsetted both 10X datasets so that they each only included shared cell types and junctions, and then ran the SpliZ pipeline separately on each dataset, resulting in a Pearson correlation of 0.776 between the two datasets.</p></sec><sec id="s4-8"><title>K-means clustering of <italic>RPS24</italic> and <italic>ATP5F1C</italic></title><p>We first subsetted to only cells in the immune, epithelial, and stromal compartments with computable SpliZ values for both <italic>RPS24</italic> and <italic>ATP5F1C</italic> in the 10X data (9712 cells in individual 1, 2370 cells in individual 2). K-means clustering was performed with the “sklearn” package in Python with k = 3 to separate all cells into three clusters based on their <italic>RPS24</italic> and <italic>ATP5F1C</italic> SpliZ values. Each resulting cluster was assigned to one compartment such as to minimize classification error, and accuracy was calculated for each compartment based on these cluster assignments.</p></sec><sec id="s4-9"><title>LiftOver shared sites</title><p>We used the UCSC LiftOver tool (<ext-link ext-link-type="uri" xlink:href="https://genome.ucsc.edu/cgi-bin/hgLiftOver">https://genome.ucsc.edu/cgi-bin/hgLiftOver</ext-link>) with the recommended settings to convert the coordinates between human (hg38), mouse (mm10), and mouse lemur (Mmur3). To find shared SpliZsites between human, mouse, and mouse lemur, we subset to only those junctions that had been successfully and uniquely converted by the LiftOver tool.</p></sec><sec id="s4-10"><title>Spermatogenesis analysis</title><p>To find genes with regulated splicing during sperm development, for each gene, Spearman’s correlation was computed between the SpliZ and pseudotime values across the cells with computable SpliZ scores for that gene. We considered only genes with computable SpliZ in at least 100 cells, and for each organism (human, mouse, mouse lemur), the genes with |Spearman’s coefficient| &gt;0.1 and Bonferroni-corrected p-value &lt; 0.05 were selected as significantly splicing regulated genes. Only those genes that have names in all three organisms were considered for the conservation analysis.</p></sec><sec id="s4-11"><title>Subpopulation analysis</title><p>To find subcluster of cells within cell types that can be distinguished based on the splice profile as quantified by the SpliZ score, we take advantage of the fact that under the null hypothesis all cells within a cell type should follow a univariate normal distribution for the SpliZ score of each gene; however, if there are subcluster of cells with distinct splice profiles, the distribution should be better modeled via a GMM. To find the optimal number of components for the distribution of the SpliZ score for each gene within a cell type, we used the ICL, which is a model selection criterion, and selected the optimal number of component as the number that attains the knee point in the ICL curve for different component numbers. If ICL selects at least two subclusters within a cell type, we assigned cells to one of the clusters based on the fitted GMM with the optimal number of components. After clustering cells, we checked to see if subclusters are disjoint enough by computing the Bhattacharyya distance between the subclusters. The subclusters for a pair of gene and cell type are called, if the distance between subclusters is &gt;0.5. Otherwise, we reduced the optimal number of components by 1 and then again ran the GMM clustering to see if the new subclusters can satisfy the Bhattacharyya criterion. We keep doing this until either we end up with only one cluster (which means no subcluster is found) or the resulting subclusters have enough distance.</p><p>To test whether the subpopulations of <italic>SAT1</italic> were the result of a ‘bimodal splicing’ artifact as reported in <xref ref-type="bibr" rid="bib7">Buen Abad Najar et al., 2020</xref>, we performed the following analysis. Considering the blood classical monocytes in human individual 1 together, we calculated the fraction <italic>p</italic> of junctional reads aligning to the 5′ splice site 23,785,328 in <italic>SAT1</italic> that partner with the 3′ splice site 23,783,883 rather than 23,784,403. We found p = 83/102 = 0.814 (the probability was 0.893 in individual 2). We then subset to only cells with exactly two reads mapping to 5′ splice site 23,785,328, resulting in 16 cells. Out of these 16 cells, all had either both reads mapping to 3′ splice site 23,783,883 or both reads mapping to 3′ splice site 23,784,403. We calculated the probability of zero cells having one read mapping to each splice site under the null hypothesis as follows: (1 - binom.pmf(1,2,0.814))<sup>16</sup> = 0.00312 (exact binomial test). Because processing of 10X data includes a UMI deduplication step through SICILIAN (<xref ref-type="bibr" rid="bib9">Dehghannasiri et al., 2021</xref>), these duplicates are not PCR duplicates.</p></sec><sec id="s4-12"><title>SpliZsite analysis</title><p>23 out of 148 (fraction = 0.1554) and 11 out of 138 (fraction = 0.0797) SpliZsites found in human individual 1 were also found to be SpliZsites in mouse lemur and mouse, respectively. We limited the comparison with mouse (resp. mouse lemur) SpliZsites to only those SpliZsites whose corresponding genes had computable SpliZ in mouse (resp. mouse lemur). To compute the expected fraction of shared SpliZsites between human and one of the other organisms under the null, we first need the null probability of each SpliZsite being shared between human and the other organism by considering the number of splice sites for the gene. This probability is <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <inline-formula><mml:math id="inf3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the number of distinct splice sites (we considered only 5′ splice sites) with junctional reads according to SICILIAN. If there are <inline-formula><mml:math id="inf4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> SpliZsites in human whose genes have also computable SpliZ scores in the other organism, the expected fraction of shared SpliZsites between human and that organism is <inline-formula><mml:math id="inf5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>I</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mfrac></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, resulting in the expected fractions 0.071 and 0.088 of human SpliZsites shared with mouse lemur and mouse, respectively. The p-value for the observed fraction 0.1554 for mouse lemur can be approximated using a binomial test (the binomial test is an approximation as the success probability for each SpliZsite changes according to <inline-formula><mml:math id="inf6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula>), which results in a p-value of 0.0003.</p></sec><sec id="s4-13"><title id="E1">Choosing effect size filters</title><p>To choose the filters for differential analysis in 10X data, we subsetted the TSP1 and TSP2 10X data to only junctions shared in both individuals and cell types shared in both individuals. We then ran the SpliZ pipeline. Using a p-value threshold of 0.05 for the SpliZ, we tested the correlation between median SpliZ values matching on cell type for genes called as significant in both datasets based on that effect size cutoff. Without an effect size cutoff, there was already a correlation of 0.2. We chose the effect size threshold 0.5 because it yielded a correlation of 0.6 between the datasets (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>). To decide the SpliZVD cutoff for SS2 data, we perform the same procedure, except this time use the SS2 data from TSP1 and TSP2, again restricted to only shared junctions and shared cell types. Because the correlation never reaches 0.6 for this data, we choose a cutoff of 3.5 because it maximizes correlation (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>).</p></sec><sec id="s4-14"><title>Sanger sequencing confirms low levels of the +<italic>a</italic>-<italic>b+c RPS24</italic> isoform in mouse kidney</title><p>Whole-tissue RNA was amplified from mouse adult kidney and human fetal kidney, assuming that a large fraction of the RNA would be coming from epithelial cells. The PCR products were cloned and then ligated so that multiple could be read out per each Sanger read. However due to the repetitive nature of the inserts, the read quality was poor and only the first couple of inserts could be interpreted. Generally, our results confirm that +a-b+c is not very abundant in mouse: in human kidney, it is 45% of total, while in mouse kidney it is only 14% (data not shown).</p></sec><sec id="s4-15"><title>Bowtie2 alignment of <italic>RPS24</italic> reads</title><p>Custom fasta files were created for human, mouse, and mouse lemur. Each includes eight transcripts corresponding to all combinations of inclusion of the a, b, and c exons in <italic>RPS24</italic>. Each sequence is centered on whichever of these exons are included, and padded on either side with sequence from exon 4 and exon 6 such that each sequence is 150 base pairs long. A Bowtie2 index was created based on these transcriptomes for each species, and all fasta files for individual 1 from human, mouse, and mouse lemur were aligned to the respective index using Bowtie2 with the command <monospace>bowtie2 --no-unal index/{params.species}_RPS24_bwt -U {input} -S {output}</monospace>, where <monospace>{params.species}</monospace> is the name of the species’ index, <monospace>{input}</monospace> is the input fasta and <monospace>{output}</monospace> is the output file name. Because exons a, b, and c are 3, 18, and 22 base pairs long, respectively, and 10X reads are around 90 base pairs long, each read aligning to one of these transcripts uniquely identifies the isoform.</p></sec><sec id="s4-16"><title>Enrichment of genes significant in all three species analysis</title><p>We test whether the number of genes significant in all three species is more than expected under the null hypothesis, which is that there is no evolutionary conservation between the species. Let <italic>s</italic> be the number of genes that are shared between human, lemur, and mouse that are significant at least once in all three species. Let <italic>n</italic> be the total number of genes present in at least 20 cells in a cell type in all three species (note: mapping between species is not perfect, so some genes present in all are probably missing). The probability that a given gene is significant in all three species is<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mn>3</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mtext> </mml:mtext><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mtext> </mml:mtext><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mspace linebreak="newline"/><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mtext> </mml:mtext><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mtext> </mml:mtext><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Under the null hypothesis, assume that a gene being significant in one species is independent from it being significant in either other species. Therefore, under the null hypothesis, <inline-formula><mml:math id="inf7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mn>3</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p><p>We can estimate the quantities on the right-hand side of the question for each species. For every species, calculate <italic>p<sub>species</sub></italic>, where<disp-formula id="equ2"><mml:math id="m2"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="normal">#</mml:mi><mml:mtext> </mml:mtext><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mtext> </mml:mtext><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>e</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mtext> </mml:mtext><mml:mi>h</mml:mi><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>m</mml:mi><mml:mi>z</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>s</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mtext> </mml:mtext><mml:mspace linebreak="newline"/><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">#</mml:mi><mml:mtext> </mml:mtext><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mtext> </mml:mtext><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mtext> </mml:mtext><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mtext> </mml:mtext><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>Z</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mtext> </mml:mtext><mml:mn>20</mml:mn><mml:mtext> </mml:mtext><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext> </mml:mtext><mml:mi>a</mml:mi><mml:mtext> </mml:mtext><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mtext> </mml:mtext><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Therefore, under the null hypothesis, the probability that at least <italic>x</italic> genes are significant in all three species out of <italic>n</italic> genes is given by <inline-formula><mml:math id="inf8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>b</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi mathvariant="normal">_</mml:mi><mml:mi>c</mml:mi><mml:mi>d</mml:mi><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>∗</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>∗</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>. The estimates of the degree of regulated splicing are a lower bound as they are based on (a) sampling only a subset of organs and (b) based on studying only a subset of genes that are sampled sufficiently with current sequencing depth conventions. Incomplete gene naming conventions, especially in the mouse and lemur, may restrict the power of this analysis.</p></sec></sec></body><back><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Resources, Software, Validation, Visualization, Writing – original draft</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Resources, Software, Validation, Visualization, Writing – original draft</p></fn><fn fn-type="con" id="con3"><p>Investigation, Methodology, Validation, Visualization</p></fn><fn fn-type="con" id="con4"><p>Formal analysis, Investigation, Validation</p></fn><fn fn-type="con" id="con5"><p>Formal analysis, Investigation, Validation</p></fn><fn fn-type="con" id="con6"><p>Formal analysis, Investigation, Validation</p></fn><fn fn-type="con" id="con7"><p>Data curation, Resources</p></fn><fn fn-type="con" id="con8"><p>Data curation, Resources</p></fn><fn fn-type="con" id="con9"><p>Project administration, Resources, Supervision</p></fn><fn fn-type="con" id="con10"><p>Project administration, Resources, Supervision</p></fn><fn fn-type="con" id="con11"><p>Conceptualization, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Writing – original draft, Writing – review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Dataset summary.</title><p>Brief overview of the datasets used in this paper, including tissues analyzed, number of cells, median number of spliced reads per cell, sex, and age.</p></caption><media mime-subtype="tab-separated-values" mimetype="text" xlink:href="elife-70692-supp1-v2.tsv"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Differential alternative splicing per compartment.</title><p>Separate table with the p-value based on the SpliZ and SpliZVD for each gene in each dataset, testing differences between compartments. The gene name, SpliZ p-value, SpliZVD p-value, and largest magnitude median for all compartments for SpliZ and SpliZVD for each gene are given by the <monospace>geneR1A_uniq, perm_pval_adj_scZ, perm_pval_adj_svd_z0, max_abs_median_scZ</monospace>, and <monospace>max_abs_median_svd_z0</monospace> columns, respectively. (<bold>A</bold>) Human individual 1 10X; (<bold>B</bold>) human individual 2 10X; (<bold>C</bold>) human individual 1 SS2; (<bold>D</bold>) human individual 2 SS2; (<bold>E</bold>) lemur individual 1 10X; (F) lemur individual 2 10X; (<bold>G</bold>) mouse individual 1 10X; (<bold>H</bold>) mouse individual 2 10X.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-70692-supp2-v2.xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Differential alternative splicing per cell type.</title><p>Separate table with the p-values based on the SpliZ and SpliZVD for each gene in each dataset, testing differences between cell types. The gene name, SpliZ p-value, SpliZVD p-value, and largest magnitude median for all compartments for SpliZ and SpliZVD for that gene are given by the geneR1A_uniq, perm_pval_adj_scZ, perm_pval_adj_svd_z0, max_abs_median_scZ, and max_abs_median_svd_z0 columns, respectively. (<bold>A</bold>) human individual 1 10X; (<bold>B</bold>) human individual 2 10X; (<bold>C</bold>) human individual 1 SS2; (<bold>D</bold>) human individual 2 SS2; (<bold>E</bold>) lemur individual 1 10X; (<bold>F</bold>) lemur individual 2 10X; (<bold>G</bold>) mouse individual 1 10X; (<bold>H</bold>) mouse individual 2 10X.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-70692-supp3-v2.xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Most variable splice sites (SpliZsites).</title><p>The most variable splice sites (SpliZsites) for genes with significant alternative splicing in human individual 1 10X data. Each line reports a SpliZsite and contains the coordinates, whether it is an annotated exon, whether it is an exon with known alternative splicing, whether the splice site is in 5′ or 3′ UTR of the gene, and whether the SpliZsite found to be a SpliZsite in mouse lemur and mouse datasets.</p></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-70692-supp4-v2.csv"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Regulated alternative splicing events in spermatogenesis.</title><p>The list of genes with significantly regulated alternative splicing during sperm development. Each line contains information about the number of cells, Spearman’s correlation, and its p-value for the human gene and also the same information (based on the mouse and mouse lemur sperm data) for its orthologous genes in mouse and mouse lemur genomes.</p></caption><media mime-subtype="plain" mimetype="text" xlink:href="elife-70692-supp5-v2.txt"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-70692-transrepform1-v2.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>The fastq files for the Tabula Sapiens data (both 10X Chromium and Smart-seq2) were downloaded from <ext-link ext-link-type="uri" xlink:href="https://tabula-sapiens-portal.ds.czbiohub.org/">https://tabula-sapiens-portal.ds.czbiohub.org/</ext-link>. The pilot 2 individual is referred to as individual 1, and the pilot 1 individual is referred to as individual 2 in this manuscript. Pancreas data was removed from individual 2. Cell type annotations were downloaded on March 19th, 2021, and the &quot;ground truth&quot; column was used as the within-tissue-compartment cell type. The Tabula Muris data was downloaded from a public AWS S3 bucket according to <ext-link ext-link-type="uri" xlink:href="https://registry.opendata.aws/tabula-muris-senis/">https://registry.opendata.aws/tabula-muris-senis/</ext-link>. The P1 (30-M-2) mouse is referred to as individual 1 and P2 (30-M-4) is referred to as individual 2 in this manuscript. Compartment annotations were assigned based on knowledge of cell type. The fastq files for the Tabula Microcebus mouse lemur data were downloaded from <ext-link ext-link-type="uri" xlink:href="https://tabula-microcebus.ds.czbiohub.org">https://tabula-microcebus.ds.czbiohub.org</ext-link>. Mouse lemurs 4 and 2 are referred to as individuals 1 and 2, respectively, in this manuscript. The propagated_cell_ontology_class column was used as the within-tissue-compartment cell type. Because tissue compartments in the mouse lemur were annotated more finely, we collapsed the lymphoid, myeloid, and megakaryocyte-erythroid compartments into the immune compartment. Human and mouse unselected spermatogenesis data was downloaded from the SRA databases with accession IDs SRR6459190 (AdultHuman_17-3), SRR6459191 (AdultHuman_17-4), and SRR6459192 (AdultHuman_17-5) for human, and accession IDs SRR6459155 (AdultMouse-Rep1), SRR6459156 (AdultMouse-Rep2), and SRR6459157 (AdultMouse-Rep3) for mouse. The files containing SpliZ values can be accessed at the following FigShare repository: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.14531721">https://doi.org/10.6084/m9.figshare.14531721</ext-link>.</p><p>The following dataset was generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Olivieri</surname><given-names>JO</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>RNA splicing programs define tissue compartments and cell types at single cell resolution</data-title><source>figshare</source><pub-id pub-id-type="doi">10.6084/m9.figshare.14531721</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation id="dataset2" publication-type="data" specific-use="references"><person-group person-group-type="author"><collab>Tabula Microcebus Consortium</collab></person-group><year iso-8601-date="2021">2021</year><data-title>Tabula Microcebus</data-title><source>Tabula Microcebus</source><pub-id pub-id-type="accession" xlink:href="https://tabula-microcebus.ds.czbiohub.org/about">czbiohub</pub-id></element-citation></p><p><element-citation id="dataset3" publication-type="data" specific-use="references"><person-group person-group-type="author"><collab>Tabula Muris Consortium </collab></person-group><year iso-8601-date="2018">2018</year><data-title>Tabula Muris</data-title><source>Tabula Muris</source><pub-id pub-id-type="accession" xlink:href="https://tabula-muris.ds.czbiohub.org/">ds.czbiohub</pub-id></element-citation></p><p><element-citation id="dataset4" publication-type="data" specific-use="references"><person-group person-group-type="author"><collab>Tabula Sapiens Consortium</collab></person-group><year iso-8601-date="2021">2021</year><data-title>Tabula Sapiens</data-title><source>Tabula Sapiens</source><pub-id pub-id-type="accession" xlink:href="https://tabula-sapiens-portal.ds.czbiohub.org/">portal.ds.czbiohub</pub-id></element-citation></p><p><element-citation id="dataset5" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultHuman_17-3</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459190">SRR6459190</pub-id></element-citation></p><p><element-citation id="dataset6" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultHuman_17-4</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459191">SRR6459191</pub-id></element-citation></p><p><element-citation id="dataset7" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultHuman_17-5</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459192">SRR6459192</pub-id></element-citation></p><p><element-citation id="dataset8" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultMouse-Rep1</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459155">SRR6459155</pub-id></element-citation></p><p><element-citation id="dataset9" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultMouse-Rep2</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459156">SRR6459156</pub-id></element-citation></p><p><element-citation id="dataset10" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>AdultMouse-Rep3</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=SRR6459157">SRR6459157</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Manny Ares, Douglas Black, Maria Barna, and members of the Salzman lab for insightful discussions. We thank Kyle Travaglini (Krasnow lab) for aliquots of single-cell RT-PCR preamplification from the Human Lung Cell Atlas samples. We thank Jessica Klein for creating part of <xref ref-type="fig" rid="fig1">Figure 1</xref>. JO is supported by the National Science Foundation Graduate Research Fellowship under Grant No. DGE-1656518 and a Stanford Graduate Fellowship. RD is supported by the Cancer Systems Biology Scholars Program Grant R25 CA180993 and the Clinical Data Science Fellowship Grant T15 LM7033-36. JS is supported by the National Institute of General Medical Sciences Grant R01 GM116847 and NSF Faculty Early Career Development Program Award MCB1552196. None of these funding sources were involved in study design, data collection and interpretation, or the decision to submit the work for publication.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Alexander</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>STAR 2.7.9A</data-title><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/alexdobin/STAR">https://github.com/alexdobin/STAR</ext-link></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Anczuków</surname><given-names>O</given-names></name><name><surname>Krainer</surname><given-names>AR</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Splicing-factor alterations in cancers</article-title><source>RNA</source><volume>22</volume><fpage>1285</fpage><lpage>1301</lpage><pub-id pub-id-type="doi">10.1261/rna.057919.116</pub-id><pub-id pub-id-type="pmid">27530828</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arzalluz-Luque</surname><given-names>Á</given-names></name><name><surname>Conesa</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Single-cell RNAseq for the study of isoforms—how is that possible?</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>110</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1496-z</pub-id><pub-id pub-id-type="pmid">30097058</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baralle</surname><given-names>FE</given-names></name><name><surname>Giudice</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Alternative splicing as a regulator of development and tissue identity</article-title><source>Nature Reviews Molecular Cell Biology</source><volume>18</volume><fpage>437</fpage><lpage>451</lpage><pub-id pub-id-type="doi">10.1038/nrm.2017.27</pub-id><pub-id pub-id-type="pmid">28488700</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bo</surname><given-names>H</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Zhu</surname><given-names>F</given-names></name><name><surname>Zhou</surname><given-names>D</given-names></name><name><surname>Tan</surname><given-names>Y</given-names></name><name><surname>Zhu</surname><given-names>W</given-names></name><name><surname>Fan</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Long noncoding RNAs expression profile and long noncoding RNA-mediated competing endogenous RNA network in nonobstructive azoospermia patients</article-title><source>Epigenomics</source><volume>12</volume><fpage>673</fpage><lpage>684</lpage><pub-id pub-id-type="doi">10.2217/epi-2020-0008</pub-id><pub-id pub-id-type="pmid">32174164</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brozovich</surname><given-names>F</given-names></name><name><surname>Nicholson</surname><given-names>CJ</given-names></name><name><surname>Degen</surname><given-names>C</given-names></name><name><surname>Gao</surname><given-names>YZ</given-names></name><name><surname>Aggarwal</surname><given-names>M</given-names></name><name><surname>Morgan</surname><given-names>KG</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mechanisms of vascular smooth muscle contraction and the basis for pharmacologic treatment of smooth muscle disorders</article-title><source>Pharmacological Reviews</source><volume>68</volume><fpage>476</fpage><lpage>532</lpage><pub-id pub-id-type="doi">10.1124/pr.115.010652</pub-id><pub-id pub-id-type="pmid">27037223</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buen Abad Najar</surname><given-names>CF</given-names></name><name><surname>Yosef</surname><given-names>N</given-names></name><name><surname>Lareau</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Coverage-dependent bias creates the appearance of binary splicing in single cells</article-title><source>eLife</source><volume>9</volume><elocation-id>e54603</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.54603</pub-id><pub-id pub-id-type="pmid">32597758</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buljan</surname><given-names>M</given-names></name><name><surname>Chalancon</surname><given-names>G</given-names></name><name><surname>Eustermann</surname><given-names>S</given-names></name><name><surname>Wagner</surname><given-names>GP</given-names></name><name><surname>Fuxreiter</surname><given-names>M</given-names></name><name><surname>Bateman</surname><given-names>A</given-names></name><name><surname>Babu</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Tissue-specific splicing of disordered segments that embed binding motifs rewires protein interaction networks</article-title><source>Molecular Cell</source><volume>46</volume><fpage>871</fpage><lpage>883</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2012.05.039</pub-id><pub-id pub-id-type="pmid">22749400</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dehghannasiri</surname><given-names>R</given-names></name><name><surname>Olivieri</surname><given-names>JE</given-names></name><name><surname>Damljanovic</surname><given-names>A</given-names></name><name><surname>Salzman</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Specific splice junction detection in single cells with SICILIAN</article-title><source>Genome Biology</source><volume>22</volume><elocation-id>219</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-021-02434-8</pub-id><pub-id pub-id-type="pmid">34353340</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dobin</surname><given-names>A</given-names></name><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Schlesinger</surname><given-names>F</given-names></name><name><surname>Drenkow</surname><given-names>J</given-names></name><name><surname>Zaleski</surname><given-names>C</given-names></name><name><surname>Jha</surname><given-names>S</given-names></name><name><surname>Batut</surname><given-names>P</given-names></name><name><surname>Chaisson</surname><given-names>M</given-names></name><name><surname>Gingeras</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>STAR: ultrafast universal RNA-seq aligner</article-title><source>Bioinformatics</source><volume>29</volume><fpage>15</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id><pub-id pub-id-type="pmid">23104886</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ellis</surname><given-names>JD</given-names></name><name><surname>Barrios-Rodiles</surname><given-names>M</given-names></name><name><surname>Colak</surname><given-names>R</given-names></name><name><surname>Irimia</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>T</given-names></name><name><surname>Calarco</surname><given-names>JA</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Pan</surname><given-names>Q</given-names></name><name><surname>O’Hanlon</surname><given-names>D</given-names></name><name><surname>Kim</surname><given-names>PM</given-names></name><name><surname>Wrana</surname><given-names>JL</given-names></name><name><surname>Blencowe</surname><given-names>BJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Tissue-specific alternative splicing remodels protein-protein interaction networks</article-title><source>Molecular Cell</source><volume>46</volume><fpage>884</fpage><lpage>892</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2012.05.037</pub-id><pub-id pub-id-type="pmid">22749401</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ezkurdia</surname><given-names>I</given-names></name><name><surname>Rodriguez</surname><given-names>JM</given-names></name><name><surname>Carrillo-de Santa Pau</surname><given-names>E</given-names></name><name><surname>Vázquez</surname><given-names>J</given-names></name><name><surname>Valencia</surname><given-names>A</given-names></name><name><surname>Tress</surname><given-names>ML</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Most highly expressed protein-coding genes have a single dominant isoform</article-title><source>Journal of Proteome Research</source><volume>14</volume><fpage>1880</fpage><lpage>1887</lpage><pub-id pub-id-type="doi">10.1021/pr501286b</pub-id><pub-id pub-id-type="pmid">25732134</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Florea</surname><given-names>L</given-names></name><name><surname>Song</surname><given-names>L</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Thousands of exon skipping events differentiate among splicing patterns in sixteen human tissues</article-title><source>F1000Research</source><volume>2</volume><elocation-id>188</elocation-id><pub-id pub-id-type="doi">10.12688/f1000research.2-188.v2</pub-id><pub-id pub-id-type="pmid">24555089</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ganesh</surname><given-names>S</given-names></name><name><surname>Horvat</surname><given-names>F</given-names></name><name><surname>Drutovic</surname><given-names>D</given-names></name><name><surname>Efenberkova</surname><given-names>M</given-names></name><name><surname>Pinkas</surname><given-names>D</given-names></name><name><surname>Jindrova</surname><given-names>A</given-names></name><name><surname>Pasulka</surname><given-names>J</given-names></name><name><surname>Iyyappan</surname><given-names>R</given-names></name><name><surname>Malik</surname><given-names>R</given-names></name><name><surname>Susor</surname><given-names>A</given-names></name><name><surname>Vlahovicek</surname><given-names>K</given-names></name><name><surname>Solc</surname><given-names>P</given-names></name><name><surname>Svoboda</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The most abundant maternal lncRNA Sirena1 acts post-transcriptionally and impacts mitochondrial distribution</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>3211</fpage><lpage>3227</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz1239</pub-id><pub-id pub-id-type="pmid">31956907</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gaviraghi</surname><given-names>M</given-names></name><name><surname>Vivori</surname><given-names>C</given-names></name><name><surname>Pareja Sanchez</surname><given-names>Y</given-names></name><name><surname>Invernizzi</surname><given-names>F</given-names></name><name><surname>Cattaneo</surname><given-names>A</given-names></name><name><surname>Santoliquido</surname><given-names>BM</given-names></name><name><surname>Frenquelli</surname><given-names>M</given-names></name><name><surname>Segalla</surname><given-names>S</given-names></name><name><surname>Bachi</surname><given-names>A</given-names></name><name><surname>Doglioni</surname><given-names>C</given-names></name><name><surname>Pelechano</surname><given-names>V</given-names></name><name><surname>Cittaro</surname><given-names>D</given-names></name><name><surname>Tonon</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Tumor suppressor PNRC1 blocks rRNA maturation by recruiting the decapping complex to the nucleolus</article-title><source>The EMBO Journal</source><volume>37</volume><elocation-id>23</elocation-id><pub-id pub-id-type="doi">10.15252/embj.201899179</pub-id><pub-id pub-id-type="pmid">30373810</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Genuth</surname><given-names>NR</given-names></name><name><surname>Barna</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Heterogeneity and specialized functions of translation machinery: from genes to organisms</article-title><source>Nature Reviews. Genetics</source><volume>19</volume><fpage>431</fpage><lpage>452</lpage><pub-id pub-id-type="doi">10.1038/s41576-018-0008-z</pub-id><pub-id pub-id-type="pmid">29725087</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giudice</surname><given-names>J</given-names></name><name><surname>Loehr</surname><given-names>JA</given-names></name><name><surname>Rodney</surname><given-names>GG</given-names></name><name><surname>Cooper</surname><given-names>TA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Alternative Splicing of Four Trafficking Genes Regulates Myofiber Structure and Skeletal Muscle Physiology</article-title><source>Cell Reports</source><volume>17</volume><fpage>1923</fpage><lpage>1933</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2016.10.072</pub-id><pub-id pub-id-type="pmid">27851958</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gonzàlez-Porta</surname><given-names>M</given-names></name><name><surname>Frankish</surname><given-names>A</given-names></name><name><surname>Rung</surname><given-names>J</given-names></name><name><surname>Harrow</surname><given-names>J</given-names></name><name><surname>Brazma</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Transcriptome analysis of human tissues and cell lines reveals one dominant transcript per gene</article-title><source>Genome Biology</source><volume>14</volume><elocation-id>R70</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2013-14-7-r70</pub-id><pub-id pub-id-type="pmid">23815980</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gooding</surname><given-names>C</given-names></name><name><surname>Smith</surname><given-names>CWJ</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Tropomyosin exons as models for alternative splicing</article-title><source>Advances in Experimental Medicine and Biology</source><volume>644</volume><fpage>27</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1007/978-0-387-85766-4_3</pub-id><pub-id pub-id-type="pmid">19209811</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Green</surname><given-names>CD</given-names></name><name><surname>Ma</surname><given-names>Q</given-names></name><name><surname>Manske</surname><given-names>GL</given-names></name><name><surname>Shami</surname><given-names>AN</given-names></name><name><surname>Zheng</surname><given-names>X</given-names></name><name><surname>Marini</surname><given-names>S</given-names></name><name><surname>Moritz</surname><given-names>L</given-names></name><name><surname>Sultan</surname><given-names>C</given-names></name><name><surname>Gurczynski</surname><given-names>SJ</given-names></name><name><surname>Moore</surname><given-names>BB</given-names></name><name><surname>Tallquist</surname><given-names>MD</given-names></name><name><surname>Li</surname><given-names>JZ</given-names></name><name><surname>Hammoud</surname><given-names>SS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A Comprehensive roadmap of murine spermatogenesis defined by single-cell RNA-Seq</article-title><source>Developmental Cell</source><volume>46</volume><fpage>651</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1016/j.devcel.2018.07.025</pub-id><pub-id pub-id-type="pmid">30146481</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname><given-names>V</given-names></name><name><surname>Warner</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Ribosome-omics of the human ribosome</article-title><source>RNA</source><volume>20</volume><fpage>1004</fpage><lpage>1013</lpage><pub-id pub-id-type="doi">10.1261/rna.043653.113</pub-id><pub-id pub-id-type="pmid">24860015</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hartmann</surname><given-names>B</given-names></name><name><surname>Castelo</surname><given-names>R</given-names></name><name><surname>Blanchette</surname><given-names>M</given-names></name><name><surname>Boue</surname><given-names>S</given-names></name><name><surname>Rio</surname><given-names>DC</given-names></name><name><surname>Valcárcel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Global analysis of alternative splicing regulation by insulin and wingless signaling in <italic>Drosophila</italic> cells</article-title><source>Genome Biology</source><volume>10</volume><elocation-id>R11</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2009-10-1-r11</pub-id><pub-id pub-id-type="pmid">19178699</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hermann</surname><given-names>BP</given-names></name><name><surname>Cheng</surname><given-names>K</given-names></name><name><surname>Singh</surname><given-names>A</given-names></name><name><surname>Roa-De La Cruz</surname><given-names>L</given-names></name><name><surname>Mutoji</surname><given-names>KN</given-names></name><name><surname>Chen</surname><given-names>IC</given-names></name><name><surname>Gildersleeve</surname><given-names>H</given-names></name><name><surname>Lehle</surname><given-names>JD</given-names></name><name><surname>Mayo</surname><given-names>M</given-names></name><name><surname>Westernströer</surname><given-names>B</given-names></name><name><surname>Law</surname><given-names>NC</given-names></name><name><surname>Oatley</surname><given-names>MJ</given-names></name><name><surname>Velte</surname><given-names>EK</given-names></name><name><surname>Niedenberger</surname><given-names>BA</given-names></name><name><surname>Fritze</surname><given-names>D</given-names></name><name><surname>Silber</surname><given-names>S</given-names></name><name><surname>Geyer</surname><given-names>CB</given-names></name><name><surname>Oatley</surname><given-names>JM</given-names></name><name><surname>McCarrey</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The mammalian spermatogenesis single-cell transcriptome, from spermatogonial stem cells to spermatids</article-title><source>Cell Reports</source><volume>25</volume><fpage>1650</fpage><lpage>1667</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2018.10.026</pub-id><pub-id pub-id-type="pmid">30404016</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hochberg</surname><given-names>Y</given-names></name><name><surname>Benjamini</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="1990">1990</year><article-title>More powerful procedures for multiple significance testing</article-title><source>In Statistics in Medicine</source><volume>9</volume><fpage>811</fpage><lpage>818</lpage><pub-id pub-id-type="doi">10.1002/sim.4780090710</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>Y</given-names></name><name><surname>Sanguinetti</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>BRIE: transcriptome-wide splicing quantification in single cells</article-title><source>Genome Biology</source><volume>18</volume><elocation-id>123</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-017-1248-5</pub-id><pub-id pub-id-type="pmid">28655331</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Julia</surname><given-names>EO</given-names></name></person-group><year iso-8601-date="2021">2021a</year><data-title>Diffsplice</data-title><version designator="swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830">swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8ce4ed66a931f705fc924095639b3c02efe9a944;origin=https://github.com/juliaolivieri/DiffSplice;visit=swh:1:snp:3c0aaffa3eedcba8221494094dc49ecdbf13cb7a;anchor=swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830">https://archive.softwareheritage.org/swh:1:dir:8ce4ed66a931f705fc924095639b3c02efe9a944;origin=https://github.com/juliaolivieri/DiffSplice;visit=swh:1:snp:3c0aaffa3eedcba8221494094dc49ecdbf13cb7a;anchor=swh:1:rev:6fa54f473eb55c9e68692a6aa1d92d479e56b830</ext-link></element-citation></ref><ref id="bib27"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Julia</surname><given-names>EO</given-names></name></person-group><year iso-8601-date="2021">2021b</year><data-title>SPLIZ pipeline</data-title><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/juliaolivieri/SpliZ_pipeline">https://github.com/juliaolivieri/SpliZ_pipeline</ext-link></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Keren</surname><given-names>H</given-names></name><name><surname>Lev-Maor</surname><given-names>G</given-names></name><name><surname>Ast</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Alternative splicing and evolution: diversification, exon definition and function</article-title><source>Nature Reviews Genetics</source><volume>11</volume><fpage>345</fpage><lpage>355</lpage><pub-id pub-id-type="doi">10.1038/nrg2776</pub-id><pub-id pub-id-type="pmid">20376054</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kondrashov</surname><given-names>N</given-names></name><name><surname>Pusic</surname><given-names>A</given-names></name><name><surname>Stumpf</surname><given-names>CR</given-names></name><name><surname>Shimizu</surname><given-names>K</given-names></name><name><surname>Hsieh</surname><given-names>AC</given-names></name><name><surname>Ishijima</surname><given-names>J</given-names></name><name><surname>Shiroishi</surname><given-names>T</given-names></name><name><surname>Barna</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Ribosome-mediated specificity in Hox mRNA translation and vertebrate tissue patterning</article-title><source>Cell</source><volume>145</volume><fpage>383</fpage><lpage>397</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2011.03.028</pub-id><pub-id pub-id-type="pmid">21529712</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname><given-names>JT</given-names></name><name><surname>Ham</surname><given-names>S</given-names></name><name><surname>Jeon</surname><given-names>S</given-names></name><name><surname>Kim</surname><given-names>Y</given-names></name><name><surname>Oh</surname><given-names>S</given-names></name><name><surname>Cho</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Expression of uncharacterized male germ cell-specific genes and discovery of novel sperm-tail proteins in mice</article-title><source>PLOS ONE</source><volume>12</volume><elocation-id>e0182038</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0182038</pub-id><pub-id pub-id-type="pmid">28742876</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname><given-names>B</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Fast gapped-read alignment with Bowtie 2</article-title><source>Nature Methods</source><volume>9</volume><fpage>357</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id><pub-id pub-id-type="pmid">22388286</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Lee</surname><given-names>JA</given-names></name><name><surname>Black</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Neuronal regulation of alternative pre-mRNA splicing</article-title><source>Nature Reviews Neuroscience</source><volume>8</volume><fpage>819</fpage><lpage>831</lpage><pub-id pub-id-type="doi">10.1038/nrn2237</pub-id><pub-id pub-id-type="pmid">17895907</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>L</given-names></name><name><surname>Park</surname><given-names>JW</given-names></name><name><surname>Ramachandran</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Tseng</surname><given-names>YT</given-names></name><name><surname>Shen</surname><given-names>S</given-names></name><name><surname>Waldvogel</surname><given-names>HJ</given-names></name><name><surname>Curtis</surname><given-names>MA</given-names></name><name><surname>Faull</surname><given-names>RLM</given-names></name><name><surname>Troncoso</surname><given-names>JC</given-names></name><name><surname>Pletnikova</surname><given-names>O</given-names></name><name><surname>Ross</surname><given-names>CA</given-names></name><name><surname>Davidson</surname><given-names>BL</given-names></name><name><surname>Xing</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Transcriptome sequencing reveals aberrant alternative splicing in Huntington’s disease</article-title><source>Human Molecular Genetics</source><volume>25</volume><fpage>3454</fpage><lpage>3466</lpage><pub-id pub-id-type="doi">10.1093/hmg/ddw187</pub-id><pub-id pub-id-type="pmid">27378699</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>YH</given-names></name><name><surname>Zhen</surname><given-names>YY</given-names></name><name><surname>Chien</surname><given-names>KY</given-names></name><name><surname>Lee</surname><given-names>IC</given-names></name><name><surname>Lin</surname><given-names>WC</given-names></name><name><surname>Chen</surname><given-names>MY</given-names></name><name><surname>Pai</surname><given-names>LM</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>LIMCH1 regulates nonmuscle myosin-II activity and suppresses cell migration</article-title><source>Molecular Biology of the Cell</source><volume>28</volume><fpage>1054</fpage><lpage>1065</lpage><pub-id pub-id-type="doi">10.1091/mbc.e15-04-0218</pub-id><pub-id pub-id-type="pmid">28228547</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martinez</surname><given-names>NM</given-names></name><name><surname>Pan</surname><given-names>Q</given-names></name><name><surname>Cole</surname><given-names>BS</given-names></name><name><surname>Yarosh</surname><given-names>CA</given-names></name><name><surname>Babcock</surname><given-names>GA</given-names></name><name><surname>Heyd</surname><given-names>F</given-names></name><name><surname>Zhu</surname><given-names>W</given-names></name><name><surname>Ajith</surname><given-names>S</given-names></name><name><surname>Blencowe</surname><given-names>BJ</given-names></name><name><surname>Lynch</surname><given-names>KW</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Alternative splicing networks regulated by signaling in human T cells</article-title><source>RNA</source><volume>18</volume><fpage>1029</fpage><lpage>1040</lpage><pub-id pub-id-type="doi">10.1261/rna.032243.112</pub-id><pub-id pub-id-type="pmid">22454538</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Megill</surname><given-names>C</given-names></name><name><surname>Martin</surname><given-names>B</given-names></name><name><surname>Weaver</surname><given-names>C</given-names></name><name><surname>Bell</surname><given-names>S</given-names></name><name><surname>Prins</surname><given-names>L</given-names></name><name><surname>Badajoz</surname><given-names>S</given-names></name><name><surname>McCandless</surname><given-names>B</given-names></name><name><surname>Pisco</surname><given-names>AO</given-names></name><name><surname>Kinsella</surname><given-names>M</given-names></name><name><surname>Griffin</surname><given-names>F</given-names></name><name><surname>Kiggins</surname><given-names>J</given-names></name><name><surname>Haliburton</surname><given-names>G</given-names></name><name><surname>Mani</surname><given-names>A</given-names></name><name><surname>Weiden</surname><given-names>M</given-names></name><name><surname>Dunitz</surname><given-names>M</given-names></name><name><surname>Lombardo</surname><given-names>M</given-names></name><name><surname>Huang</surname><given-names>T</given-names></name><name><surname>Smith</surname><given-names>T</given-names></name><name><surname>Chambers</surname><given-names>S</given-names></name><name><surname>Carr</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Cellxgene: A Performant, Scalable Exploration Platform for High Dimensional Sparse Matrices</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.04.05.438318</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merkin</surname><given-names>J</given-names></name><name><surname>Russell</surname><given-names>C</given-names></name><name><surname>Chen</surname><given-names>P</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Evolutionary dynamics of gene and isoform regulation in Mammalian tissues</article-title><source>Science</source><volume>338</volume><fpage>1593</fpage><lpage>1599</lpage><pub-id pub-id-type="doi">10.1126/science.1228186</pub-id><pub-id pub-id-type="pmid">23258891</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Olivieri</surname><given-names>JE</given-names></name><name><surname>Dehghannasiri</surname><given-names>R</given-names></name><name><surname>Salzman</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The SpliZ generalizes percent spliced in to reveal regulated splicing at single-cell resolution</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.11.10.377572</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patrick</surname><given-names>R</given-names></name><name><surname>Humphreys</surname><given-names>DT</given-names></name><name><surname>Janbandhu</surname><given-names>V</given-names></name><name><surname>Oshlack</surname><given-names>A</given-names></name><name><surname>Ho</surname><given-names>JWK</given-names></name><name><surname>Harvey</surname><given-names>RP</given-names></name><name><surname>Lo</surname><given-names>KK</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Sierra: discovery of differential transcript usage from polyA-captured single-cell RNA-seq data</article-title><source>Genome Biology</source><volume>21</volume><elocation-id>167</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-020-02071-7</pub-id><pub-id pub-id-type="pmid">32641141</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Roozbeh</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>SICILIAN</data-title><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/salzmanlab/SICILIAN">https://github.com/salzmanlab/SICILIAN</ext-link></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>See</surname><given-names>P</given-names></name><name><surname>Lum</surname><given-names>J</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Ginhoux</surname><given-names>F</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Corrigendum: a single-cell sequencing guide for immunologists</article-title><source>Frontiers in Immunology</source><volume>10</volume><elocation-id>278</elocation-id><pub-id pub-id-type="doi">10.3389/fimmu.2019.00278</pub-id><pub-id pub-id-type="pmid">30863399</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Setarehbadi</surname><given-names>R</given-names></name><name><surname>Vatannejad</surname><given-names>A</given-names></name><name><surname>Vaisi-Raygani</surname><given-names>A</given-names></name><name><surname>Amiri</surname><given-names>I</given-names></name><name><surname>Esfahani</surname><given-names>M</given-names></name><name><surname>Fattahi</surname><given-names>A</given-names></name><name><surname>Tavilani</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Apolipoprotein E genotypes of fertile and infertile men</article-title><source>Systems Biology in Reproductive Medicine</source><volume>58</volume><fpage>263</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.3109/19396368.2012.684134</pub-id><pub-id pub-id-type="pmid">22568769</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sha</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Yuan</surname><given-names>J</given-names></name><name><surname>Zhu</surname><given-names>X</given-names></name><name><surname>Su</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Xu</surname><given-names>X</given-names></name><name><surname>Wei</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Loss-of-function mutations in centrosomal protein 112 is associated with human acephalic spermatozoa phenotype</article-title><source>Clinical Genetics</source><volume>97</volume><fpage>321</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1111/cge.13662</pub-id><pub-id pub-id-type="pmid">31654588</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname><given-names>SR</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Zhao</surname><given-names>J</given-names></name><name><surname>Zeng</surname><given-names>X</given-names></name><name><surname>Hou</surname><given-names>SX</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The novel tumour suppressor madm regulates stem cell competition in the <italic>Drosophila</italic> testis</article-title><source>Nature Communications</source><volume>7</volume><elocation-id>10473</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms10473</pub-id><pub-id pub-id-type="pmid">26792023</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Song</surname><given-names>Y</given-names></name><name><surname>Botvinnik</surname><given-names>OB</given-names></name><name><surname>Lovci</surname><given-names>MT</given-names></name><name><surname>Kakaradov</surname><given-names>B</given-names></name><name><surname>Liu</surname><given-names>P</given-names></name><name><surname>Xu</surname><given-names>JL</given-names></name><name><surname>Yeo</surname><given-names>GW</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Single-cell alternative splicing analysis with expedition reveals splicing dynamics during neuron differentiation</article-title><source>Molecular Cell</source><volume>67</volume><fpage>148</fpage><lpage>161</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2017.06.003</pub-id><pub-id pub-id-type="pmid">28673540</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soumillon</surname><given-names>M</given-names></name><name><surname>Necsulea</surname><given-names>A</given-names></name><name><surname>Weier</surname><given-names>M</given-names></name><name><surname>Brawand</surname><given-names>D</given-names></name><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Gu</surname><given-names>H</given-names></name><name><surname>Barthès</surname><given-names>P</given-names></name><name><surname>Kokkinaki</surname><given-names>M</given-names></name><name><surname>Nef</surname><given-names>S</given-names></name><name><surname>Gnirke</surname><given-names>A</given-names></name><name><surname>Dym</surname><given-names>M</given-names></name><name><surname>de Massy</surname><given-names>B</given-names></name><name><surname>Mikkelsen</surname><given-names>TS</given-names></name><name><surname>Kaessmann</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Cellular source and mechanisms of high transcriptome complexity in the mammalian testis</article-title><source>Cell Reports</source><volume>3</volume><fpage>2179</fpage><lpage>2190</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2013.05.031</pub-id><pub-id pub-id-type="pmid">23791531</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Svensson</surname><given-names>V</given-names></name><name><surname>da Veiga Beltrame</surname><given-names>E</given-names></name><name><surname>Pachter</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A curated database reveals trends in single-cell transcriptomics</article-title><source>Database</source><volume>2020</volume><elocation-id>073</elocation-id><pub-id pub-id-type="doi">10.1093/database/baaa073</pub-id><pub-id pub-id-type="pmid">33247933</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="web"><person-group person-group-type="author"><collab>Tabula Microcebus Consortium</collab></person-group><year iso-8601-date="2021">2021</year><article-title>Tabula Microcebus</article-title><ext-link ext-link-type="uri" xlink:href="https://tabula-microcebus.ds.czbiohub.org/about">https://tabula-microcebus.ds.czbiohub.org/about</ext-link><date-in-citation iso-8601-date="2021-01-10">January 10, 2021</date-in-citation></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>Tabula Muris Consortium</collab></person-group><year iso-8601-date="2018">2018</year><article-title>Single-cell transcriptomics of 20 mouse organs creates a Tabula Muris</article-title><source>Nature</source><volume>562</volume><fpage>367</fpage><lpage>372</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0590-4</pub-id><pub-id pub-id-type="pmid">30283141</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="preprint"><person-group person-group-type="author"><collab>Tabula Sapiens Consortium</collab></person-group><year iso-8601-date="2021">2021</year><article-title>The Tabula Sapiens: A Single Cell Transcriptomic Atlas of Multiple Organs from Individual Human Donors</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.07.19.452956</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Travaglini</surname><given-names>KJ</given-names></name><name><surname>Nabhan</surname><given-names>AN</given-names></name><name><surname>Penland</surname><given-names>L</given-names></name><name><surname>Sinha</surname><given-names>R</given-names></name><name><surname>Gillich</surname><given-names>A</given-names></name><name><surname>Sit</surname><given-names>R</given-names></name><name><surname>Chang</surname><given-names>S</given-names></name><name><surname>Conley</surname><given-names>SD</given-names></name><name><surname>Mori</surname><given-names>Y</given-names></name><name><surname>Seita</surname><given-names>J</given-names></name><name><surname>Berry</surname><given-names>GJ</given-names></name><name><surname>Shrager</surname><given-names>JB</given-names></name><name><surname>Metzger</surname><given-names>RJ</given-names></name><name><surname>Kuo</surname><given-names>CS</given-names></name><name><surname>Neff</surname><given-names>N</given-names></name><name><surname>Weissman</surname><given-names>IL</given-names></name><name><surname>Quake</surname><given-names>SR</given-names></name><name><surname>Krasnow</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A molecular cell atlas of the human lung from single-cell RNA sequencing</article-title><source>Nature</source><volume>587</volume><fpage>619</fpage><lpage>625</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2922-4</pub-id><pub-id pub-id-type="pmid">33208946</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ule</surname><given-names>J</given-names></name><name><surname>Blencowe</surname><given-names>BJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Alternative Splicing Regulatory Networks: Functions, Mechanisms, and Evolution</article-title><source>Molecular Cell</source><volume>76</volume><fpage>329</fpage><lpage>345</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2019.09.017</pub-id><pub-id pub-id-type="pmid">31626751</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vedula</surname><given-names>P</given-names></name><name><surname>Kurosaka</surname><given-names>S</given-names></name><name><surname>Leu</surname><given-names>NA</given-names></name><name><surname>Wolf</surname><given-names>Y</given-names></name><name><surname>Shabalina</surname><given-names>SA</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Sterling</surname><given-names>S</given-names></name><name><surname>Dong</surname><given-names>DW</given-names></name><name><surname>Kashina</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Diverse functions of homologous actin isoforms are defined by their nucleotide, rather than their amino acid sequence</article-title><source>eLife</source><volume>6</volume><elocation-id>e31661</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.31661</pub-id><pub-id pub-id-type="pmid">29244021</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Chen</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Lou</surname><given-names>G</given-names></name><name><surname>Chen</surname><given-names>S</given-names></name><name><surname>Zhou</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Identification and characterization of PNRC splicing variants</article-title><source>Gene</source><volume>423</volume><fpage>116</fpage><lpage>124</lpage><pub-id pub-id-type="doi">10.1016/j.gene.2008.07.018</pub-id><pub-id pub-id-type="pmid">18703122</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Westoby</surname><given-names>J</given-names></name><name><surname>Artemov</surname><given-names>P</given-names></name><name><surname>Hemberg</surname><given-names>M</given-names></name><name><surname>Ferguson-Smith</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Obstacles to detecting isoforms using full-length scRNA-seq data</article-title><source>Genome Biology</source><volume>21</volume><elocation-id>74</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-020-01981-w</pub-id><pub-id pub-id-type="pmid">32293520</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>X</given-names></name><name><surname>Coulombe-Huntington</surname><given-names>J</given-names></name><name><surname>Kang</surname><given-names>S</given-names></name><name><surname>Sheynkman</surname><given-names>GM</given-names></name><name><surname>Hao</surname><given-names>T</given-names></name><name><surname>Richardson</surname><given-names>A</given-names></name><name><surname>Sun</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>F</given-names></name><name><surname>Shen</surname><given-names>YA</given-names></name><name><surname>Murray</surname><given-names>RR</given-names></name><name><surname>Spirohn</surname><given-names>K</given-names></name><name><surname>Begg</surname><given-names>BE</given-names></name><name><surname>Duran-Frigola</surname><given-names>M</given-names></name><name><surname>MacWilliams</surname><given-names>A</given-names></name><name><surname>Pevzner</surname><given-names>SJ</given-names></name><name><surname>Zhong</surname><given-names>Q</given-names></name><name><surname>Trigg</surname><given-names>SA</given-names></name><name><surname>Tam</surname><given-names>S</given-names></name><name><surname>Ghamsari</surname><given-names>L</given-names></name><name><surname>Vidal</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Widespread Expansion of Protein Interaction Capabilities by Alternative Splicing</article-title><source>Cell</source><volume>164</volume><fpage>805</fpage><lpage>817</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.01.029</pub-id><pub-id pub-id-type="pmid">26871637</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Chen</surname><given-names>MH</given-names></name><name><surname>Wu</surname><given-names>X</given-names></name><name><surname>Kodani</surname><given-names>A</given-names></name><name><surname>Fan</surname><given-names>J</given-names></name><name><surname>Doan</surname><given-names>R</given-names></name><name><surname>Ozawa</surname><given-names>M</given-names></name><name><surname>Ma</surname><given-names>J</given-names></name><name><surname>Yoshida</surname><given-names>N</given-names></name><name><surname>Reiter</surname><given-names>JF</given-names></name><name><surname>Black</surname><given-names>DL</given-names></name><name><surname>Kharchenko</surname><given-names>P</given-names></name><name><surname>Sharp</surname><given-names>PA</given-names></name><name><surname>Walsh</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Cell-Type-Specific Alternative Splicing Governs Cell Fate in the Developing Cerebral Cortex</article-title><source>Cell</source><volume>166</volume><fpage>1147</fpage><lpage>1162</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.07.025</pub-id><pub-id pub-id-type="pmid">27565344</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname><given-names>F</given-names></name><name><surname>Yan</surname><given-names>P</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Cui</surname><given-names>Y</given-names></name><name><surname>Zheng</surname><given-names>M</given-names></name><name><surname>Cheng</surname><given-names>Y</given-names></name><name><surname>Guo</surname><given-names>Y</given-names></name><name><surname>Yang</surname><given-names>X</given-names></name><name><surname>Guo</surname><given-names>X</given-names></name><name><surname>Zhu</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Deficiency of TPPP2, a factor linked to oligoasthenozoospermia, causes subfertility in male mice</article-title><source>Journal of Cellular and Molecular Medicine</source><volume>23</volume><fpage>2583</fpage><lpage>2594</lpage><pub-id pub-id-type="doi">10.1111/jcmm.14149</pub-id><pub-id pub-id-type="pmid">30680919</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zipursky</surname><given-names>SL</given-names></name><name><surname>Sanes</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Chemoaffinity revisited: dscams, protocadherins, and neural circuit assembly</article-title><source>Cell</source><volume>143</volume><fpage>343</fpage><lpage>353</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2010.10.009</pub-id><pub-id pub-id-type="pmid">21029858</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.70692.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Yeo</surname><given-names>Gene W</given-names></name><role>Reviewing Editor</role><aff><institution>University of California, San Diego</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2021.05.01.442281">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2021.05.01.442281v1">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>This study describes an analysis of cell type-specific alternative splicing using 10x scRNA-seq data. This work shows that in spite of the challenges associated with the analysis of such datasets, it is possible to identify alternative exons with differential splicing between tissue compartments and to some extent reveal cell types by splicing profiles of single cells. This work is informative regarding what can be done to analyze alternative splicing using 10X data and fills in a gap in the field. Your revised manuscript addresses reviewers' concerns and strengthen the manuscript for the general audience and we are most appreciative of it.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;RNA splicing programs define tissue compartments and cell types at single cell resolution&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 2 peer reviewers, one of whom is a member of our Board of Reviewing Editors, and the evaluation has been overseen by Patricia Wittkopp as the Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>1) The computational analysis appears to be solid in general but the presentation, as in the current form, need to be improved before publication.</p><p>2) Overall, there is little doubt that alternative exons near the 3' end of the transcripts can be studied by 10X data, but the scope is relatively limited. This is confirmed in this study, as only 1353 genes can be quantified at the exon level and only 22 genes were identified to have differential splicing. The study needs to be very clearly discuss this major limitation and balance the pros and cons of their method.</p><p>3) The algorithm is under review at another journal and has made the review process here difficult (several reviewers bowed out for this reason) – thus is is very important for SpliZ to be thoroughly discussed here. It would actually be even better if the paper describing the method were accepted for publication prior to publication of this work.</p><p>4) Some claims are made that are not fully substantiated by the data, which could be addressed by adding more context to the analyses. Furthermore, some figures can be tweaked or expanded upon in the text to improve clarity. In particular, the following sections should be amended:</p><p>&quot;Supporting the idea that SpliZsites 31 discover real biological signal, 13% of the SpliZsites in the human are also identified as SpliZsites in the mouse lemur and/or mouse (Methods).&quot;</p><p>Insufficient context is provided here for readers for this conclusion. The authors should compare this number to a shuffled dataset or other background data to demonstrate the supposed significance of the 13% number provided.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>Some claims are made that are not fully substantiated by the data, which could be addressed by adding more context to the analyses. Furthermore, some figures can be tweaked or expanded upon in the text to improve clarity. In particular, the following sections should be amended:</p><p>&quot;Supporting the idea that SpliZsites 31 discover real biological signal, 13% of the SpliZsites in the human are also identified as SpliZsites in the mouse lemur and/or mouse (Methods).&quot;</p><p>Insufficient context is provided here for readers for this conclusion. The authors should compare this number to a shuffled dataset or other background data to demonstrate the supposed significance of the 13% number provided.</p><p>In Figure 2B the authors demonstrate prediction of cellular compartment of cells using k-means clustering analysis on SpliZ scores from two genes. The claim in the main text: &quot;Setting k=3, cells from stromal, epithelial, and immune compartments were classified with accuracies of 78%, 84%, and 95% respectively independent of gene expression&quot;. Though these were the results from one of the two individuals in the dataset, the accuracies were much worse for the other individual, and the text is misleading here in only focusing on the cleaner data. The authors should acknowledge this in the text, and address the possible factors causing lower accuracy in individual 2. Furthermore, given the cells come from 4 tissue compartments (immune, epithelial, endothelial and stromal), the authors should elaborate on the decision to set k to 3.</p><p>Figure 3A, a legend for the squares and the circles, indicating 10X and Smart-seq can be more clear. (like those in Supp Figure 2)</p><p>Figure 7B, though the splicing changes are correlated across the species, the directionality of splice site usage is inverted across species. The authors can discuss more on the biological meaning.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>The computational analysis appears to be solid in general but the presentation, as in the current form, need to be improved before publication.</p><p>1. Overall, I do not have doubt that alternative exons near the 3' end of the transcripts can be studied by 10X data, but the extend is relatively limited. This is confirmed in this study, as only 1353 genes can be quantified at the exon level and only 22 genes were identified to have differential splicing. I think despite the limitation, efforts to mine splicing using 10X data should still be encouraged, given the explosive number of datasets available. However, the discussion of the pro and cons should be balanced (e.g., the 3' bias should be discussed).</p><p>2. The authors used a new pipeline named SpliZ for analysis (preprint cited as ref. 26), and several variations SpliZsite and SpliZVD were also used. While it is fine to present technical details in separate publications, key features have to be described in the manuscript, which is required for understanding the results. For example, what does SpliZ measure (something similar to PSI I assume), what is SpliZsite (filtered splice sites from STAR alignments?), what is the difference beteen SpliZ and SpliZVD? How dropout and UMIs are handled in the pipeline?</p><p>3. Insufficient descriptions were provided in multiple figures.</p><p>Figure 2A, circles and squares were not explained (they were explained in Figure 3 below). How the dot plots are related to the splice sites shown in the sashimi plots and how the splice sites in the sashimi plots are related to the gene structure schematics (need some guesswork)? What is shown in the boxplot (labelled &quot;Average 3' splice site per cell&quot;, which I assume it SpliZ score, but again is this the fraction of reads that uses the upstream 3' splice site?)</p><p>Figure 2C, D. Is each dot a single cell or a &quot;meta cell&quot; that averages a certain number of individual cells?</p><p>Figure 3A, how the gene structure schematic relate to the boxplot above is confusing. Also in the gene schematics of the three species further down, it is confusing why two parts of the human gene were highlighted (only the two 3' splice sites near the 3' end are relevant?). Unclear what is shown in the illustrations on the right of the gene schematics.</p><p>Figure 3D. How exons labeled 5,6,7 are related to site 1 and 2 in A-C (need guesswork)?</p><p>Figure 4. Only read fraction is shown but not SpliZ scores?</p><p>Figure 5A, similar to the question for Figure 2A, how the gene structure schematic relate to the boxplot above is confusing.</p><p>Figure 6C. It does not seem to be the evidence of a single exon to distinguish two subpopulations of monocytes are convincing. How do we know whether the two subpopulations reflect certain technical issues (potentially similar to the controversial &quot;bimodal splicing&quot; proposed in previous publications)?</p><p>Similar confusion in Figure 7 as in Figures 2 and 5.</p><p>4. I thought some global evaluation on the reliability of 10X results using independent datasets will be important (e.g., correlation of differential splicing in comparison with results from bulk RNA-seq and/or SMART-seq data). Some of the results presented in Supplementary Figures(e.g., Figure S6) can probably presented as the main figure.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.70692.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>1) The computational analysis appears to be solid in general but the presentation, as in the current form, need to be improved before publication.</p></disp-quote><p>We are pleased to hear that you find the analysis solid in general. We have now addressed all suggestions and comments made by the reviewers to greatly improve the presentation of this manuscript for publication, including revising figures to make interpretation more clear.</p><disp-quote content-type="editor-comment"><p>2) Overall, there is little doubt that alternative exons near the 3' end of the transcripts can be studied by 10X data, but the scope is relatively limited. This is confirmed in this study, as only 1353 genes can be quantified at the exon level and only 22 genes were identified to have differential splicing. The study needs to be very clearly discuss this major limitation and balance the pros and cons of their method.</p></disp-quote><p>We appreciate the concern that the limited scope of 10X data for splicing discovery was not emphasized enough in the original draft. To clarify, 22 genes were found to have differential compartment-specific splicing, while 129 genes were found to have cell-type-specific differential splicing out of 1,416 genes (page 6). We have added further clarification of the pros and cons of our method on page 9 in the revised manuscript:</p><p>“Although the SpliZ method enables biological discovery of splicing differences based on droplet-based sequencing data, droplet-based data still presents major challenges for splicing analysis compared to full-length data. In this study, droplet-based sequencing has much lower sequencing coverage than full-length data, resulting in only 1,416 genes with measurable SpliZ values in the first human individual based on 10X data compared to 9,802 genes with measurable SpliZ values in SS2 data. Additionally, current droplet-based data is 3-prime-biased, meaning that some splicing events will never be sequenced by the technology and therefore cannot be analyzed. Despite these challenges, the ubiquity of droplet-based data, its utility for profiling rare cell types, and its unprecedented scale make it a useful resource for splicing analysis.”</p><disp-quote content-type="editor-comment"><p>3) The algorithm is under review at another journal and has made the review process here difficult (several reviewers bowed out for this reason) – thus is is very important for SpliZ to be thoroughly discussed here. It would actually be even better if the paper describing the method were accepted for publication prior to publication of this work.</p></disp-quote><p>We understand that the SpliZ paper not being published makes it more difficult to review this manuscript. The methods paper for the SpliZ is currently in review at Nature Methods. It has been recently reviewed and based on the positive reviewers’ comments the Editor invited us for resubmission. We have now submitted the revision and it is currently being re-evaluated by reviewers. We are happy to share the comments from the reviewers on that manuscript if it would help you make a decision. We have also now added a thorough explanation of the SpliZ to the methods in a new section called “Explanation of the SpliZ method” on page 12 and added several more sentences of explanation to the main text at top of page 3:</p><p>“A large negative (resp. positive) SpliZ score for a gene in a cell means that the cell has shorter (resp. longer) introns than average for that gene. In the simplest exon skipping case, the SpliZ reduces to PSI.”</p><disp-quote content-type="editor-comment"><p>4) Some claims are made that are not fully substantiated by the data, which could be addressed by adding more context to the analyses. Furthermore, some figures can be tweaked or expanded upon in the text to improve clarity.</p></disp-quote><p>We are grateful for the thoughtful reading of the paper that revealed the lack of clarity in some analyses and figures. We have now added more context everywhere it was requested.</p><disp-quote content-type="editor-comment"><p>In particular, the following sections should be amended:</p><p>&quot;Supporting the idea that SpliZsites 31 discover real biological signal, 13% of the SpliZsites in the human are also identified as SpliZsites in the mouse lemur and/or mouse (Methods).&quot;</p><p>Insufficient context is provided here for readers for this conclusion. The authors should compare this number to a shuffled dataset or other background data to demonstrate the supposed significance of the 13% number provided.</p></disp-quote><p>We appreciate the point that the 13% fraction is provided without enough context for the reader to understand its significance. To add more clarity to our analysis, we now report the fraction of shared SpliZsites with mouse lemur and mouse separately. We have now changed the text on page 7 as follows:</p><p>“15.5% of LiftOver human SpliZsites were also SpliZsites in the mouse lemur compared to 7% expected under the null (Methods). Only 8.0% of LiftOver SpliZsites in human were called as SpliZsites in mouse compared to 8.8% expected under the null. This could be due to many factors including a larger evolutionary distance between mouse and human, smaller number of analyzed mouse cells, or lower sequencing depth (Suppl. Table 1)”</p><p>We also provided the details of our statistical analysis in a new section “SpliZsite analysis” in the Methods on page 16. We should point out that SpliZsites between organisms are not completely comparable due to lack of perfect LiftOver mappings between organisms and also different read depth and number of cells for the same gene in different organisms which can lead to a SpliZsite for a gene not being detected in all organisms.</p><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>Some claims are made that are not fully substantiated by the data, which could be addressed by adding more context to the analyses. Furthermore, some figures can be tweaked or expanded upon in the text to improve clarity. In particular, the following sections should be amended:</p><p>&quot;Supporting the idea that SpliZsites 31 discover real biological signal, 13% of the SpliZsites in the human are also identified as SpliZsites in the mouse lemur and/or mouse (Methods).&quot;</p><p>Insufficient context is provided here for readers for this conclusion. The authors should compare this number to a shuffled dataset or other background data to demonstrate the supposed significance of the 13% number provided.</p></disp-quote><p>We appreciate the point that the 13% fraction is provided without enough context for the reader to understand its significance. To add more clarity to our analysis, we now report the fraction of shared SpliZsites with mouse lemur and mouse separately. We have now changed the text on page 7 as follows:</p><p>“15.5% of LiftOver human SpliZsites were also SpliZsites in the mouse lemur compared to 7% expected under the null (Methods). Only 8.0% of LiftOver SpliZsites in human were called as SpliZsites in mouse compared to 8.8% expected under the null. This could be due to many factors including a larger evolutionary distance between mouse and human, smaller number of analyzed mouse cells, or lower sequencing depth (Suppl. Table 1)”</p><p>We also provided the details of our statistical analysis in a new section “SpliZsite analysis” in the Methods on page 16. We should point out that SpliZsites between organisms are not completely comparable due to lack of perfect LiftOver mappings between organisms and also different read depth and number of cells for the same gene in different organisms which can lead to a SpliZsite for a gene not being detected in all organisms.</p><disp-quote content-type="editor-comment"><p>In Figure 2B the authors demonstrate prediction of cellular compartment of cells using k-means clustering analysis on SpliZ scores from two genes. The claim in the main text: &quot;Setting k=3, cells from stromal, epithelial, and immune compartments were classified with accuracies of 78%, 84%, and 95% respectively independent of gene expression&quot;. Though these were the results from one of the two individuals in the dataset, the accuracies were much worse for the other individual, and the text is misleading here in only focusing on the cleaner data. The authors should acknowledge this in the text, and address the possible factors causing lower accuracy in individual 2. Furthermore, given the cells come from 4 tissue compartments (immune, epithelial, endothelial and stromal), the authors should elaborate on the decision to set k to 3.</p></disp-quote><p>Thank you for pointing this out. We have now added the accuracies for the second individual to the main text, included a discussion of why the accuracies may be worse in the second individual, and moved the explanation of not including the endothelial compartment from the methods to the main text (see paragraph 2 on page 4):</p><p>“Setting k=3, cells from stromal, epithelial, and immune compartments were classified with accuracies of 78%, 84%, and 95% respectively independent of gene expression in the first human individual (70%, 100%, and 49% in the second individual) (Figure 2B-D, Methods). The lower accuracy for individual 2 may be caused by individual 2 having only a third as many cells. The endothelial compartment was not included because it had a small proportion of cells in both datasets (3.7% in individual 1, 4.5% in individual 2).”</p><disp-quote content-type="editor-comment"><p>Figure 3A, a legend for the squares and the circles, indicating 10X and Smart-seq can be more clear. (like those in Supp Figure 2)</p></disp-quote><p>Thank you for your comment. We have now added a legend for Figure 3A (to the right of panel C) in the revised figure.</p><disp-quote content-type="editor-comment"><p>Figure 7B, though the splicing changes are correlated across the species, the directionality of splice site usage is inverted across species. The authors can discuss more on the biological meaning.</p></disp-quote><p>Thank you for pointing this out. The reason for the opposite direction of the splicing changes is that in human, <italic>CEP112</italic> is one the minus strand but it is on the plus strand in mouse and lemur genomes. We have now clarified this in the revised caption.</p><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>The computational analysis appears to be solid in general but the presentation, as in the current form, need to be improved before publication.</p></disp-quote><p>Thank you for your detailed suggestions for how to improve the presentation of the paper. We have taken all feedback into account to make the figures and paper overall much clearer.</p><disp-quote content-type="editor-comment"><p>My major comments:</p><p>1. Overall, I do not have doubt that alternative exons near the 3' end of the transcripts can be studied by 10X data, but the extend is relatively limited. This is confirmed in this study, as only 1353 genes can be quantified at the exon level and only 22 genes were identified to have differential splicing. I think despite the limitation, efforts to mine splicing using 10X data should still be encouraged, given the explosive number of datasets available. However, the discussion of the pro and cons should be balanced (e.g., the 3' bias should be discussed).</p></disp-quote><p>We appreciate the concern that the limited scope of 10X data for splicing discovery was not emphasized enough in the original draft. To clarify, 22 genes were found to have differential compartment-specific splicing, while 129 genes were found to have cell-type-specific differential splicing out of 1,416 genes (page 6). We have added further clarification of the pros and cons of our method to page 9 in the manuscript:</p><p>“Although the SpliZ method enables biological discovery of splicing differences based on droplet-based sequencing data, droplet-based data still presents major challenges for splicing analysis compared to full-length data. In this study, droplet-based sequencing has much lower sequencing coverage than full-length data, resulting in only 1,416 genes with measurable SpliZ values in the first human individual based on 10X data compared to 9,802 genes with measurable SpliZ values in SS2 data. Additionally, current droplet-based data is 3-prime-biased, meaning that some splicing events will never be sequenced by the technology and therefore cannot be analyzed. Despite these challenges, the ubiquity of droplet-based data, its utility for profiling rare cell types, and its unprecedented scale make it a powerful approach to discover regulated splicing.”</p><disp-quote content-type="editor-comment"><p>2. The authors used a new pipeline named SpliZ for analysis (preprint cited as ref. 26), and several variations SpliZsite and SpliZVD were also used. While it is fine to present technical details in separate publications, key features have to be described in the manuscript, which is required for understanding the results. For example, what does SpliZ measure (something similar to PSI I assume), what is SpliZsite (filtered splice sites from STAR alignments?), what is the difference beteen SpliZ and SpliZVD? How dropout and UMIs are handled in the pipeline?</p></disp-quote><p>We appreciate your comment on the lack of enough description for the SpliZ pipeline in the paper. We have now added a new section “Explanation of the SpliZ method” in the Methods to better describe the SpliZ pipeline on page 12. We have also added several more sentences of explanation to the main text at the beginning of page 3:</p><p>“A large negative (resp. positive) SpliZ score for a gene in a cell means that the cell has shorter (resp. longer) introns than average for that gene. In the simplest exon skipping case, the SpliZ reduces to PSI.”</p><p>In summary, we recommend SpliZ for 10x data (as it is less likely to capture multiple independent splicing events in a gene through 10x sequencing) and SpliZVD for the SmartSeq2 as it can systematically provide more strength for analyzing multiple splicing events by projecting them onto lower dimensions. We should also note that SplizSites are obtained as the splice sites that have the largest contribution to the overall variation in the genes found to be significantly regulated by the SpliZ and/or SpliZVD scores (we have now added a description for SPliZsites on page 3). SpliZ and SpliZVD are the only scores that could be used for finding genes with regulated splicing while SpliZsites is used to pinpoint the differential splicing pattern of a gene to one of its splice sites. Data is preprocessed with the SICILIAN pipeline (Dehghannasiri et al., 2021), which handles UMI deduplication (added on page 12).</p><disp-quote content-type="editor-comment"><p>3. Insufficient descriptions were provided in multiple figures.</p></disp-quote><p>We would like to thank you for pointing this out. We have now clarified the figures by adding more description to the legends, text, and to the figures themselves.</p><disp-quote content-type="editor-comment"><p>Figure 2A, circles and squares were not explained (they were explained in Figure 3 below).</p></disp-quote><p>Thank you for pointing out that circles and squares were not explained until figure 3. We have now moved that explanation up into the caption for figure 2.</p><disp-quote content-type="editor-comment"><p>How the dot plots are related to the splice sites shown in the sashimi plots and how the splice sites in the sashimi plots are related to the gene structure schematics (need some guesswork)? What is shown in the boxplot (labelled &quot;Average 3' splice site per cell&quot;, which I assume it SpliZ score, but again is this the fraction of reads that uses the upstream 3' splice site?)</p></disp-quote><p>We appreciate your comment. The dot plots show the fraction of junctional reads for each splice site at the cell type level. The thickness of the sashimi arcs show the fraction of reads when all cells for the associated group of cell types (right below each sashimi) and all datasets (individuals and technologies) are considered together as pseudo bulk. Each vertical line for a splice site in the sashimi arc shows the fraction of reads corresponding to a 10x (circles) or SS2 (squares) dataset from a certain individual. The box plot shows the distribution of the average 3’ splice site for each cell within a cell type by assigning 1 and 2 to the closer and farther 3’ SS, respectively, and then computing their weighted average according to their number of junctional reads. We have now clarified this in the revised caption for Figure 2 and have also modified Figure 2A to make it clearer. We had also provided the SpliZ scores for <italic>MYL6</italic> in Figure 2-—figure supplement 1 (previously Suppl. Figure 1A). We have now mentioned this in the revised caption.</p><disp-quote content-type="editor-comment"><p>Figure 2C, D. Is each dot a single cell or a &quot;meta cell&quot; that averages a certain number of individual cells?</p></disp-quote><p>In Figures 2C, D, each dot shows the SpliZ score (Figure 2C) and the number of spliced reads (Figure 2D) for a single cell and dots are color coded according to the compartment of the cell. We have now clarified this in the revised caption for Figure 2.</p><disp-quote content-type="editor-comment"><p>Figure 3A, how the gene structure schematic relate to the boxplot above is confusing. Also in the gene schematics of the three species further down, it is confusing why two parts of the human gene were highlighted (only the two 3' splice sites near the 3' end are relevant?). Unclear what is shown in the illustrations on the right of the gene schematics.</p></disp-quote><p>We would like to thank you for your comment. To obtain the average 3’ splice site per cell, splice sites are ranked from 1 (closest to the 5’ SS) to 3 (farthest from the 5’ SS) and then the ranks were used to obtain the weighted average 3’ splice site per cell (based on the junctional reads from the 5’ splice site to each 3’ splice site). We put the schematic for protein domains in a separate panel (Figure 3E). The schematic shows how the 3 different protein domains in <italic>MYL6</italic> are organized in each <italic>MYL6</italic> isoform. We have now revised the caption to better describe the figure. We have also modified the figure itself to make it clearer.</p><disp-quote content-type="editor-comment"><p>Figure 3D. How exons labeled 5,6,7 are related to site 1 and 2 in A-C (need guesswork)?</p></disp-quote><p>Thank you for your comment. We have removed exon numbers in Figure 3D and instead have used consistent Refseq isoform IDs throughout different panels in Figure 3.</p><disp-quote content-type="editor-comment"><p>Figure 4. Only read fraction is shown but not SpliZ scores?</p></disp-quote><p>The SpliZ scores for genes <italic>RPS24</italic>, <italic>MYL6</italic>, and <italic>ATP5F1C</italic> are shown in Figure 2-—figure supplement 1-3 (previously Supplementary Figure 1). To visualize the scores across the entire cell population, we utilized the cellxgene software. In these figures, UMAP embedding is based on the gene expression but the cells in the second UMAP for each gene are color-coded according to the SpliZ score of the gene in each cell. We clarified this by adding the following text to the caption of Figure 3:</p><p>“The SpliZ scores (and also gene expression values) for <italic>MYL6</italic> across all 10x cells in human individual 1 are shown in Figure 2-—figure supplement 1”</p><disp-quote content-type="editor-comment"><p>Figure 5A, similar to the question for Figure 2A, how the gene structure schematic relate to the boxplot above is confusing.</p></disp-quote><p>We appreciate your comment. We have now clarified in the revised caption that we have used the same approach as in Figures 2 and 3 to obtain the box plots: ranking 3’ splice sites from 1 to 3 (from the closest one to the farthest one) and then obtaining the weighted average of 3’ splice site for each cell within a cell type. We have now modified Figure 6 (previously Figure 5) and added the following text to its caption to clarify this:</p><p>“The box plot shows the distribution of the average 3’ splice site (obtained as the weighted average of 3’ splice sites when ranked from 1 to 3 from the closest to the farthest according to their fraction of junctional reads) for the cells within a celltype (See Figures 2 and 3 for more explanation of dot and box plots.)”</p><disp-quote content-type="editor-comment"><p>Figure 6C. It does not seem to be the evidence of a single exon to distinguish two subpopulations of monocytes are convincing. How do we know whether the two subpopulations reflect certain technical issues (potentially similar to the controversial &quot;bimodal splicing&quot; proposed in previous publications)?</p></disp-quote><p>Thank you for bringing up the possibility that the distributions we’re seeing in the subpopulation analysis could be due to a technical artifact such as bimodal splicing (Buen Abad Najar et al., 2020). We agree that this is a big concern, and we appreciate the opportunity to provide more evidence for the reliability of the subclusters. We have now revised the first paragraph of page 8 and also added the following analysis into the Methods section on page 16 of the paper:</p><p>“To test whether the subpopulations of <italic>SAT1</italic> were the result of a “bimodal splicing” artifact as reported in (Buen Abad Najar, Yosef, and Lareau 2020), we performed the following analysis. Subsetting to only blood classical monocytes in human individual 1, we calculated the fraction of junctional reads <italic>p</italic> aligning to the 5’ splice site 23785328 in <italic>SAT1</italic> that partner with the 3’ splice site 23783883 rather than 23784403. We found <italic>p</italic> = 83/102 = 0.814 (the probability was 0.893 considering only individual 2). We then subset to only cells with exactly 2 reads mapping to 5’ splice site 23785328 in <italic>SAT1</italic>, resulting in 16 cells. Out of these 16 cells, all had either both reads mapping to 3’ splice site 23783883 or both reads mapping to 3’ splice site 23784403. We calculated the probability of zero cells having one read mapping to each splice site under the null hypothesis as follows: (1 – binom.pmf(1,2,0.814))<sup>16</sup> = 0.00312 (binomial exact test). Because processing of 10X data includes a UMI deduplication step through SICILIAN (Dehghannasiri et al., 2021), these duplicates are not PCR duplicates.”</p><disp-quote content-type="editor-comment"><p>Similar confusion in Figure 7 as in Figures 2 and 5.</p></disp-quote><p>Thank you for pointing this out. The gray dashed lines between the gene structure and the dot plot show the corresponding splice site for each column of the dot plot and regarding box plots we similarly obtained the average 3’ splice site by ranking them from 1 to 4 and computing the weighted average of their ranks according to the number of junctional reads to each 3’ splice site. We have now better explained this in the revised caption for Figure 7 (now Figure 8).</p><disp-quote content-type="editor-comment"><p>4. I thought some global evaluation on the reliability of 10X results using independent datasets will be important (e.g., correlation of differential splicing in comparison with results from bulk RNA-seq and/or SMART-seq data). Some of the results presented in Supplementary Figures(e.g., Figure S6) can probably presented as the main figure.</p></disp-quote><p>We appreciate the interest in seeing global comparisons of the reliability of 10X results. We did not attempt a comparison of 10X with bulk data because bulk data represents a mixture of single cells with unknown proportions, which would confound the comparison between scRNA-Seq and bulk data. Instead, we validated two of our main discoveries, patterns in <italic>MYL6</italic> and <italic>RPS24</italic>, using FISH, a completely orthogonal method of validation. We have also already included global comparison with SS2 as you pointed out in Supplementary Figure 6, which we have now included as a main figure (Figure 5).</p></body></sub-article></article>