<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">51254</article-id><article-id pub-id-type="doi">10.7554/eLife.51254</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group></article-categories><title-group><article-title>Gene regulatory network reconstruction using single-cell RNA sequencing of barcoded genotypes in diverse environments</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes" id="author-124327"><name><surname>Jackson</surname><given-names>Christopher A</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8769-2710</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes" id="author-140061"><name><surname>Castro</surname><given-names>Dayanne M</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-140060"><name><surname>Saldi</surname><given-names>Giuseppe-Antonio</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes" id="author-112725"><name><surname>Bonneau</surname><given-names>Richard</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4354-7906</contrib-id><email>bonneau@nyu.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="equal-contrib2">‡</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund7"/><xref ref-type="other" rid="fund8"/><xref ref-type="other" rid="fund4"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes" id="author-3100"><name><surname>Gresham</surname><given-names>David</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4028-0364</contrib-id><email>dgresham@nyu.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib2">‡</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution content-type="dept">Center For Genomics and Systems Biology</institution><institution>New York University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution content-type="dept">Department of Biology</institution><institution>New York University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution content-type="dept">Courant Institute of Mathematical Sciences, Computer Science Department</institution><institution>New York University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution content-type="dept">Center For Data Science</institution><institution>New York University</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution content-type="dept">Flatiron Institute, Center for Computational Biology</institution><institution>Simons Foundation</institution><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Barkai</surname><given-names>Naama</given-names></name><role>Reviewing Editor</role><aff><institution>Weizmann Institute of Science</institution><country>Israel</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution>Max Planck Institute for Developmental Biology</institution><country>Germany</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn><fn fn-type="con" id="equal-contrib2"><label>‡</label><p>These authors also contributed equally to this work</p></fn></author-notes><pub-date date-type="publication" publication-format="electronic"><day>27</day><month>01</month><year>2020</year></pub-date><pub-date pub-type="collection"><year>2020</year></pub-date><volume>9</volume><elocation-id>e51254</elocation-id><history><date date-type="received" iso-8601-date="2019-08-21"><day>21</day><month>08</month><year>2019</year></date><date date-type="accepted" iso-8601-date="2020-01-10"><day>10</day><month>01</month><year>2020</year></date></history><permissions><copyright-statement>© 2020, Jackson et al</copyright-statement><copyright-year>2020</copyright-year><copyright-holder>Jackson et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-51254-v3.pdf"/><abstract><p>Understanding how gene expression programs are controlled requires identifying regulatory relationships between transcription factors and target genes. Gene regulatory networks are typically constructed from gene expression data acquired following genetic perturbation or environmental stimulus. Single-cell RNA sequencing (scRNAseq) captures the gene expression state of thousands of individual cells in a single experiment, offering advantages in combinatorial experimental design, large numbers of independent measurements, and accessing the interaction between the cell cycle and environmental responses that is hidden by population-level analysis of gene expression. To leverage these advantages, we developed a method for scRNAseq in budding yeast (<italic>Saccharomyces cerevisiae</italic>). We pooled diverse transcriptionally barcoded gene deletion mutants in 11 different environmental conditions and determined their expression state by sequencing 38,285 individual cells. We benchmarked a framework for learning gene regulatory networks from scRNAseq data that incorporates multitask learning and constructed a global gene regulatory network comprising 12,228 interactions.</p></abstract><abstract abstract-type="executive-summary"><title>eLife digest</title><p>Organisms switch their genes on and off to adapt to changing environments. This takes place thanks to complex networks of regulators that control which genes are actively ‘read’ by the cell to create the RNA molecules that are needed at the time. Piecing together these networks is key to fully understand the inner workings of living organisms, and how to potentially modify or artificially create them.</p><p>Single-cell RNA sequencing is a powerful new tool that can measure which genes are turned on (or ‘expressed’) in an individual cell. Datasets with millions of gene expression profiles for individual cells now exist for organisms such as mice or humans. Yet, it is difficult to use these data to reconstruct networks of regulators; this is partly because scientists are not sure if the computational methods normally used to build these networks also work for single-cell RNA sequencing data.</p><p>One way to check if this is the case is to use the methods on single-cell datasets from organisms where the networks of regulators are already known, and check whether the computational tools help to reach the same conclusion. Unfortunately, the regulatory networks in the organisms for which scientists have a lot of single-cell RNA sequencing data are still poorly known. There are living beings in which the networks are well characterised – such as yeast – but it has been difficult to do single-cell sequencing in them at the scale seen in other organisms.</p><p>Jackson, Castro et al. first adapted a system for single-cell sequencing so that it would work in yeast. This generated a gene expression dataset of over 40,000 yeast cells. They then used a computational method (called the Inferelator) on these data to construct networks of regulators, and the results showed that the method performed well. This allowed Jackson, Castro et al. to start mapping how different networks connect, for example those that control the response to the environment and cell division. This is one of the benefits of single-cell RNA methods: cell division for example is not a process that can be examined at the level of a population, since the cells may all be at different life stages. In the future, the dataset will also be useful to scientists to benchmark a variety of single cell computational tools.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>single cell RNA sequencing</kwd><kwd>gene regulatory networks</kwd><kwd>transcription factors</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>S. cerevisiae</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000062</institution-id><institution>National Institute of Diabetes and Digestive and Kidney Diseases</institution></institution-wrap></funding-source><award-id>R01DK103358</award-id><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000057</institution-id><institution>National Institute of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R01GM107466</award-id><principal-award-recipient><name><surname>Gresham</surname><given-names>David</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>MCB1818234</award-id><principal-award-recipient><name><surname>Gresham</surname><given-names>David</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100009633</institution-id><institution>Eunice Kennedy Shriver National Institute of Child Health and Human Development</institution></institution-wrap></funding-source><award-id>R01HD096770</award-id><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>IOS1546218</award-id><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000054</institution-id><institution>National Cancer Institute</institution></institution-wrap></funding-source><award-id>R01CA229235</award-id><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution>Flatiron Institute</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund8"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000893</institution-id><institution>Simons Foundation</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Bonneau</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Single cell expression data can be used to determine how regulatory transcription factors and target genes are connected, and is especially useful when studying transcription factors controlling heterogeneous cell states.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Elucidating relationships between genes, and the products they encode, remains one of the central challenges in experimental and computational biology. A gene regulatory network (GRN) is a directed graph in which regulators of gene expression are connected to target gene nodes by interaction edges. Regulators of gene expression include transcription factors (TF) which can act as activators and repressors, RNA binding proteins, and regulatory RNAs. Identifying regulatory relationships between transcriptional regulators and their targets is essential for understanding biological phenomena ranging from cell growth and division to cell differentiation and development (<xref ref-type="bibr" rid="bib27">Davidson, 2012</xref>). Reconstruction of GRNs is required to understand how gene expression dysregulation contributes to cancer and complex heritable diseases (<xref ref-type="bibr" rid="bib8">Barabási et al., 2011</xref>; <xref ref-type="bibr" rid="bib53">Hu et al., 2016</xref>).</p><p>Genome-scale methods provide an efficient means of identifying gene regulatory relationships. Efforts of the past two decades have resulted in the development of a variety of experimental and computational methods that leverage advances in technology and machine learning for constructing GRNs. Previously, we developed a method for inferring transcriptional regulatory networks based on regression with regularization that we have called the Inferelator (<xref ref-type="bibr" rid="bib10">Bonneau et al., 2006</xref>; <xref ref-type="bibr" rid="bib23">Ciofani et al., 2012</xref>). This method takes as inputs gene expression data and sources of prior information, and outputs regulatory relationships between transcription factors and their target genes that explain the observed gene expression levels. Subsequent work has enhanced this approach by selecting regulators for each gene more effectively (<xref ref-type="bibr" rid="bib77">Madar et al., 2010</xref>), incorporating orthogonal data types that can be used to generate constraints on network structure (<xref ref-type="bibr" rid="bib46">Greenfield et al., 2013</xref>), and explicitly estimating latent biophysical parameters including transcription factor activity (<xref ref-type="bibr" rid="bib5">Arrieta-Ortiz et al., 2015</xref>; <xref ref-type="bibr" rid="bib35">Fu et al., 2011</xref>) and mRNA decay rates (<xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>). We have successfully applied this approach to construct GRNs from gene expression data acquired from variation across time, conditions, and genotypes in microbes (<xref ref-type="bibr" rid="bib5">Arrieta-Ortiz et al., 2015</xref>; <xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>), plants (<xref ref-type="bibr" rid="bib124">Wilkins et al., 2016</xref>), and mammalian cells (<xref ref-type="bibr" rid="bib23">Ciofani et al., 2012</xref>; <xref ref-type="bibr" rid="bib84">Miraldi et al., 2019</xref>).</p><p>Recently, single-cell RNA sequencing (scRNAseq) has exploded in popularity with the development of droplet systems for rapid encapsulation and labeling of thousands of cells in parallel. The DROP-seq system (<xref ref-type="bibr" rid="bib76">Macosko et al., 2015</xref>) based on bead capture, and the inDrop (<xref ref-type="bibr" rid="bib127">Zilionis et al., 2017</xref>) and 10x Genomics (<xref ref-type="bibr" rid="bib126">Zheng et al., 2017</xref>) systems based on hydrogel beads, provide a facile means of generating RNA sequencing data for tens of thousands of individual cells. Although scRNAseq has primarily been used for defining cell types and states, this technology holds great potential for efficient construction of GRNs (<xref ref-type="bibr" rid="bib54">Hwang et al., 2018</xref>). By combining genetic perturbation of transcriptional regulators using CRISPR/Cas9 with scRNAseq, mixtures of genetic perturbations can be assayed in a single reaction (<xref ref-type="bibr" rid="bib1">Adamson et al., 2016</xref>; <xref ref-type="bibr" rid="bib31">Dixit et al., 2016</xref>; <xref ref-type="bibr" rid="bib58">Jaitin et al., 2016</xref>). This approach, known as Perturb-seq, presents a new opportunity for efficiently inferring GRNs from thousands of individual cells in which different regulators have been disrupted. There are considerable advantages in both scalability and detection of intra-sample heterogeneity with Perturb-seq, but quantifying the the effectiveness of CRISPR/Cas9 targeting in individual cells and distinguishing gene expression variability from noise inherent to mRNA undersampling in scRNAseq (<xref ref-type="bibr" rid="bib12">Brennecke et al., 2013</xref>; <xref ref-type="bibr" rid="bib47">Grün et al., 2014</xref>) present technical challenges. Computational methods to take advantage of scRNAseq data for inferring GRNs are under active development (<xref ref-type="bibr" rid="bib2">Aibar et al., 2017</xref>; <xref ref-type="bibr" rid="bib17">Chan et al., 2017</xref>; <xref ref-type="bibr" rid="bib114">van Dijk et al., 2018</xref>). However, benchmarking these methods is difficult; in the absence of a known GRN, model performance is often estimated using simulated data (<xref ref-type="bibr" rid="bib20">Chen and Mar, 2018</xref>), and issues regarding the appropriate experimental and computational approaches to GRN construction from scRNAseq data remain unresolved.</p><p>The budding yeast <italic>Saccharomyces cerevisiae</italic> is ideally suited to constructing GRNs from experimental data and benchmarking computational methods. Decades of work have provided a plethora of transcriptional regulatory data comprising functional and biochemical information (<xref ref-type="bibr" rid="bib28">de Boer and Hughes, 2012</xref>; <xref ref-type="bibr" rid="bib111">Teixeira et al., 2018</xref>). As a result, yeast is well suited to constructing GRNs using methods that leverage the rich available information and for assessing the performance of those methods by comparison to experimentally validated interactions (<xref ref-type="bibr" rid="bib75">Ma et al., 2014</xref>; <xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>). Budding yeast presents several technical challenges for single cell analysis, and as a result scRNAseq methods for budding yeast reported to date (<xref ref-type="bibr" rid="bib38">Gasch et al., 2017</xref>; <xref ref-type="bibr" rid="bib87">Nadal-Ribelles et al., 2019</xref>) yield far fewer individual cells (~10<sup>2</sup>) than are now routinely generated for mammalian studies (&gt;10<sup>4</sup>). The limitations of existing scRNAseq methods for budding yeast cells limits our ability to investigate eukaryotic cell biology as many signaling and regulatory pathways are highly conserved in yeast (<xref ref-type="bibr" rid="bib14">Carmona-Gutierrez et al., 2010</xref>; <xref ref-type="bibr" rid="bib44">Gray et al., 2004</xref>), including the Ras/protein kinase A (PKA), AMP Kinase (AMPK) and target of rapamycin (TOR) pathways (<xref ref-type="bibr" rid="bib43">González and Hall, 2017</xref>; <xref ref-type="bibr" rid="bib72">Loewith and Hall, 2011</xref>). However, recent work has successfully established single cell sequencing in the fission yeast <italic>Schizosaccharomyces pombe</italic> (<xref ref-type="bibr" rid="bib98">Saint et al., 2019</xref>).</p><p>In budding yeast, the TOR complex 1 (TORC1 or mTORC1 in human) coordinates the transcriptional response to changes in nitrogen sources (<xref ref-type="bibr" rid="bib42">Godard et al., 2007</xref>; <xref ref-type="bibr" rid="bib96">Rødkaer and Faergeman, 2014</xref>). Controlling this response are four major TF groups, which are regulated by diverse post-transcriptional processes. The Nitrogen Catabolite Repression (NCR) pathway, which is regulated principally by TORC1, consists of the TFs <italic>GAT1</italic>, <italic>GLN3</italic>, <italic>DAL80</italic>, and <italic>GZF3</italic> (<xref ref-type="bibr" rid="bib52">Hofman-Bang, 1999</xref>), and is responsible for suppressing the utilization of non-preferred nitrogen sources when preferred nitrogen sources are available. Gat1 and Gln3 are localized to the cytoplasm until activation results in relocalization to the nucleus (<xref ref-type="bibr" rid="bib25">Cox et al., 2000</xref>), where they then compete with Dal80 and Gzf3 for DNA binding motifs (<xref ref-type="bibr" rid="bib39">Georis et al., 2009</xref>). The General Amino Acid Control (GAAC) pathway consists of the TF <italic>GCN4</italic> (<xref ref-type="bibr" rid="bib51">Hinnebusch, 2005</xref>), and is responsible for activating the response to amino acid starvation, as detected by increases in uncharged tRNA levels. Gcn4 activity is translationally controlled by ribosomal pausing at upstream open reading frames in the 5’ untranslated region (<xref ref-type="bibr" rid="bib86">Mueller and Hinnebusch, 1986</xref>). The retrograde pathway, consisting of the TF heterodimer <italic>RTG1</italic> and <italic>RTG3</italic>, is responsible for altering expression of metabolic and biosynthetic genes in response to mitochondrial stress (<xref ref-type="bibr" rid="bib60">Jia et al., 1997</xref>; <xref ref-type="bibr" rid="bib70">Liao and Butow, 1993</xref>) or environmental stress (<xref ref-type="bibr" rid="bib97">Ruiz-Roig et al., 2012</xref>). The Rtg1/Rtg3 complex is localized to the cytoplasm until activation, upon which they translocate into the nucleus (<xref ref-type="bibr" rid="bib63">Komeili et al., 2000</xref>). The Ssy1-Ptr3-Ssy5-sensing (SPS) pathway (<xref ref-type="bibr" rid="bib71">Ljungdahl, 2009</xref>), consists of the TFs <italic>STP1</italic> and <italic>STP2</italic>, and is responsible for altering transporter expression (<xref ref-type="bibr" rid="bib30">Didion et al., 1998</xref>; <xref ref-type="bibr" rid="bib55">Iraqui et al., 1999</xref>) in response to changes in extracellular environment. Stp1 and Stp2 are anchored to the plasma membrane until the SPS sensor triggers proteolytic cleavage of their anchoring domain and releases them for nuclear import (<xref ref-type="bibr" rid="bib4">Andréasson and Ljungdahl, 2002</xref>).</p><p>Construction of GRNs based on the transcription factors in these pathways has had mixed success; the high redundancy of the NCR pathway has proven challenging to deconvolute (<xref ref-type="bibr" rid="bib82">Milias-Argeitis et al., 2016</xref>). The GAAC pathway is more straightforward, although separating direct and indirect regulation remains difficult, even with high-quality experimental data (<xref ref-type="bibr" rid="bib85">Mittal et al., 2017</xref>). As a result, a comprehensive GRN for nitrogen metabolism has remained elusive, despite successes in identifying genes that respond to changes in environmental nitrogen (<xref ref-type="bibr" rid="bib3">Airoldi et al., 2016</xref>) and identification of post-transcriptional control mechanisms that underlie these changes (<xref ref-type="bibr" rid="bib83">Miller et al., 2018</xref>).</p><p>Many signalling regulators involved in environmental response interact with cell cycle programs (<xref ref-type="bibr" rid="bib61">Johnston et al., 1977</xref>; <xref ref-type="bibr" rid="bib108">Talarek et al., 2017</xref>), including the TOR pathway (<xref ref-type="bibr" rid="bib128">Zinzalla et al., 2007</xref>); however, how regulation of the mitotic cell cycle and environmentally responsive gene expression is coordinated is unknown. The regulation of nitrogen responsive gene expression in yeast is well-suited to the development of generalizable methods as the degree of TF redundancy, post-transcriptional regulation of TF activity, which precludes straightforward relationships between TF abundance and target expression, and multifactorial impact on gene expression, including intrinsic and extrinsic processes and stimuli, provide a tractable model system for addressing these challenges in higher eukaryotes.</p><p>Here, we have developed a method for scRNAseq in budding yeast using Chromium droplet-based single cell encapsulation (10x Genomics). We engineered TF gene deletions by precisely excising the entire TF open reading frame and introducing a unique transcriptional barcode that enables multiplexed analysis of genotypes using scRNAseq. We pooled 72 different strains, corresponding to 12 different genotypes, and determined their gene expression profiles in 11 conditions using scRNAseq analysis of 38,000 cells. We show that our method enables identification of cells from complex mixtures of genotypes in asynchronous cultures that correspond to specific mutants, and to specific stages of the cell cycle. Identification of mutants can be used to identify differentially expressed genes between genotypes providing an efficient means of multiplexed gene expression analysis. We used scRNAseq data in yeast to benchmark computational aspects of GRN reconstruction, and show that multi-task learning integrates information across environmental conditions without requiring complex normalization, resulting in improved GRN reconstruction. We find that imputation of missing data does not improve GRN reconstruction and can lead to prediction of spurious interactions. Using scRNAseq data, we constructed a global GRN for budding yeast comprising 12,228 regulatory interactions. We discover novel regulatory relationships, including previously unknown connections between regulators of cell cycle gene expression and nitrogen responsive gene expression. Our study provides a generalizable framework for GRN reconstruction from scRNAseq, a rich data set that will enable benchmarking of future computational methods, and establishes the use of droplet-based scRNAseq analysis of multiplexed genotypes in yeast.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Engineering a library of Prototrophic, Transcriptionally-Barcoded Gene Deletion Strains</title><p>The yeast gene knockout collection (<xref ref-type="bibr" rid="bib40">Giaever et al., 2002</xref>) facilitates pooled analysis of mutants using unique DNA barcode sequences that identify each gene deletion strain, but these barcodes are only present at the DNA level, precluding their use with scRNAseq. Therefore, we constructed an array of prototrophic, diploid yeast strains with homozygous deletions of TFs that control distinct regulatory modules: 1) NCR, 2) GAAC, 3) SPS-sensing, and 4) the retrograde pathway that coordinately control nitrogen-related gene expression in yeast. We engineered eleven different TF knockout genotypes, using six independently constructed biological replicates for each genotype. In addition, we constructed six biological replicates of the wild-type control in which we deleted the neutral HO locus. Genes were deleted using a modified kanMX cassette such that each of the 72 strains contains a unique transcriptional barcode in the 3’ untranslated region (UTR) of the G418 resistance gene, that can be recovered by RNA sequencing (<xref ref-type="fig" rid="fig1">Figure 1</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1A</xref>). Homozygous diploids were constructed by mating to a strain containing the same TF knockout marked with a nourseothricin drug resistance cassette. On rich media plates, the 72 strains have an approximately wild-type growth; under nutritional stress, some TF knockouts exhibit growth advantages or disadvantages (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1B</xref>).</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Single-Cell RNA-Seq Experimental Workflow in <italic>Saccharomyces Cerevisiae.</italic></title><p>Schematic workflows for: (<bold>A</bold>) Growth of a transcriptionally-barcoded pool of 11 nitrogen metabolism transcription factor (TF) knockout strains and a wild-type control strain each analyzed with six biological replicates (<bold>B</bold>) Synthesis in microfluidic droplets of single-cell cDNA with a cell-specific index sequence (IDX) attached to the oligo-dT primer, and a common template switch oligo (TSO). cDNA is processed for whole-transcriptome libraries, to quantify gene expression. In parallel, PCR products are amplified containing the genotype-specific transcriptional barcode (BC) encoded on the Kan<sup>R</sup> antibiotic resistance marker mRNA, to identify cell genotype. Expression DNA libraries and PCR products are separately indexed for multiplexed sequencing (<bold>C</bold>) Processing of single-cell sequencing data using Unique Molecular Identifiers (UMI) into a count matrix which is used to learn a gene regulatory network using multi-task network inference from several different growth conditions.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig1-v3.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Strain Construction Workflow and Validation.</title><p>(<bold>A</bold>) Construction of KanMX[BC] deletion cassettes containing degenerate barcodes in the 3’ untranslated region, followed by transformation into yeast to create a transcription factor knockout array (<bold>B</bold>) Spotting of the deletion array onto YPD (control) and media containing different nitrogen sources plates. Each column is a separate TF deletion, and each spot is a uniquely barcoded biological replicate. There are six biological replicates for each of 12 genotypes, for a total of 72 unique strains in the array.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig1-figsupp1-v3.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Single-Cell RNA sequencing of pooled libraries in diverse growth conditions</title><p>ScRNAseq in yeast presents several challenges: cells are small (40–90 µm<sup>3</sup>), enclosed in a polysaccharide-rich cell wall, and contain fewer mRNAs per cell (40 k-60k) than higher eukaryotes. We developed and validated a protocol using the droplet-based 10x genomics chromium platform, and it used it to perform scRNAseq of the pool of TF knockouts in eleven growth conditions that provide a range of metabolic challenges (<xref ref-type="table" rid="table1">Table 1</xref>). In addition to variable nitrogen sources in minimal media with excess [MM] and limiting [NLIM-NH<sub>4</sub>, NLIM-GLN, NLIM-PRO, NLIM-UREA] nitrogen, some conditions result in fermentative metabolism of glucose in rich media [YPD], and inhibition of the TOR signaling pathway in rich media by the small molecule rapamycin [RAPA]. We also studied conditions that require respiratory metabolism of ethanol in rich [YPEtOH] and minimal media [MMEtOH], and in rich media after sugars had been fully metabolized to ethanol and cells have undergone the diauxic shift [DIAUXY]. We tested two different starvation conditions, carbon [CSTARVE] and nitrogen starvation [NSTARVE]; however, the latter condition did not pass quality control during single-cell transcriptome library preparation and was discarded.</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Environmental Growth Conditions.</title><p>Environmental growth conditions are listed with their respective nitrogen and carbon sources. Yeast Extract + Peptone (YP) is a rich, complex nitrogen source. YP + Dextrose [YPD] is standard yeast rich media. Minimal media contains a standard base of trace metals, vitamins, and salts. All cultures were harvested 4 hr after inoculation, except for the culture harvested after the diauxic shift [DIAUXY], which was harvested 10 hr after inoculation. Rapamycin was added to YPD in the [RAPA] culture 30 min prior to harvest. Specific media formulations are listed in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>-Supplemental Table 4.</p></caption><table frame="hsides" rules="groups"><thead><tr><th>Growth condition</th><th>Abbrv.</th><th>Nitrogen source</th><th>Carbon source</th></tr></thead><tbody><tr><td>Yeast Extract, Peptone, Glucose</td><td>YPD</td><td>YP</td><td>D-Glucose</td></tr><tr><td>YPD (Harvested after Post-Diauxic Shift)</td><td>DIAUXY</td><td>YP</td><td>D-Glucose</td></tr><tr><td>YPD + 200 ng/mL Rapamycin</td><td>RAPA</td><td>YP</td><td>D-Glucose</td></tr><tr><td>Yeast Extract, Peptone, Ethanol</td><td>YPEtOH</td><td>YP</td><td>Ethanol</td></tr><tr><td>Minimal Media (Glucose)</td><td>MMD</td><td>20 mM (NH<sub>4</sub>)<sub>2</sub>SO<sub>4</sub></td><td>D-Glucose</td></tr><tr><td>Minimal Media (Ethanol)</td><td>MMEtOH</td><td>20 mM (NH<sub>4</sub>)<sub>2</sub>SO<sub>4</sub></td><td>Ethanol</td></tr><tr><td>Nitrogen Limited Minimal Media (with Glutamine)</td><td>NLIM-GLN</td><td>0.8 mM L-Glutamine</td><td>D-Glucose</td></tr><tr><td>Nitrogen Limited Minimal Media (with Proline)</td><td>NLIM-PRO</td><td>0.8 mM L-Proline</td><td>D-Glucose</td></tr><tr><td>Nitrogen Limited Minimal Media (with NH<sub>4</sub>)</td><td>NLIM-NH<sub>4</sub></td><td>0.8 mM (NH<sub>4</sub>)<sub>2</sub>SO<sub>4</sub></td><td>D-Glucose</td></tr><tr><td>Nitrogen Limited Minimal Media (with Urea)</td><td>NLIM-UREA</td><td>0.8 mM Urea</td><td>D-Glucose</td></tr><tr><td>Carbon Starvation</td><td>CSTARVE</td><td>1 mM (NH<sub>4</sub>)<sub>2</sub>SO<sub>4</sub></td><td>None</td></tr></tbody></table></table-wrap><p>Cells from the eleven different conditions were sequenced and processed using cellranger (10x genomics) and our custom analysis pipeline (fastqToMat0), yielding a digital expression matrix (<xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>) in which each cell is annotated with the environmental growth condition and genotype. Genotype-specific barcodes facilitate identification and removal of droplets that have multiple cells (doublets) by determining cell IDs that have more than one annotated genotype. Using our pool of 72 strains, we detect and remove 98.5% of doublets. PCR artifacts and duplicates are removed using Unique Molecular Identifiers (UMIs) (<xref ref-type="bibr" rid="bib62">Kivioja et al., 2012</xref>) to quantify gene expression as unique transcript reads (counts). Following sequence processing, quality control, removal of doublets, and assigning metadata, we recovered 83,703,440 transcript counts from a total of 38,225 individual cells.</p><p>To initially assess the quality of our data, we examined the expression of genes that are characteristic of different metabolic states. Consistent with our expectations, the core fermentative (anaerobic) genes <italic>PDC1</italic> and <italic>ENO2</italic> are expressed in cells in fermenting culture conditions only, and the core respirative (aerobic) gene <italic>ADH2</italic> is expressed in cells in respiring culture conditions (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). The number of cells recovered varies by over an order of magnitude between conditions; stressful conditions of low nitrogen have lower cell yields overall. The yeast stress response is linked to increased resistance to zymolyase digestion (<xref ref-type="bibr" rid="bib88">Nagarajan et al., 2014</xref>), which may be reflected in decreased cell yield during single-cell sequencing. Each of the 72 strains is found in each of the 11 conditions, although the number of each strain and genotype varies by environmental condition (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A</xref>), and some strains are disproportionately affected. However, the number of transcripts per cell is generally equivalent between strains even when they differ in representation within libraries (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1B</xref>). By contrast, we find that total transcript counts per cell are highly linked to environmental growth conditions (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1C</xref>), which is consistent with decreased total transcriptome pool size in suboptimal conditions (<xref ref-type="bibr" rid="bib6">Athanasiadou et al., 2019</xref>). For cells growing in rich medium (YPD) we recover a median of 2250 unique transcripts per cell, from a median of 695 distinct genes, indicating a capture rate of approximately 3–5% of total transcripts from each cell. The strain genotype does not strongly influence transcript counts per cell (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1D</xref>). There is a high correlation between single-cell expression data and bulk RNA expression data (spearman correlation 0.941) for wild-type cells grown in YPD (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>) indicating that the effect of technical bias caused by single-cell processing is minimal. We also find good correlation to other published single-cell yeast data sets, and a comparable published bulk RNAseq experiment,.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Gene expression of single Yeast Cells Cluster Based on Environmental Growth Condition.</title><p>(<bold>A</bold>) Normalized density histograms of raw UMI counts of the core glycolytic genes <italic>ENO2</italic> and <italic>PDC1</italic>, and the alcohol respiration gene <italic>ADH2</italic> in each environmental growth condition. Mean UMI count for each of the 12 different strain genotypes within each growth condition are plotted as dots on the X axis. (<bold>B–C</bold>) Uniform Manifold Approximation and Projection (UMAP) projection of log-transformed and batch-normalized scRNAseq data. Axes are dimensionless variables V1 and V2 with arbitrary units, here omitted. Individual cells are colored by environmental growth condition (<bold>B</bold>) or by strain genotype (<bold>C</bold>). Growth conditions are abbreviated as in <xref ref-type="table" rid="table1">Table 1</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig2-v3.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Quality Control of Single-Cell RNA Sequencing Data.</title><p>(<bold>A</bold>) The number of cells that pass all quality control and preprocessing filters for each growth condition. Each genotype has multiple independently barcoded biological replicates that are plotted separately within each condition. The mean number of cells for each genotype within a condition is plotted as a horizontal line. (<bold>B</bold>) The mean count of unique transcripts, determined by UMI, for each growth condition. Biological replicates are plotted separately for each genotype within each condition. The mean number of transcripts per genotype is plotted as a horizontal line. (<bold>C</bold>) The distribution of unique transcript count per cell across all growth conditions. (<bold>D</bold>) The distribution of unique transcript counts per cell across all genotypes.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig2-figsupp1-v3.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Single-Cell RNA Expression Comparison.</title><p>(<bold>A</bold>) Pairwise ranked gene expression plots of cells grown in YPD to mid-log phase. Data sets are bulk counts [TRIZOL] from FY4/FY5 diploids (n = 6), 10x-based 3’ end-labeled single-cell counts [10x-scRNA] from FY4/FY5 diploids (n = 976), 5’ end-labeled single-cell counts [yscRNA (2019)] from BY4741 haploids (n = 127), SCnorm-calculated normalized counts <xref ref-type="bibr" rid="bib38">Gasch et al. (2017)</xref> from BY4741 haploids (n = 163), and bulk transcripts per kilobase million [GSE135430] from BY4741 haploids (n = 12). (<bold>B</bold>) Correlation heatmap showing spearman’s rank correlation between each sample. (<bold>C</bold>) Counts per sample after sequencing and de-artifacting with UMIs for the bulk experiment on FY4/FY5 wild-type RNA extracted with TRIZOL (n = 6).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig2-figsupp2-v3.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Expression of Categories of Genes in Single Cells.</title><p>UMAP projection of log-transformed and batch-normalized scRNAseq data, colored by: (<bold>A</bold>) Condition, as <xref ref-type="fig" rid="fig2">Figure 2B</xref> (<bold>B</bold>) Total raw count of transcripts (<bold>C</bold>) Percentage of transcripts which are ribosomal genes (<bold>D</bold>) Percentage of transcripts which are ribosomal biogenesis genes (<bold>E</bold>) Percentage of transcripts which are induced environmental stress response (iESR) genes (<bold>F</bold>) Percentage of transcripts which are mitochondrially-encoded genes.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig2-figsupp3-v3.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Measures of Gene Variance in Each Condition.</title><p>(<bold>A</bold>) Coefficient of variation (standard deviation over mean) plotted against mean of each gene for each growth condition. Both axes are plotted on a log scale. (<bold>B</bold>) The mean pearson residuals (residuals over expected standard deviation) of a regularized negative binomial regression model calculated for each gene by the R package sctransform. The mean gene expression is plotted on a log scale.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig2-figsupp4-v3.tif"/></fig></fig-group><p>Mapping the digital expression matrix into two-dimensional space with a Uniform Manifold Approximation and Projection [UMAP] results in clear separation of individual cells into groups based on environmental condition (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Cells from different minimal media or nitrogen-limited growth conditions localize near each other, and cells grown in different rich nitrogen sources are clearly separate from each other. Within environmentally-determined grouping there appears to be no strong ordering by genotype (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). These clusters are not driven by sequencing depth (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3B</xref>), although there are some stress conditions which have subpopulations which are downregulated for ribosomal genes and upregulated for induced environmental stress response (iESR) genes (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3C–F</xref>). The increased relative abundance of ribosomal related gene expression in rich media conditions is consistent with previously-observed correlation of ribosomal gene expression and cellular growth rate (<xref ref-type="bibr" rid="bib11">Brauer et al., 2008</xref>). Some measures of variance per-gene differ in different growth conditions (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). Interactive figures are provided (<ext-link ext-link-type="uri" xlink:href="http://shiny.bio.nyu.edu/YeastSingleCell2019/">http://shiny.bio.nyu.edu/YeastSingleCell2019/</ext-link>) facilitating exploration of expression levels for all genes.</p></sec><sec id="s2-3"><title>The mitotic cell cycle underlies heterogeneity in single cell gene expression</title><p>To identify sources of gene expression differences between cells within environments, we clustered single cells within each environmental condition separately by constructing a Shared Nearest Neighbor graph (<xref ref-type="bibr" rid="bib125">Xu and Su, 2015</xref>) and clustering using the Louvain method (<xref ref-type="bibr" rid="bib9">Blondel et al., 2008</xref>). Genes with known roles in mitotic cell cycle are highly represented among the most differentially expressed genes between clusters (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>). Overlaying the expression of three of these genes (<italic>PIR1</italic>, <italic>DSE2</italic>, and <italic>HTB1</italic>/<italic>HTB2</italic>) on UMAP plots illustrates cell cycle effects on single cell gene expression (<xref ref-type="fig" rid="fig3">Figure 3A</xref> and <xref ref-type="fig" rid="fig3">Figure 3B</xref>). <italic>PIR1</italic> expression, a marker for early G1 (<xref ref-type="bibr" rid="bib106">Spellman et al., 1998</xref>), is diagnostic of a distinct cluster. <italic>DSE2</italic> is expressed only in daughter cells (<xref ref-type="bibr" rid="bib24">Colman-Lerner et al., 2001</xref>), which allows daughter cells in G1 to be distinguished from mother cells in G1. Cells that have high expression of the histone 2B genes, which are upregulated in S-phase (<xref ref-type="bibr" rid="bib33">Eriksson et al., 2012</xref>), are localized together in the UMAP plots (<xref ref-type="fig" rid="fig3">Figure 3B</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Cells Within Conditions Cluster According to Cell Cycle Genes.</title><p>(<bold>A</bold>) Cells from each growth condition were separately normalized and transformed into 2-dimensional space using UMAP. The log-transformed, normalized expression for each cell of (i) the G1-phase specific marker <italic>PIR1</italic>, (ii) the G1-phase daughter-cell specific marker <italic>DSE2</italic>, (iii) the S-phase specific marker histone 2B (<italic>HTB</italic>) is shown; (iv) the genotype and (v) the cluster membership of each cell. (<bold>B</bold>) Summary of clustered single cell expression within the YPD and RAPA growth conditions (i) Proportion of cells from a specific strain genotype within each cluster (ii) The mean log-transformed, normalized expression of the G1- and S-phase marker genes, as well as a hexokinase gene <italic>HXK2</italic> for each cluster (<bold>C</bold>) Schematic of the mitotic cell cycle with expression of <italic>DSE2</italic>, <italic>PIR1</italic>, and <italic>HTB</italic> genes annotated.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig3-v3.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Expression of Important Genes For Clustering.</title><p>(<bold>A</bold>) Gene expression heatmap of genes that are specific for clusters of cells in YPD. Genes name are colored green for G1 phase genes, yellow for S phase genes, blue for G2 phase genes, and purple for M phase genes (<bold>B</bold>) Summary of clustered single cells within growth conditions (i) Proportion of cells within each cluster that consist of a specific strain genotype (ii) The mean within each cluster of log-transformed, normalized expression of the G1- and S-phase marker genes, as well as a hexokinase gene <italic>HXK2</italic> that is unlikely to be responding to cell cycle.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig3-figsupp1-v3.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Some Conditions Have Stress Response Clusters Cells from each growth condition were separately normalized and transformed into 2-dimensional space using UMAP.</title><p>These single cells are colored by (<bold>A</bold>) Total raw UMI count (<bold>B</bold>) Percentage of transcripts which are ribosomal genes (<bold>C</bold>) Percentage of transcripts which are induced environmental stress response (iESR) genes (<bold>D</bold>) Percentage of transcripts which annotated as G1 phase genes (<bold>E</bold>) Percentage of genes which are annotated as S phase genes (<bold>F</bold>) Percentage of genes which are annotated as G2 phase genes (<bold>G</bold>) Percentage of genes which are annotated as M phase genes.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig3-figsupp2-v3.tif"/></fig></fig-group><p>For each cluster of cells within a growth condition we plotted the proportion of cells belonging to each TF deletion genotype, and the mean expression of several cell cycle genes (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Some clusters predominantly contain cells from a single TF deletion genotype; for example, cells deleted for <italic>GLN3</italic> (<italic>gln3Δ</italic>) form a separate cluster in YPD and RAPA conditions, as do cells deleted for one of the RTG heterodimer components (<italic>rtg1Δ</italic> and <italic>rtg3Δ</italic>). However, differences in expression due to genotype do not appear to be a primary source of expression differences within conditions, as most clusters show a uniform distribution of genotypes (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>). Similarly, we do not find that differences in expression of metabolic genes underlie overall differences in expression (e.g. <italic>HXK2</italic>) suggesting that the yeast metabolic cycle (<xref ref-type="bibr" rid="bib103">Silverman et al., 2010</xref>; <xref ref-type="bibr" rid="bib112">Tu et al., 2005</xref>) is not readily identifiable in single cells using scRNAseq. Three of the high-stress growth conditions (NLIM-GLN, NLIM-PRO, and MMEtOH) have clusters that are separate from the majority of the cells analyzed in those conditions. We find that these clusters have higher levels of stress response genes and lower levels of ribosomal genes than other cells in these conditions (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2B–C</xref>) These clusters may reflect cells undergoing early entry into quiescence and provide evidence for a heterogeneous response to stressful conditions.</p></sec><sec id="s2-4"><title>Deletion of Transcription Factors causes gene expression changes that differ between growth conditions</title><p>To assess our ability to determine differential gene expression between TF knockout strains, we examine the expression of genes known to respond to nitrogen signalling. <italic>GAP1</italic> (General Amino acid Permease) is a transporter responsible for importing amino acids under conditions of nitrogen limitation; <italic>GAP1</italic> expression is regulated by the NCR activators <italic>GAT1</italic> and <italic>GLN3</italic>, the NCR repressors <italic>GZF3</italic> and <italic>DAL80</italic> (<xref ref-type="bibr" rid="bib107">Stanbrough and Magasanik, 1995</xref>), and potentially <italic>GCN4</italic> (<xref ref-type="bibr" rid="bib89">Natarajan et al., 2001</xref>). We identify differing degrees of dysregulation of <italic>GAP1</italic> expression when these TFs are deleted (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). The effect of deleting TFs varies by condition: <italic>GAP1</italic> is not expressed in YPD and its expression increases in nitrogen-limited media and in response to rapamycin. Deletion of <italic>GAT1</italic> results in decreased expression in nitrogen limiting media, but deletion of <italic>GLN3</italic> does not affect <italic>GAP1</italic> expression. By contrast, in the presence of rapamycin deletion of <italic>GLN3</italic> results in reduced <italic>GAP1</italic> expression. Deletion of <italic>GCN4</italic> only impacts <italic>GAP1</italic> expression in the presence of urea. <italic>MEP2</italic> and <italic>GLN1</italic> are also responsive to nitrogen TFs, and are dysregulated when certain TFs are deleted; expression of the glycolytic gene <italic>HXK2</italic> decreases when <italic>GLN3</italic>, <italic>GCN4</italic>, or <italic>RTG1</italic>/<italic>RTG3</italic> are deleted, but only in conditions of nitrogen limitation (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref>). These environmentally dependent impacts of genotype on gene expression demonstrate the importance of exploration of variable conditions for studying genotypic effects on expression.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Impact of Deleting Transcription Factors on Gene Expression.</title><p>(<bold>A</bold>) Violin plots of the log<sub>2</sub> batch-normalized expression of the general amino acid permease gene <italic>GAP1</italic> in YPD, RAPA, ammonium-limited media, and urea-limited media. (<bold>B</bold>) Count of differentially expressed genes in each combination of growth condition and strain genotype. Data were transformed to pseudobulk values by summing all counts for each the six biological replicates for each genotype and then analyzed for differential gene expression using DESeq2 [1.5-fold change; p.adj &lt;0.05]. (<bold>C</bold>) Log<sub>2</sub>(fold change) of genes differentially expressed in TF knockout strains compared to wildtype, when grown in YPD. Asterisks denote statistically significant differences in gene expression [1.5-fold change; p.adj &lt;0.05].</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig4-v3.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Differential Gene Expression Varies by Condition.</title><p>(<bold>A</bold>) Distribution by strain genotype of the log<sub>2</sub> batch-normalized expression of the ammonium permease gene <italic>MEP2</italic> and the glutamine synthetase gene <italic>GLN1</italic> in YPD, RAPA, and ammonium limitation. (<bold>B</bold>) Number of differentially expressed genes identified after bulking between wild-type cells in each growth condition (<bold>C</bold>) Log2(fold change) of genes differentially expressed in TF knockout strains compared to wildtype, when grown in YPD and then treated with rapamycin. Asterisks denote statistically significant differences in gene expression [1.5-fold change; p.adj &lt;0.05].</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig4-figsupp1-v3.tif"/></fig></fig-group><p>A variety of statistical methods have been proposed and benchmarked for testing different expression of scRNAseq data (<xref ref-type="bibr" rid="bib105">Soneson and Robinson, 2018</xref>). Our experimental design allows single-cell measurements to be collapsed into a total count (pseudobulk) measurement by summing counts across all cells that correspond to each of the six individual replicates of each genotype within a condition. When we analyze this data using standard approaches to RNAseq analysis (DESeq2) we detect several genes with significant (adjusted p-value&lt;0.05) differences in expression (fold change &gt;1.5) between wild-type and TF deletion strains (<xref ref-type="fig" rid="fig4">Figure 4B</xref>) that are consistent with known regulatory pathways. There are considerably fewer changes in gene expression as a result of TF deletions compared to the hundreds of genes that change expression between different conditions (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1B</xref>). However, in cells grown in rich media [YPD], we found 96 genes that are differentially expressed in TF deletion strains compared to wildtype (<xref ref-type="fig" rid="fig4">Figure 4C</xref>), and expression of 160 genes are perturbed in TF deletion strains compared to wildtype when exposed to rapamycin [RAPA] (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1C</xref>). Many of these differentially expressed genes are annotated as functioning in amino acid metabolism and biosynthesis.</p></sec><sec id="s2-5"><title>Optimal modeling parameters for network inference from Single-Cell yeast data</title><p>Differential gene expression in a TF knockout strain is not sufficient evidence of a direct regulatory relationship as many significant changes in gene expression upon deleting a TF are indirect, and many direct effects may be subtle. Therefore, we constructed a gene regulatory network using the Inferelator, a regression-based network inference method which is based on three main modeling assumptions. First, we assume that Transcription Factor Activity (TFA) is a latent biophysical parameter that represents the effect of a TF binding to DNA and modulating its transcription activity (<xref ref-type="bibr" rid="bib5">Arrieta-Ortiz et al., 2015</xref>; <xref ref-type="bibr" rid="bib35">Fu et al., 2011</xref>). The TFA values are not directly measured, and instead must be estimated as a relative value based on prior knowledge of a regulatory network of TF and target relationships. This TFA estimation is essential as many TFs are post-transcriptionally regulated, or are expressed at levels that are not reliably detected by scRNAseq (<xref ref-type="bibr" rid="bib34">Filtz et al., 2014</xref>). Second, we assume that expression of a gene can be described as a weighted sum of the activities of TFs (<xref ref-type="bibr" rid="bib10">Bonneau et al., 2006</xref>) using an additive model in which activators and repressors increase or decrease the expression of targets linearly. Finally, we assume that each gene is regulated by a small number of TFs, and that regularization of gene expression models is required to enforce this biologically relevant property of target regulation. <italic>Saccharomyces cerevisiae</italic>, as a preeminent model organism in systems biology, has a well defined set of known interactions that are of considerably higher quality than is available for more complex eukaryotes providing a validated gold standard for testing model performance (<xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>).</p><p>To evaluate the performance of data processing methods and model parameter selections within the Inferelator on scRNAseq data, we perform ten cross-validations using the existing gold standard network. During cross-validation, we infer a GRN using half of the gold standard target genes as priors, then evaluate performance based on recovery of TF-target gene interactions for gold standard interactions that are left out of the priors. We tested preprocessing and prior selection options by inferring networks using gene expression models that are regularized by best subset regression to minimize Bayesian Information Criterion (<xref ref-type="bibr" rid="bib5">Arrieta-Ortiz et al., 2015</xref>; <xref ref-type="bibr" rid="bib46">Greenfield et al., 2013</xref>) and quantified performance in predicting TF-target interactions using the area under the precision-recall curve (AUPR). As negative controls, we employed the same procedure after shuffling priors and after simulating scRNAseq data in which all variance is due to sampling noise. The negative control with shuffled priors establishes a random classifier baseline AUPR of 0.02; the negative control with simulated data establishes a circular recovery baseline AUPR of 0.06 (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). Performance of the Inferelator on our scRNAseq data far exceeds these baselines, with a mean AUPR of 0.20. This performance from our single dataset is comparable to that of a GRN constructed from 2577 experimental observations using bulk gene expression data (<xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Model Performance and Impact of Data Imputation, Prior Selection, and Multitask learning on Network Inference using the Inferelator.</title><p>(<bold>A</bold>) Model performance of Inferelator (TFA-BBSR) network inference after shuffling priors [Neg.Shuffled], on a simulated negative data set [Neg. Data], on the unaltered count matrix [No Imputation], and after imputing missing data from the count matrix using the MAGIC, ScImpute, and VIPER packages. Model performance is shown using area under the precision-recall curve [AUPR], as well as the number of network edges using a precision (&gt;0.5) cutoff, and the number of network edges using a confidence (&gt;0.95) cutoff. Each point plotted in gray is a separate cross-validation analysis, with mean +/- one standard deviation plotted in black (n = 10). (<bold>B</bold>) Median AUPR after cross-validation (n = 10) and resampling to different numbers of cells, for priors extracted from the gold standard [GS], the YEASTRACT database, Bussemaker et al, priors predicted from ATAC-seq data and motif searching, and no prior data. (<bold>C</bold>) AUPR of separate cross-validation network inference using cells from all growth conditions, or from individual conditions separately. Each cross-validation (n = 10) was downsampled to the same number of cells. (<bold>D</bold>) Cross-validation (n = 10) using the YEASTRACT prior data. Networks are learned for all conditions together [BBSR (ALL) ●], for all conditions individually with TFA-BBSR followed by combination [BBSR (BY TASK) ▲], and for all conditions together in multi-task learning followed by combination [AMuSR (MTL) ◆]. Models are evaluated by (i) AUPR on the aggregate, final network and (ii) AUPR for each task-specific subnetwork from BBSR (BY TASK) (▲) and AMuSR (MTL) (◆).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig5-v3.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Low-Dimensional Clustering of Imputed Data Scatter plot after UMAP into 2-dimensional space.</title><p>Unmodified data (<bold>A</bold>) is compared to imputation methods to recover missing data using MAGIC (<bold>B</bold>), ScImpute (<bold>C</bold>), and VIPER (<bold>D</bold>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig5-figsupp1-v3.tif"/></fig></fig-group><p>The sparsity of data for each cell acquired using scRNAseq may negatively impact its utility in GRN construction. A commonly used technique to address missing data is data imputation. We tested the impact of several imputation packages on network inference: MAGIC (<xref ref-type="bibr" rid="bib114">van Dijk et al., 2018</xref>), ScImpute (<xref ref-type="bibr" rid="bib69">Li and Li, 2018</xref>), and VIPER (<xref ref-type="bibr" rid="bib21">Chen and Zhou, 2018</xref>). Whereas these methods can enhance separation of gene expression states in low-dimensionality projections (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1A</xref>), we find that they are either ineffective or detrimental to network inference (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). When the GRN is reconstructed from interactions selected at a precision threshold of 0.5, which takes into account how many interactions are correct according to the gold standard, no imputation method increases the number of recovered interactions compared to unmodified data. Data imputation with MAGIC increases the total number of confidently predicted (confidence &gt;0.95) interactions, but recovers fewer interactions that are correct according to the gold standard.</p></sec><sec id="s2-6"><title>Selection of priors for inference from Single-Cell yeast data</title><p>Algorithms for network inference perform poorly when making predictions based only on expression data (<xref ref-type="bibr" rid="bib45">Greenfield et al., 2010</xref>). Including prior knowledge of regulatory relationships and network topology improves model selection, and allows approximation of latent variables like TFA. Priors can be generated from regulatory interactions defined using methods such as chromatin immunoprecipitation sequencing (ChIP-seq) or analysis of transposase-accessible chromatin (ATAC-seq) and TF binding motifs, or from curated databases of interactions derived from literature. The source and processing of prior knowledge has a substantial effect on the size and accuracy of the learned network (<xref ref-type="bibr" rid="bib7">Azizi et al., 2018</xref>; <xref ref-type="bibr" rid="bib102">Siahpirani and Roy, 2017</xref>). We tested the impact on GRN reconstruction of prior data derived from literature, and from high-throughput experimental assays that encompass interactions between the entire yeast genome and the majority of known TFs (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). The best performance is obtained using a curated set of known TF-gene interactions obtained from YEASTRACT (<xref ref-type="bibr" rid="bib111">Teixeira et al., 2018</xref>). Generating priors using motif searching within open chromatin regions determined by ATAC-seq (<xref ref-type="bibr" rid="bib16">Castro et al., 2019</xref>; <xref ref-type="bibr" rid="bib84">Miraldi et al., 2019</xref>), and by modeling TF-DNA affinities in promoters (<xref ref-type="bibr" rid="bib116">Ward and Bussemaker, 2008</xref>) provides a considerable improvement over GRN reconstruction from TF expression without priors, but have lower performance than priors derived from curated data.</p></sec><sec id="s2-7"><title>Multi-task learning improves network inference and enables reconstruction of a unified Gene Regulatory Network from multiple conditions</title><p>Numerous methods exist for integrating information across different conditions and experiments that aim to reduce technical variation while retaining biologically meaningful differences (<xref ref-type="bibr" rid="bib50">Hicks et al., 2018</xref>; <xref ref-type="bibr" rid="bib67">Leek et al., 2010</xref>). The appropriate approach to integrating scRNAseq data for the purpose of GRN reconstruction remains unknown. We find that when we separate data based on environmental conditions and infer GRNs we obtain unique networks of differing quality (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). Learning a single network from all conditions by first combining the data can be compromised by technical variability and imbalance in the number of cells between conditions. Furthermore, normalizing batches to equal transcript depth risks suppressing differences which are true biological variability. An alternative approach is to treat the cells from each environmental condition as separate tasks. Separate tasks can be learned independently, without sharing information between tasks (implemented as BBSR (BY TASK)). This entails learning networks from each task, and then combining task-specific networks into a global network. Alternatively, networks can be learned together in a multitask learning (MTL) framework (<xref ref-type="bibr" rid="bib66">Lam et al., 2016</xref>), sharing information between tasks while they are learned, which we have implemented as Adaptive Multiple Sparse Regression (AMuSR) (<xref ref-type="bibr" rid="bib16">Castro et al., 2019</xref>). We find that, compared to network inference using all data simultaneously [BBSR (ALL)], treating conditions as separate network inference tasks provides a considerable improvement in performance (<xref ref-type="fig" rid="fig5">Figure 5Di</xref>). This is likely due to the retention of environmentally specific interactions that would otherwise be obscured using methods for normalizing data prior to GRN construction. The performance of the information sharing network inference approach [AMuSR (MTL)] and the non-sharing network inference approach [BBSR (BY TASK)] are very similar overall. We find that some individual tasks had modest improvements in model performance with AMuSR and others with BBSR (<xref ref-type="fig" rid="fig5">Figure 5Dii</xref>).</p><p>We constructed a global gene regulatory network using the YEASTRACT priors (as determined above) and our multi-task network inference (AMuSR) procedure. Eleven GRNs were jointly learned from each of the eleven environmental growth conditions; for each task a confidence score for each regulator-target interaction was calculated. GRNs learned for each condition were combined by rank summing condition-specific confidence scores to create a global confidence score for each potential interaction. All potential interactions are ranked by global confidence score, and a global GRN is constructed from interactions that meet the precision threshold of 0.5, as measured by recovery of known interactions (<xref ref-type="fig" rid="fig6">Figure 6A</xref>, <xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>). The resulting GRN comprises 6114 new interactions and 6114 interactions present in the priors, resulting in a total of 12,228 regulator-target interactions. We find that 5372 interactions from the priors are not recovered (recall of 0.532). The global GRN comprises an identified regulator for approximately half of all known genes (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1A</xref>). There is a positive correlation between expression level for a gene and the number of regulators for that gene (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1B</xref>) and 90% of the identified interactions are predicted to have activating effects (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1C</xref>). Many condition-specific networks have uniquely identified interactions (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1D</xref>), but more than 75% of the final network is composed of TF-gene interactions found in multiple conditions (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1E</xref>). Of the novel learned interactions (i.e. those not in the prior data), 60% have evidence of a TF-gene regulatory relationship when compared to the YEASTRACT database (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1F</xref>). 573 learned TF-gene interactions have evidence for physical localization of the TF to the target gene, and 2957 learned TF-gene interactions have evidence of expression changes when the TF is perturbed.</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Reconstruction of a Gene Regulatory Network Identifies New Regulatory Relationships.</title><p>A network inferred from the single-cell expression data using multi-task learning and the YEASTRACT TF-gene interaction prior, with a cutoff at precision &gt;0.5. (<bold>A</bold>) Network graph with known interaction edges from the prior in gray and new inferred interaction edges in red (<bold>B</bold>) Network graph of the 11 nitrogen-responsive transcription factors with known edges from the prior in gray and new edges in red (<bold>C</bold>) The number of interactions for each TF; interaction edges present in the prior that are not in the final network are included in black. The nitrogen TFs knocked out in this work are labeled in blue, and TFs with gene ontology annotations for mitotic cell cycle are annotated in green (<bold>D</bold>) Gene ontology classification of network interactions by the GO slim biological process terms annotated for the target gene and the regulatory TF (the GO term <underline>transcription from RNA pol II</underline> is omitted from the annotations for regulatory TFs).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig6-v3.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Summary of Learned GRN.</title><p>(<bold>A</bold>) Histogram of the number of regulators per target gene in the learned and prior network (<bold>B</bold>) Hexagonal heatmap of the ranked expression of a gene against the number of regulators for that gene; r<sup>2</sup> is calculated using Spearman’s Rank correlation (<bold>C</bold>) The number of interactions for a TF that are activating (positive) or repressing (negative). Bars are colored by the number of separate conditions in which the interaction is identified, with condition-specific networks selected using a precision threshold of 0.5 (<bold>D</bold>) The number of unique TF-gene interactions identified in condition-specific networks (<bold>E</bold>) The distribution of learned and prior interactions from the final, aggregate network in condition-specific networks (<bold>F</bold>) Evidence for learned network edges which are not in the prior. Interactions where YEASTRACT has evidence for a change in gene expression when the TF is perturbed are in tan. Interactions where YEASTRACT has evidence that the TF physically localizes to the gene are in blue. Interactions where YEASTRACT has no annotated evidence are in red.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig6-figsupp1-v3.tif"/></fig></fig-group><p>Within the nitrogen-regulated TF subnetwork comprising the 11 deleted TFs (<xref ref-type="fig" rid="fig6">Figure 6B</xref>) we identify 885 regulator-target interactions, of which 447 are novel, and 438 are present in the priors. This subnetwork contains many features consistent with expectations including co-regulation of targets by the NCR TFs. Overall, the global GRN has the largest number of target genes for general TFs (including <italic>ABF1</italic>, <italic>RAP1, CBF1,</italic> and <italic>SFP1</italic>), but we also define regulatory relationships for a total of 129 of the predicted 207 yeast TFs (<xref ref-type="fig" rid="fig6">Figure 6C</xref>). The poorest recovery of prior data is found for TFs that regulate environmental responses not included in our experimental design, such as the stress response TF <italic>MSN2</italic> and the mating TF <italic>STE12</italic>, highlighting the necessity of exploration of condition space for complete network reconstruction. Regulators and target genes can be mapped to Gene Ontology (GO) biological process slim terms, which are broad categorizations that facilitate pathway analysis. Ordering GO slim terms by the number of interactions in the learned GRN, we find that for target genes eight of the top ten GO slim terms are metabolism-related (<xref ref-type="fig" rid="fig6">Figure 6D</xref> i); in contrast, for regulatory TFs, five of the top ten GO slim terms are stress response related (<xref ref-type="fig" rid="fig6">Figure 6D</xref> ii).</p></sec><sec id="s2-8"><title>Identification of coregulation by cell cycle and environmental response TFs</title><p>Analysis of single cell expression in asynchronous cultures allows detection of cell cycle regulated relationships. The learned global GRN contains 257 genes that are regulated both by nitrogen TFs and by cell cycle TFs (<xref ref-type="fig" rid="fig7">Figure 7A</xref>). Many of these regulatory connections are novel; likely due to the fact that identifying interactions between metabolism and cell cycle are challenging in asynchronous cultures without single-cell techniques. Of these genes, 38 are annotated with the amino acid metabolic biological process GO term and 20 are annotated with the ion or transmembrane transport biological process GO term. Only 11 are annotated with the mitotic cell cycle biological process GO term, indicating that the majority of the interconnection between cell cycle and nitrogen response genes is due to regulation of metabolism-related genes by cell cycle TFs.</p><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Coordinated regulation of Nitrogen Response and Cell Cycle.</title><p>(<bold>A</bold>) A gene regulatory network showing target genes that are regulated by at least one nitrogen TF (blue) and at least one cell cycle TF (green). Target gene nodes are colored by GO slim term. Newly inferred regulatory edges are red and known regulatory edges from the prior are in gray. Transcription factor activity (TFA) is calculated from the learned network and then scaled to a z-score over all cells which do not have that TF deleted (e.g. <italic>gcn4Δ</italic> cells are omitted from the calculation for <italic>GCN4</italic> TFA). The mean TFA z-score for four selected conditions is inset for GAAC and NCR TFs (<bold>B</bold>) TFA for cell cycle TFs for each cell in the YPD growth condition.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig7-v3.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>Cell Cycle TF Activity Clusters within Growth Conditions.</title><p>Cells grown in YPD are plotted after UMAP with (<bold>A–B</bold>) z-Score of the calculated transcription factor activity (TFA) based on the learned network for (<bold>A</bold>) select nitrogen TFs and (<bold>B</bold>) cell cycle TFs (<bold>C</bold>) Log<sub>2</sub> of the expression of cell cycle TFs (<bold>D</bold>) Log<sub>2</sub> of the expression of cell cycle target genes.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-51254-fig7-figsupp1-v3.tif"/></fig></fig-group><p>We estimated the TFA for every TF in each cell, using the learned GRN and the single-cell expression matrix. The TFA of nitrogen responsive TFs is principally linked to growth condition as these TFs vary in activity between conditions (<xref ref-type="fig" rid="fig7">Figure 7A</xref>), but are generally similar within condition (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1A</xref>). As expected, we find that cells grown in rich media (YPD) have low TFA for the NCR TFs <italic>GLN3</italic> and the GAAC TF <italic>GCN4</italic>. The TFA for these TFs increases substantially upon treatment with rapamycin. By contrast, the estimated TFAs of cell cycle TFs varies within condition (<xref ref-type="fig" rid="fig7">Figure 7B</xref>); and are concordant with cell cycle responsive gene expression (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1B–D</xref>).</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><sec id="s3-1"><title>A robust scRNAseq and transcriptional barcoding method in yeast</title><p>Since the inception of single-cell RNA sequencing (<xref ref-type="bibr" rid="bib109">Tang et al., 2009</xref>), technological advances have resulted in the scale of datasets increasing from tens of cells to tens of thousands in a diversity of organisms. However, the number of cells recovered during scRNAseq in budding yeast has been comparatively limited in studies published to date (<xref ref-type="bibr" rid="bib38">Gasch et al., 2017</xref>; <xref ref-type="bibr" rid="bib87">Nadal-Ribelles et al., 2019</xref>). We present here the first report of droplet-based scRNAseq in this widely used model eukaryotic cell. Using a diverse library of transcriptionally barcoded gene deletion strains we were able to efficiently analyze the gene expression state of 38,255 cells using 11 experiments. In addition to facilitating multiplexed analysis of genotypes, transcriptional barcoding provides a facile means of identifying doublet cells within droplets thereby increasing the accuracy of single cell analysis.</p><p>Consistent with our understanding of global gene expression variation first characterized in foundational studies of the transcriptome (<xref ref-type="bibr" rid="bib29">DeRisi et al., 1997</xref>; <xref ref-type="bibr" rid="bib37">Gasch et al., 2000</xref>), we find that environmental condition is the primary determinant of the gene expression state of individual yeast cells. However, we observe significant heterogeneity in individual cell gene expression within conditions. Much of this variation can be explained by the mitotic cell-cycle. It is important to note that we do not remove or suppress this cell-cycle driven variance. The cell cycle is itself driven by transcriptional regulators, and our goal is to build a network that integrates cell-cycle regulation with regulated responses to the environment. The ability to access the crosstalk between signalling pathways and the cell cycle program is a key advantage to performing single-cell sequencing in asynchronous cultures, which bypasses many of the limitations of synchronized bulk sequencing experiments. It is also important to note that in several stressful growth conditions, we see heterogeneous cellular responses; some cells appear to be proliferative, while other cells have downregulated translational machinery and upregulated stress response genes. This is an interesting outcome by itself, as it is further evidence of bet-hedging strategies (<xref ref-type="bibr" rid="bib68">Levy et al., 2012</xref>), and we expect that the presence of multiple distinct transcriptional states between cells in the same environmental condition is advantageous for network inference. Model performance, as measured by AUPR, can vary considerably when learning networks from any single growth condition (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). Cells in rich YPD media do not require many anabolic pathways to be active, and primarily express genes required for the cell-cycle, translation, and glycolysis; in contrast, cells in minimal MMD media must express these pathways plus many anabolic pathways to synthesize nitrogenous bases, cofactors and amino acids. We find that this increased transcriptional diversity results in better overall performance. Nonetheless, the largest performance gain comes from aggregating networks from cells in different conditions (<xref ref-type="fig" rid="fig5">Figure 5D</xref>), which demonstrates a general advantage to learning GRNs from heterogeneous data.</p><p>Deletion of specific transcription factors results in changes in single cell gene expression for some TFs in some conditions. However, genotypic effects are comparatively minor. We believe that this is due to multiple factors including functional redundancy between TFs, physiological adaptation to the genetic perturbation and the conditional specificity of TFs. It is likely that perturbations that are transiently induced, and result in increased TF activity (<xref ref-type="bibr" rid="bib80">McIsaac et al., 2013</xref>) may be effective in eliciting detectable responses in gene expression, facilitating causal inference. The use of precise gene deletions does provide several advantages over the use of CRISPR/Cas9-based perturbations as engineered deletions are unambiguous whereas the efficiency of perturbation by CRISPR/Cas9 varies for different guide RNAs.</p></sec><sec id="s3-2"><title>A generalizable framework for GRN construction using scRNAseq</title><p>Constructing GRNs from single cell gene expression data is a universal goal in all organisms. A yeast single-cell expression matrix has several beneficial properties for design and testing of gene regulatory network inference models as there exist high quality known interactions and TF binding motifs. The issues of data sparsity and low sampling rates are likely to be problems common in experiments in any organism using scRNAseq. We find that techniques that have been developed for normalization and imputation do not improve performance of the additive linear model-based inference of the Inferelator algorithm (<xref ref-type="fig" rid="fig5">Figure 5</xref>). However, there are significant opportunities for development of smoothing techniques that would enhance network inference, perhaps targeting latent biophysical parameters like transcription factor activity. It seems reasonable to assume that these biophysical parameters should be stable within the local neighborhood of samples, and the activity calculation that we have used is ill-conditioned and potentially unstable. This is of particular concern when working with undersampled single-cell data and we are actively addressing this issue.</p><p>We find that the application of multitask learning is well suited to GRN reconstruction from scRNAseq data. Jointly learning multiple related tasks improves generalization accuracy, especially in scenarios in which datasets are undersampled (<xref ref-type="bibr" rid="bib15">Caruana, 1998</xref>), and has the desirable side benefit of mitigating the need for complex batch-correction techniques that aim to address technical variation between experiments. Removing batch-effect technical noise from data without suppressing interbatch biological variability remains an unsolved problem, and therefore application of multitask learning approaches to network inference from single-cell data is likely to be generally applicable to integrating scRNAseq data from different cell types and conditions.</p></sec><sec id="s3-3"><title>A global GRN for budding yeast</title><p>Using our scRNAseq dataset, we reconstructed a global GRN with several novel regulatory relationships. Among the most novel of these interactions are those between cell-cycle associated TFs and targets and nitrogen TFs and target genes. The cell cycle and metabolism are, by necessity, interconnected, and the mechanism of rapamycin in arresting cell cycle through TOR is well-established (<xref ref-type="bibr" rid="bib49">Heitman et al., 1991</xref>). Several studies have identified metabolic cycling patterns which are believed to be driven by the cell cycle (<xref ref-type="bibr" rid="bib13">Burnetti et al., 2016</xref>; <xref ref-type="bibr" rid="bib104">Slavov and Botstein, 2011</xref>; <xref ref-type="bibr" rid="bib112">Tu et al., 2005</xref>). Although regulatory connections between environmental sensors, metabolism, and the cell cycle have been previously reported, a comprehensive regulatory network does not exist, in large part because of the difficulty of experimentally perturbing cell cycle without confounding metabolic changes. Our study provides a valuable first step in identifying specific regulatory connections that were previously inaccessible, and which are necessary to create a complete map of the yeast regulome.</p><p>Incorporation of additional information into the network inference process, including information about interactions between transcription factors such as functional redundancy and heterodimerization, would likely improve learning of the network. We note that several TFs have few learned targets reflecting the requirement for surveying conditions in which particular TFs are active. For example, <italic>STE12</italic> and <italic>TEC1</italic> are mating-related TFs that we expect to be entirely inactive in our diploid cells; <italic>MSN2</italic> and <italic>YAP1</italic> are stress-responsive TFs that respond to specific stimuli that were not tested in our study. Targeted analysis of the GRN with rationally designed genetic perturbations and environmental conditions will maximize the additional information that can be recovered from future experiments.</p></sec><sec id="s3-4"><title>Conclusion</title><p>Single-cell sequencing is a transformative method for systems biology. To date, scRNAseq has been widely applied to the problem of defining different cell types. However, the ability to simultaneously study the expression of hundreds of genotypes in different conditions, and sample the expression state of thousands of cells, presents a rich source of information for the purpose of GRN reconstruction. Our study implements this approach in budding yeast, the workhorse of systems biology, and establishes a generalizable framework for GRN reconstruction from scRNAseq data in any organism.</p></sec></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><p>Requests for strains and reagents should be directed to David Gresham (<ext-link ext-link-type="uri" xlink:href="https://as.nyu.edu/content/nyu-as/as/faculty/david-gresham.html">dgresham@nyu.edu</ext-link>). Requests related to computational analysis and code should be directed to Richard Bonneau (<ext-link ext-link-type="uri" xlink:href="https://cims.nyu.edu/people/profiles/BONNEAU_Richard.html">rb133@nyu.edu</ext-link>). There are no restrictions on the materials or the code used in this work. All materials are released under CC-BY 4.0 and all code is available under the permissive MIT or BSD licenses.</p><sec id="s4-1"><title>Yeast strain construction and growth</title><p>All yeast strains were generated from the prototrophic FY4 (MAT<bold>a</bold>) or FY5 (MAT<bold>ɑ</bold>) background strains. Yeast were transformed using the standard lithium acetate transformation protocol (<xref ref-type="bibr" rid="bib41">Gietz and Schiestl, 2007</xref>). <italic>E. coli</italic> were transformed using the standard chemically competent transformation protocol. Plasmid constructions were confirmed by sanger sequencing. Yeast genotypes, plasmid sequences, and oligonucleotide sequences are provided as <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>-supplemental tables 1-3. Media formulations are provided as <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>-supplemental table 4.</p><sec id="s4-1-1"><title>Construction of barcoded deletion cassettes</title><p>The deletion cassette plasmid was constructed by amplifying pTEF::KAN<sup>R</sup> from pUG6 (Euroscarf) and tTEF from pUG6, with an overlapping junction between KAN<sup>R</sup> and tTEF containing two BbsI sites for golden-gate mediated barcode cloning. These pieces were assembled into pUC19 using gibson isothermal assembly to generate DGP304. This plasmid was then modified by linearizing with BamHI and XbaI, amplifying a bacterial GFP expression cassette from pWS158 (Addgene), and assembled using gibson isothermal assembly to generate DGP306.</p><p>Gene deletion barcodes were created by synthesizing an oligonucleotide containing flanking PCR handles (M13F and M13R), flanking BbsI sites for golden gate cloning, and the degenerate sequence caNNgNNgtNNgNNgtNNgNNgt. The mixture of oligonucleotides was double-stranded using <italic>E. coli</italic> DNA Polymerase I, Large (Klenow) fragment. Klenow buffer (1x NEB Buffer 2.1 [NEB #B7202S]) was mixed with 250 nM barcode oligonucleotide, 250 nM M13R primer, 200 nM/each dNTP [NEB #N0447S], incubated at 80°C and slowly cooled to room temperature. The DNA Polymerase I, Large (Klenow) Fragment (NEB #M0210S) was added to 0.1 U/µL and the reaction was incubated at 37°C for 30 min. The polymerase was heat-inactivated by placing the reaction at 75°C for 20 min. The resulting dsDNA cassette was used with no further cleanup.</p><p>The barcode was inserted into the 3’ untranslated region of the pTEF::KAN<sup>R</sup>::tTEF yeast selection marker cassette in DGP306 by BbsI-mediated golden gate cloning. A golden gate reaction was prepared with 1x Thermo FastDigest Buffer [Thermo #ER1011], 1 mM ATP, 10 mM DTT, 2 U/µL T4 DNA Ligase [NEB #M0202S], 1 U/µL BpiI [Thermo #ER1011], 10 ng/µL DGP306, 25 nM barcode dsDNA, and incubated in a thermocycler using the following program: 37°C 20 min; 25x cycles of 37°C for 5 min and 16°C for 5 min; 37°C for 20 min; 80°C for 20 min. An additional 1 U/µL BpiI was then added to the reaction mix and incubated at 37°C for 30 min to linearize any remaining uncloned plasmid.</p><p>The golden gate cloning reaction was transformed into One Shot TOP10 <italic>E. coli</italic> (ThermoFisher #C404003). Cloning and transformation efficiency was estimated by plating 2% of the transformation onto LB + ampicillin plate and counting GFP<sup>+</sup> and GFP<sup>-</sup> colonies. The remainder of the reaction was inoculated into 200 mL molten LB + ampicillin + 0.6% (w/v) SeaPrep Agarose (Lonza 50302) media, thoroughly mixed, snap cooled in an ice bucket, and incubated overnight at 37°C. The soft agar culture was then collected by centrifugation, washed with PBS, and resuspended in 2 mL 50% glycerol. 100 µL of this mixture was used to inoculate a culture of 100 mL LB + ampicillin and the remainder stored at −80°C in aliquots. The 100 mL culture was grown for 8 hr at 37°C, harvested, washed with PBS, and stored at −20°C until midiprepped (Qiagen) according to the manufacturer’s protocol.</p></sec><sec id="s4-1-2"><title>Construction of a barcoded Transcription Factor deletion array</title><p>The degenerate barcoded plasmid was used as template for PCR using primers containing gene-specific targeting homology arms (1x NEB Q5 Master Mix #M0494S, 1 ng template plasmid, 250 nM/each oligo). The PCR amplicon was then transformed into FY4 and plated on YPD+G418 to select transformants. Transformants containing the gene deletion were confirmed using colony PCR and gene-specific primers and a KANR primer. PCR products of correct transformants were cleaned using silica spin columns (Qiagen) according to the manufacturer’s protocol and the barcode identified by Sanger sequencing. At least six uniquely barcoded strains (i.e. biological replicates) were generated for each genotype, with the criteria that each barcode had to differ by at least three bases, ensuring that the probability of barcode collisions is extremely low.</p><p>The plasmid DGP328 (pTEF::NATR::tTEF) was used as template for PCR using primers containing the same gene-specific targeting homology. The PCR amplicon was transformed into FY5 and plated on YPD+nourseothricin. Positive transformants were confirmed using colony PCR with gene-specific primers and a NATR primer.</p><p>FY4-derived MATa strains were arrayed in a 96-well plate (Corning 3788) and then pinned (V and P Scientific #VP407FP12) onto YPD in an OmniTray (Nunc 165218). FY5-derived MATɑ strains were arrayed in a 96-well plate so that the same gene was disrupted in matching wells of the MATa and MATɑ plates and then pinned onto YPD. These arrays are grown overnight at 30°C. The MATa array and MATɑ array were then pinned to the same YPD plate to create spots where MATa and MATɑ strains were overlaid. The plate layout was designed so that some locations had only MATa strains, only MATɑ strains, or no strains, to control for mating, contamination, and the efficacy of diploid selection. The mating array was grown overnight at 30°C to allow mating to occur and then pinned to a YPD+G418+nourseothricin plate to select for MATa/MATɑ diploids. This diploid selection plate was grown overnight and then pinned to a YPD+G418+nourseothricin plate for a second round of diploid selection. The second diploid selection plate was grown overnight at 30°C and then pinned to a YPD+G418+nourseothricin plate for a third round of diploid selection at 30°C. This plate was then pinned to several replicate 96 well round-bottom plates containing 200 µL YPD+G418+nourseothricin in each well. These plates were cultured with shaking overnight at 30°C, then centrifuged and the media aspirated. The cells were resuspended in 50% glycerol and the plates stored at −80°C.</p></sec><sec id="s4-1-3"><title>Culturing and harvest</title><p>The barcoded deletion array was pinned from a frozen stock plate at −80°C onto a YPD plate for recovery and grown overnight at 30°C. The first recovery plate was then pinned to a second recovery YPD plate and grown overnight at 30°C. The second recovery plate was pinned to a 96 well round-bottom plate containing 200 µL YPD in each well and grown overnight at 30°C. The cultures from this plate were pooled, washed 2x with 50 mL PBS, and then resuspended in 1 mL PBS. 250 µL of the washed cells were used to inoculate 50 mL of the relevant media for the specific experimental condition in a shake flask. These flasks were grown for 4 hr. The experiment grown to diauxic shift was grown for 10 hr. We confirmed that glucose in the media was exhausted between hour 9 and hour 10 using a hexokinase-based assay (R-Biopharm #10716251035). All other steps of harvesting cells were identical to the 4 hr experiments. The experiment treated with rapamycin was grown for 3 hr and 30 min in YPD, and then 10 µL of rapamycin stock (1 mg/mL Millipore #553210 in ethanol) was added to a final concentration of 200 ng/mL. Cells were then harvested at 4 hr (after 30 min in rapamycin).</p><p>Cell count per mL at harvest was determined using a Beckman Coulter Z2 Particle Counter #6605700. Cell density (cells/mL) for each condition at harvest was as follows: (YPD 1.4e7; RAPA 1.2e7; YPEtOH 1.0e7; NLIM-GLN 0.5e7; NLIM-NH4 0.8e7; NLIM-PRO 0.4e7; NLIM-UREA 0.5e7; MMD 1.1e7; MMEtOH 0.7e7; CSTARVE 0.1e7) A volume of culture containing ~10<sup>8</sup> cells was collected and the cells pelleted by centrifugation. These cells were immediately resuspended in 1 mL RNALater (Qiagen #76104), washed 2x with 1 mL RNALater and resuspended in a final volume of 500 µL RNALater. This suspension was stored at −20°C for 12 to 72 hr.</p></sec></sec><sec id="s4-2"><title>Library preparation and sequencing</title><p>All steps below used RNAse-free reagents.</p><sec id="s4-2-1"><title>Single cell library preparation</title><p>Cells stored in RNALater were removed from −20°C and ~10<sup>7</sup> cells were washed 2x with 1 mL spheroplasting buffer (50 mM Sodium Phosphate pH 7.5, 1M Sorbitol, 10 mM EDTA, 2 mM DTT, 100 µg/mL BSA). Cells from fermentative phase growth cultures were then resuspended in 100 µL spheroplasting buffer + 0.1 U Zymolyase 100T (Zymo Research #E1004). Cells from respiratory phase growth cultures or starvation cultures were resuspended in 100 µL spheroplasting buffer + 0.25U Zymolyase 100T. The spheroplasting reaction was incubated at 37°C for exactly 20 min, and then the spheroplasted cells were pelleted and resuspended in 500 µL RNALater for 5 min on ice. After this incubation the spheroplasted cells were pelleted and washed 3x with 1 mL wash buffer (10 mM TRIS pH 8, 1M Sorbitol, 100 µg/mL BSA) and resuspended in 1 mL wash buffer. The cells were visualized to confirm spheroplasting and counted using a hemocytometer. A dilution equal to ~5×10<sup>6</sup> cells/mL in wash buffer was prepared and then immediately used for single cell isolation.</p><p>Single cell library preparation was done using the 10x Genomics Chromium 3’ v2 Single Cell Gene Expression Kit (10x Genomics #120237), following the kit protocol. 66.2 µL of single-cell master mix was prepared to which 27.7 µL H<sub>2</sub>O was added. The microfluidic Chromium Single Cell A Chip (10x Genomics #120236) was then prepared for use. 6 µL of prepared spheroplast cell suspension was added to the single-cell master mix, and then immediately transferred to the microfluidics chip. Hydrogel beads and partitioning oil were added according to the manufacturer’s protocol, and the cells were encapsulated with hydrogel beads using the 10x Genomics Chromium Controller. Following emulsification, reverse transcription and cleanup was performed according to the manufacturer’s protocol. Whole transcriptome amplification was performed using a total of 10 cycles of PCR. Cleanup, fragmentation, adapter ligation, and indexing was performed according to the manufacturer’s protocol, using 8–10 cycles of PCR for the indexing reaction.</p><p>Transcribed barcodes were amplified from the whole transcriptome amplification prior to fragmentation. The KAN<sup>R</sup> transcript containing the genotype barcode was amplified in a reaction(1x KAPA HiFi Hotstart Readymix [Kapa #KK2602], 200 nM/each primer, 1 µL 10x whole-transcriptome DNA), using 6 cycles of PCR (98°C for 3:00; 6 cycles of 98°C for 0:20, 63°C for 0:20, and 72°C for 0:20 min; 72°C for 1:00 min). The amplicon pool was then purified with 1x volume of SPRIselect beads (Beckman Coulter #B23317) and eluted into 24 µL H<sub>2</sub>O. To this eluate, 25 µL of 2x KAPA HiFi Hotstart Readymix was added, as well as 200 nM/each indexing primers. The indexing reaction was cycled for 8–10 cycles of PCR, using the 10x Genomics indexing PCR reaction settings (98°C for 0:45; 8-10x cycles of 98°C for 0:20, 54°C for 0:30, and 72°C for 0:20; 72°C for 1:00).</p><p>Library fragment sizes were determined using a High Sensitivity D1000 Screentape (Agilent #5067–5584) and quantified with the KAPA illumina library quantification system (Roche #KK4953) on a Roche lightcycler 480. Libraries from each condition were pooled so that 99% of the pool consisted of the single-cell transcriptome library and 1% of the pool consisted of the genotype barcode amplicon. Samples were then pooled for multiplex sequencing on an Illumina NextSeq 500 with the NextSeq 500/550 v2.5 High Output 150 Cycle kit (Illumina #20024907), using the sequencing parameters recommended by 10x Genomics (Read 1: 26 bp, Read 2: 98 bp, Index 2: 8 bp) and standard illumina read and indexing primers.</p></sec><sec id="s4-2-2"><title>Bulk RNA library preparation</title><p>Each of the six wild-type yeast strains (MAT a/ɑ Δho::KanMX/Δho::NatMX) were separately grown overnight in YPD at 30°C.~10<sup>8</sup> cells (0.5 mL) of overnight culture was subcultured into separate 50 mL flasks of pre-warmed YPD and cultured with shaking for 4 hr at 30°C. At 4 hr, for each culture flask,~10<sup>8</sup> cells were pelleted by centrifugation and immediately transferred to a microfuge tube, then snap-frozen in liquid nitrogen for storage at −80°C.</p><p>For each of six wild-type samples snap-frozen in liquid nitrogen and stored at −80°C, cell pellets were removed from −80°C storage and immediately resuspended in 1 mL TRIZOL (ThermoFisher #15596026), which is an organic extraction reagent with phenol and the chaotropic salt guanidinium thiocyanate (<xref ref-type="bibr" rid="bib22">Chomczynski and Sacchi, 2006</xref>). After sitting at RT for 5 min, 200 µL chloroform was added and tubes were mixed by inversion. Organic and aqueous phases were separated by centrifugation at 4°C. The aqueous phase was re-extracted with 500 µL acid phenol:chloroform (ThermoFisher #AM9720), then the aqueous phase from that extraction was re-extracted with 500 µL chloroform. 1:10th volume 5M NH<sub>4</sub>OAc (ThermoFisher #AM9070G) and 2.5x volumes of ice-cold absolute ethanol were added to the aqueous phase from the chloroform extraction, and RNA was precipitated overnight at −80°C. After precipitation, the RNA pellet was washed with ice-cold 70% ethanol and dissolved into 100 µL RNA elution buffer (10 mM TRIS pH8, 0.05% TWEEN-20). RNA was quantified by Qbit (ThermoFisher #Q10210) and a working stock of 5 ng/µL RNA was prepared for each sample.</p><p>15 µL of reverse transcription mix (5 µL 5x Maxima RT Buffer, 5 µL 20% (w/v) Ficoll PM-400 [GE Life Sciences #17030010], 2.5 µL 10 mM/each dNTP [New England Biolabs #N0447S], 0.5 µL Lucigen NxGen RNase Inhibitor [Lucigen #30281–1], 0.5 µL 50 µM Template Switch Oligo [IDT], 0.5 µL 50 µM Barcode/UMI/poly-dT Oligo [IDT], 0.5 µL Maxima H Minus Reverse Transcriptase [ThermoFisher #EP0752], 0.5 µL H<sub>2</sub>O) was added to 50 ng (10 µL) of RNA. Each reaction contained a separate barcoded poly-dT oligo such that each of the six biological replicate samples contain a unique, identifiable barcode sequence. Reverse transcription was carried out at 53°C for 1 hr, followed by heat inactivation at 85°C for 5 min. 98 µL RLT Buffer [Qiagen] and 2 µL MyOne Silane beads [ThermoFisher #37002D] were added, mixed, and allowed to sit at RT for 10 min. cDNA was then isolated by magnetic separation of beads, followed by 2x washes with 200 µL 80% ethanol. Beads were pooled together and all cDNA was eluted into 40 µL of DNA elution buffer (10 mM TRIS pH8, 0.05% TWEEN-20, 1 mM DTT). 60 µL WTA master mix (50 µL 2x KAPA HiFi Hotstart Readymix, 1 µL 100 µM Forward Oligo, 1 µL 100 µM Reverse Oligo, 8 µL H<sub>2</sub>O) was added and whole transcriptomes were amplified using 12 cycles of PCR (98°C for 3:00; 12 cycles of 98°C for 0:20, 55°C for 0:20, and 72°C for 1:15 min; 72°C for 3:00 min). The amplified pool was then purified with 0.6x volume of SPRIselect beads and eluted into 25 µL DNA elution buffer. Amplified DNA was quantified using a high sensitivity D5000 ScreenTape (Agilent #5067–5592).</p><p>Amplified whole-transcriptome DNA was tagmented with a nextera XT kit (Illumina #FC-131–1096) as follows. 3 ng of DNA was diluted to a total volume of 10 µL with DNA elution buffer. 20 µL TD buffer and 10 µL ATM was added and DNA was tagemented at 55C for 10 min. The reaction was halted with 10 µL NT buffer, and the fragment pool was indexed by adding 30 µL NPM buffer, 5 µL illumina index 2 (i7) adapter primer, 5 µL 5 µM DG1954 (no-index primer) and amplifying using 12 cycles of PCR (95°C for 0:30; 12 cycles of 95°C for 0:10, 55°C for 0:30, and 72°C for 0:30 min; 72°C for 5:00 min). Libraries were purified by double-sided SPRI selection. 55 µL SPRIselect beads (0.55x) were added to the nextera indexing reaction, and the unbound supernatant was transferred to a clean tube. 20 µL SPRIselect beads were added (0.75x total), and after binding and washing, DNA was eluted into 20 µL DNA elution buffer. Libraries were checked for size with a High Sensitivity D1000 Screentape, and quantified with the KAPA illumina library quantification system on a Roche lightcycler 480. Libraries were sequenced on an Illumina NextSeq 500 with the NextSeq 500/550 v2.5 High Output 150 Cycle kit, using the sequencing parameters recommended by 10x Genomics (Read 1: 26 bp, Read 2: 98 bp, Index 2: 8 bp) and standard illumina read and indexing primers.</p></sec><sec id="s4-2-3"><title>Processing sequencing data</title><p>Sequencing results were analyzed using the Cellranger pipeline (10x Genomics) v2.1.0 and custom python scripts written for this project, which are located in the fastqTomat0 GitHub repository (<ext-link ext-link-type="uri" xlink:href="https://github.com/flatironinstitute/fastqToMat0">https://github.com/flatironinstitute/fastqToMat0</ext-link>). The reference genome was obtained from Ensembl (Version R64-1-1) as a FASTA file, and the reference annotations were obtained from Ensembl (Version R64-1-1.93) in GTF format. The reference transcript annotations were altered to incorporate 5’ and 3’ untranslated regions using data from generated using TIF-seq (<xref ref-type="bibr" rid="bib92">Pelechano et al., 2013</xref>) and the gffAnnotate.py script from fastqTomat0. The antibiotic resistance marker cassettes was added to both the FASTA and GTF files using command line tools. A STAR reference genome was then created from the modified GTF and FASTA files using cellranger mkref.</p><p>Raw single-cell sequencing reads were converted into FASTQ files using cellranger mkfastq and a 10x Genomics Index CSV file. These FASTQ reads were then aligned to the reference genome and counted using cellranger count. The FASTQ files for indexes not corresponding to the 10x single-cell transcriptome library were processed with the fastqBCLinker.py script from fastqTomat0, which identifies the genotype for each single-cell read and creates a TSV file mapping cell barcodes to genotypes. The count data from cellranger count and the barcode data from fastqBCLinker.py was combined by the tenXtomatrix.py script from fastqTomat0. This processing step discards doublet single-cell reads, which are identified by removing ‘cells’ which map to more than one of the 72 genotype-specific barcodes. We expect that 1/72 of these doublets will have the same barcode, and so we expect that ~ 98.5% of doublets will be removed and ~1.5% will be retained. This script produces a dense TSV matrix of counts per gene per cell that can be imported with python’s <monospace>pandas.read_table()</monospace> or R’s <monospace>read.table()</monospace>. This matrix is provided as <xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>. This final data matrix is assembled from 11 independent single-cell sequencing batches, each corresponding to a single shake flask with a different growth condition.</p><p>Raw bulk RNA sequencing reads were converted into FASTQ files using bcl2fastq. These FASTQ reads were then aligned to the reference genome and counted using cellranger count, after adding the appropriate custom chemistry configuration and barcode whitelist to cellranger. The count data from cellranger count was processed by the tenXtomatrix.py script from fastqTomat0 into a TSV matrix of counts per gene per sample, which is included in <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>.</p></sec></sec><sec id="s4-3"><title>Network inference</title><sec id="s4-3-1"><title>Inferelator</title><p>Network inference with the Inferelator consists of three major steps; data preprocessing and filtering, estimation of transcription factor activities, and regularized regression. Cross-validation of network inference parameters was performed by randomly selecting half of the genes in the gold standard network and removing them. To prevent circularity, any genes that were used in the gold standard were removed from the prior data during cross-validation; for tests where the gold standard network was also used as a prior, this meant that half of the genes in the gold standard network were retained and defined as the gold standard, and half of the genes in the gold standard network were used as priors. A summary table of the cross-validation results is provided as <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>-Supplemental Table 5.</p><p>The randomized negative control was performed by randomly reassigning gene names in the prior data. Transcription factor labels and expression values were otherwise unchanged. The simulated negative control was performed using simulated data by constructing a probability distribution for the yeast transcriptome from estimates of absolute mRNA abundances (<xref ref-type="bibr" rid="bib65">Lahtvee et al., 2017</xref>) and randomly sampling this distribution using the synthesize_data.py script from the fastqToMat0 package. Metadata and total UMI count for each cell were retained in this negative control; only the individual gene counts were synthesized from the simulated control probability distribution.</p><p>See below for details on each step of the network inference procedure.</p></sec><sec id="s4-3-2"><title>Single-Cell preprocessing and filtering</title><p>Single cell data was loaded as an integer UMI count matrix (Cells x Genes). Genes with a variance of 0 for all cells were removed. The count matrix was then transformed by log scaling using log<sub>2</sub>(x+1). For data sets that had already undergone library normalization and transformation as a result of an imputation method, this transformation preprocessing step was skipped.</p></sec><sec id="s4-3-3"><title>Single-Cell imputation</title><p>All imputation methods used the untransformed integer UMI count matrix (Cells x Genes) in which genes with a variance of 0 had been removed. For MAGIC, count data was library size normalized with the <monospace>library.size.normalize()</monospace> function from the Rmagic package, then transformed by square-root, and subjected to imputation with the <monospace>magic()</monospace> function from the Rmagic package. For VIPER, count data was subjected directly to the <monospace>VIPER()</monospace> function from the VIPER package, using the parameters recommended by the VIPER authors for 10x genomics UMI count data. For ScImpute, count data was normalized by the method included in the ScImpute package and then subjected to imputation with the <monospace>imputation_wlabel_model8()</monospace> function from the ScImpute package. The R script to perform these imputations is included with <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>.</p></sec><sec id="s4-3-4"><title>Construction of known prior TF-Gene networks</title><p>Construction of the gold standard prior network has been previously described (<xref ref-type="bibr" rid="bib110">Tchourine et al., 2018</xref>); this gold standard network consists of 1403 signed [−1, 0, 1] interactions, for which sign represents activation (+) or inhibition (-), in a 998 genes by 98 transcription factors regulatory matrix. YEASTRACT priors were retrieved from the YEASTRACT database (<xref ref-type="bibr" rid="bib111">Teixeira et al., 2018</xref>) using the <italic>generate regulation matrix</italic> tool. Both activation and inhibition interactions were included, but only those that are supported by both DNA binding and expression evidence. The YEASTRACT prior network consists of 11486 unsigned [0, 1] interactions in a 3912 genes by 152 transcription factors regulatory matrix. Construction of the ATAC-motif priors has been previously described (<xref ref-type="bibr" rid="bib16">Castro et al., 2019</xref>; <xref ref-type="bibr" rid="bib84">Miraldi et al., 2019</xref>), and are built from chromatin accessibility data and known transcription factor binding motifs. The ATAC-motif prior network consists of 71,865 signed integer interactions with a range of [−11,. .., 26], for which sign represents activation (+) or inhibition (-) and absolute values represent the number of motif occurrences, in a 5551 genes by 138 transcription factors regulatory matrix. Bussemaker-priors were generated from modeling transcription factor affinities for regulatory DNA motifs (<xref ref-type="bibr" rid="bib116">Ward and Bussemaker, 2008</xref>). The Bussemaker prior network consists of unsigned floating-point values [0, 20] that reflect estimated binding affinities in a dense 6516 genes by 123 transcription factors regulatory matrix.</p></sec><sec id="s4-3-5"><title>Estimating transcription factor activities (TFA)</title><p>Log-transformed single-cell data was transposed into matrix <bold><italic>X</italic></bold>, in which columns are individual cells and rows are genes. <bold><italic>P</italic></bold> is the connectivity matrix of known prior regulatory interactions between transcription factors (in columns) and genes (in rows). <bold><italic>P</italic></bold><italic><sub>i,k</sub></italic> is zero if there is no known regulatory interaction between transcription factor <italic>k</italic> and gene <italic>i. <bold>A</bold></italic> is the activity matrix, where the columns are the individual cells as in <bold><italic>X</italic></bold> and rows are the transcription factors. We model the expression of gene <italic>i</italic> in individual cell <italic>j</italic> as a linear combination of the activities of the a priori known regulators of gene <italic>i</italic> in individual cell <italic>j</italic> (1). In practice, this means that we use the known targets of a transcription factor to derive its activity.<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mi>T</mml:mi><mml:mi>F</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow/></mml:munderover><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>In matrix form, <xref ref-type="disp-formula" rid="equ1">Equation 1</xref> can be written as <inline-formula><mml:math id="inf1"><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">P</mml:mi><mml:mi mathvariant="bold-italic">A</mml:mi></mml:math></inline-formula>. This is an overdetermined system, meaning that there are more equations than unknowns and therefore there is no solution if all equations are linearly independent. We approximate <inline-formula><mml:math id="inf2"><mml:mi>A</mml:mi></mml:math></inline-formula> by finding <inline-formula><mml:math id="inf3"><mml:mi>Â</mml:mi></mml:math></inline-formula> that minimizes <inline-formula><mml:math id="inf4"><mml:mo>‖</mml:mo><mml:mi>P</mml:mi><mml:mi>Â</mml:mi><mml:mo>-</mml:mo><mml:mi>X</mml:mi><mml:mo>‖</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:math></inline-formula>. If a transcription factor has no prior targets present in <bold><italic>P</italic></bold>, we use the expression of that transcription factor as a proxy for its activity.</p></sec><sec id="s4-3-6"><title>Inferring regulatory interactions, single-task (Bayesian Best Subset Regression)</title><p>We utilize a bayesian best-subset regression (BBSR) method, previously described (<xref ref-type="bibr" rid="bib46">Greenfield et al., 2013</xref>), for single-task network inference. At steady state, we model the expression of a gene <italic>i</italic> in individual cell <italic>j</italic> as a linear combination of the activities of its regulators in individual cell <italic>j</italic> (2). For each gene <italic>i</italic>, we limit the number of potential regulators <inline-formula><mml:math id="inf5"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to the ten with the highest context likelihood of relatedness, calculated from the mutual information between all regulators and the gene <italic>i</italic> (<xref ref-type="bibr" rid="bib77">Madar et al., 2010</xref>), in addition to any a priori known regulator of gene <italic>i</italic>. Limiting the regulators is necessary before best subset regression, when we find the least squares solution to all possible combinations of predictors in set <inline-formula><mml:math id="inf6"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. Because we expect a limited number of transcription factors to regulate a particular gene, our goal is to find a sparse solution for β, in which non-zero entries define both the strength and direction (activation or repression) of a regulatory relationship.<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow/></mml:munderover><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Prior knowledge can be incorporated using Zellner’s g-prior on the regression parameters β; in this work, we include prior interactions in the set of predictors to be modeled by best subset regression, but we do not further bias the predictors chosen with a g-prior on the regression parameters. We select the model with the lowest Bayesian Information Criterion, which adds a theoretically derived penalty term to the training error to account for model complexity and thereby reduce generalization error. After this step, the output is a matrix of inferred regression parameters β, where each entry corresponds to a regulatory relationship between transcription factor <italic>k</italic> and gene <italic>i</italic>.</p></sec><sec id="s4-3-7"><title>Inferring regulatory interactions, multi-task (AMuSR)</title><p>The multitask approach used here entails a joint inference of regulatory networks across multiple expression datasets. In addition to the linear assumption, in which gene expression is a linear function of the activities of regulators, we also assume that much of the underlying regulatory network is shared among related datasets (conditions). Here, we extend a previous version of the Inferelator that implements Adaptive Multiple Sparse Regression (AMuSR), which is designed to leverage cross-dataset commonalities while preserving relevant differences (<xref ref-type="bibr" rid="bib16">Castro et al., 2019</xref>).</p><p>There are multiple ways of dividing the existing yeast data into multiple network data subsets, which we refer to as tasks. Within our experimental design, cells are processed and sequenced as batches, which are taken from separate environmental growth conditions. Differences between these batches are a combination of technical and biological variation. The technical variation can come from batch effect due to stress and energy-source differences associated with differing growth conditions (for example via direct effects on cell wall and thus cell lysis/yield), as well as from differences in sample preparation and sequencing. Differences in growth condition also generate biologically significant variation in gene expression due to differences in regulatory program activation. Removing technical variation while retaining biological variation through batch normalization is not feasible, and therefore these individual sample batches from separate growth conditions are taken as individual tasks for the network inference. Thus, the index ‘d’, below, ranges from 1 to 11 and is an index over the separate datasets corresponding to growth conditions. This separation into tasks results in the joint learning of 11 networks (one for each growth condition), followed by combination into a single global network.</p><p>Briefly, the network model is represented as a matrix <bold><italic>W</italic></bold> for each target gene (where columns are individual single-cell batches <italic>d</italic> and rows are potential regulators <italic>k</italic>) with signed entries corresponding to strength and type of regulation. We then decompose the model coefficient matrix <bold><italic>W</italic></bold> into a dataset-specific component <bold><italic>S</italic></bold> and a conserved component <bold><italic>B</italic></bold> to enable us to penalize dataset-unique and conserved interactions separately for each target gene; this separation captures differences in regulatory networks across datasets. Specifically, we apply an <italic>l<sub>1</sub>/l<sub>∞</sub></italic> penalty to the <bold><italic>B</italic></bold> component to encourage similarity between network models, and an <italic>l<sub>1</sub>/ l<sub>1</sub></italic> penalty to the other to accommodate differences to <bold><italic>S</italic></bold> (<xref ref-type="bibr" rid="bib59">Jalali et al., 2010</xref>). Regularization parameters <inline-formula><mml:math id="inf7"><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf8"><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, representing the strength of each penalty, were chosen via Extended Bayesian Information Criterion (<xref ref-type="bibr" rid="bib19">Chen and Chen, 2008</xref>). We set <inline-formula><mml:math id="inf9"><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to <inline-formula><mml:math id="inf10"><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:msqrt><mml:mfrac><mml:mrow><mml:mi>d</mml:mi> <mml:mi/><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi> <mml:mi/><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:msqrt></mml:math></inline-formula>, where <italic>d</italic> is the number of tasks, <italic>n</italic> is the mean number of samples per task, and <italic>p</italic> is the number of predictors. We then search for <italic>c<sub>b</sub></italic> in the log interval [0.1, 10.0] with 20 steps. We then set <inline-formula><mml:math id="inf11"><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> such that <inline-formula><mml:math id="inf12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>d</mml:mi></mml:mfrac><mml:mo>&lt;</mml:mo><mml:mfrac><mml:msub><mml:mrow><mml:mo>ƛ</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>ƛ</mml:mo></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>&lt;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <italic>d</italic> is the number of tasks and <inline-formula><mml:math id="inf13"><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. We search for <italic>c<sub>s</sub></italic> in the linear interval [<inline-formula><mml:math id="inf14"><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula>+ 0.01, 0.99] with 10 steps.</p><p>We can incorporate prior knowledge by using adaptive weights (<inline-formula><mml:math id="inf15"><mml:mi>ɸ</mml:mi><mml:msub><mml:mrow><mml:mi>ƛ</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) when penalizing different coefficients in the <italic>l<sub>1</sub>/ l<sub>1</sub></italic> penalty (<xref ref-type="bibr" rid="bib129">Zou, 2006</xref>). In this work, however, we chose not to bias predictors to the priors using adaptive weights, and set <inline-formula><mml:math id="inf16"><mml:mi>ɸ</mml:mi></mml:math></inline-formula> to 1. For each gene, we minimize the following function (<xref ref-type="bibr" rid="bib16">Castro et al., 2019</xref>):<disp-formula id="equ3"><label>(3)</label><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mtext> </mml:mtext><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow/></mml:munderover><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mtext> </mml:mtext><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>d</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:msup><mml:mrow><mml:mover><mml:mi>A</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mo>∗</mml:mo><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mo>∗</mml:mo><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>λ</mml:mi></mml:mrow><mml:mi>s</mml:mi></mml:msub><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="normal">Φ</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>λ</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>B</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">∞</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equ4"><mml:math id="m4"><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>:</mml:mo> <mml:mi/><mml:mi>W</mml:mi><mml:mo>=</mml:mo><mml:mi>S</mml:mi> <mml:mi/><mml:mo>+</mml:mo> <mml:mi/><mml:mi>B</mml:mi> <mml:mi/></mml:math></disp-formula></p></sec><sec id="s4-3-8"><title>Ranking interactions and data resampling</title><p>Interactions were ranked by both the overall performance of the model for each gene <italic>i</italic> and the proportion of variance explained by each β<italic><sub>i,k</sub></italic>. The output of this procedure is a matrix <bold><italic>S</italic></bold> where <italic>S<sub>i,k</sub></italic> is the confidence score on the interaction between transcription factor <italic>k</italic> and gene <italic>i</italic>. In order to avoid overfitting and sampling biases, we repeat this procedure <italic>N</italic> times by resampling the input data matrix with replacement. Finally, we rank combine the confidence scores generated by running the above inference procedure on each of the <italic>N</italic> bootstrapped datasets and obtain a final matrix of combined confidence scores for the possible interactions between transcription factors (columns) and genes (rows).</p></sec><sec id="s4-3-9"><title>Network combination</title><p>Individual task networks were assembled into a global network by rank combining the confidence scores generated for each possible interaction between transcription factors and genes, obtaining a final matrix of combined confidence scores for the global network. Global interactions were ordered by combined confidence score, and the top interactions were kept to a threshold defined by precision = 0.5, as determined by recovery of the priors.</p></sec></sec><sec id="s4-4"><title>Statistical analysis and differential gene expression</title><p>To analyze all growth conditions together, the raw single-cell count matrix was normalized using <italic>multiBatchNorm</italic> from the <italic>scater</italic> package in R (<xref ref-type="bibr" rid="bib78">McCarthy et al., 2017</xref>). In short, this calculates size factors that are used to scale cells from different environmental condition batches so that each batch is of approximately the same mean UMI count. Cells were then library size normalized within batches and the normalized data was log-transformed with log<sub>2</sub>(x+1) to give a transformed and normalized count matrix.</p><sec id="s4-4-1"><title>Visualizing single cell expression data</title><p>This normalized count matrix was reduced to 50 principal components by principal component analysis (PCA) with <italic>multiBatchPCA</italic> from the scater package in R. <italic>MultiBatchPCA</italic> is standard PCA with the modification that each environmental condition batch contributes equally to the covariance matrix, even when batches are imbalanced in cell count. These principal components were projected into two dimensional space by Uniform Manifold Approximation and Projection (<italic>UMAP</italic>) (<xref ref-type="bibr" rid="bib79">McInnes et al., 2018</xref>) and plotted.</p><p>To analyze each growth condition separately, the cells corresponding to a growth condition were selected from the raw count matrix, library-size normalized and log<sub>2</sub>(x+1) transformed, and reduced to 50 principal components with PCA. These principal components were projected into two dimensional space by UMAP for plotting, and also used to generate a shared nearest-neighbor (sNN) graph, which is used to cluster cells using the Louvain clustering method. Each growth condition was processed and plotted separately.</p></sec><sec id="s4-4-2"><title>Pseudobulk differential gene expression</title><p>The raw, unmodified UMI counts of all cells from each biological replicate (with the same strain barcode) within a specific environmental growth condition were summed, resulting in 72 samples per condition (six biological replicates for each of the 12 transcription factor deletions). Summed pseudobulk expression data was then tested with DESeq2 (<xref ref-type="bibr" rid="bib73">Love et al., 2014</xref>) for differential gene expression (testing against a null hypothesis of Fold Change &lt; 1.5 and considering changes significant when p&lt;0.05 at a false discovery rate of 0.1) with no additional processing or normalization.</p></sec><sec id="s4-4-3"><title>Gene categorization</title><p>Cell-cycle associated genes are categorized using the Spellman annotations (<xref ref-type="bibr" rid="bib106">Spellman et al., 1998</xref>). Ribosomal genes, ribosomal biogenesis genes, and induced environmental stress response genes are categorized using the Gasch annotations (<xref ref-type="bibr" rid="bib38">Gasch et al., 2017</xref>). Gene category annotations are included as <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>-Supplemental Table 6.</p></sec><sec id="s4-4-4"><title>Gene Ontology</title><p>The number of interactions was determined for each gene and each transcription factor. Interactions are considered Learned (new) if they are present in the learned network and not in the prior network; Learned (In Prior) if they are present in the learned network and not in the prior; and, Not Learned (In Prior) if they are present in the prior and not in the learned network. Each gene was mapped to Gene Ontology (GO) slim terms using the YeastGenome slim mapping (<ext-link ext-link-type="uri" xlink:href="https://downloads.yeastgenome.org/curation/literature/go_slim_mapping.tab">https://downloads.yeastgenome.org/curation/literature/go_slim_mapping.tab</ext-link>), which is a curated gene ontology mapping of high-level, broad GO terms. Interactions for all genes annotated with a GO term were summed. The generic terms ‘biological_process’, ‘not_yet_annotated’, and ‘other’ are removed from both target genes and regulatory transcription factors, and the common term ‘transcription from RNA polymerase II promoter’ was removed from regulatory transcription factors GO annotations. The 25 remaining terms with the highest number of learned (new) interactions were plotted separately for both the target genes and the regulatory transcription factors.</p></sec><sec id="s4-4-5"><title>Correlation plots</title><p>Gene expression data was derived experimentally in this work (FY4/FY5) or obtained from GEO. All samples are from early-log phase growth in YPD. Single-cell yeast data sets are from GSE122392 (BY4741) (<xref ref-type="bibr" rid="bib87">Nadal-Ribelles et al., 2019</xref>) and GSE102475 (BY4741) (<xref ref-type="bibr" rid="bib38">Gasch et al., 2017</xref>). A comparable bulk RNA control is from GSE135430 (BY4741) (<xref ref-type="bibr" rid="bib101">Scholes and Lewis, 2019</xref>). Genes were ranked by expression in each cell or sample. All cells or samples from a specific experiment were rank-combined and ranks were pairwise plotted for each experiment with GGally in R.</p></sec><sec id="s4-4-6"><title>Variability plots</title><p>Coefficient of variation (mean over standard deviation) is calculated for each gene in each growth condition. Pearson residuals (model residual over expected standard deviation) are calculated for each gene in each cell and then the mean of the pearson residuals is taken for each growth condition. This calculation is done with the <italic>vst</italic> function from the sctransform package in R (<xref ref-type="bibr" rid="bib48">Hafemeister and Satija, 2019</xref>). In short, this builds for each gene a regularized negative binomial model, which is then used to calculate pearson residuals for each cell compared to the model. This is done separately for each growth condition.</p></sec></sec><sec id="s4-5"><title>Data and software availability</title><sec id="s4-5-1"><title>Sequencing data</title><p>Raw sequencing data, the output from the cellranger pipeline to count reads, and the output from the fastqToMat0 pipeline to extract and attach genotype metadata to the count matrix are available in NCBI GEO under the accession number GEO: GSE125162.</p></sec><sec id="s4-5-2"><title>Single-Cell processing pipeline</title><p>The cellranger pipeline is available from 10x Genomics under the MIT license (<ext-link ext-link-type="uri" xlink:href="https://github.com/10XGenomics/cellranger">https://github.com/10XGenomics/cellranger</ext-link>). The fastqToMat0 pipeline is available from GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/flatironinstitute/fastqToMat0">https://github.com/flatironinstitute/fastqToMat0</ext-link>; <xref ref-type="bibr" rid="bib57">Jackson, 2020</xref>; copy archived at <ext-link ext-link-type="uri" xlink:href="https://github.com/elifesciences-publications/fastqToMat0">https://github.com/elifesciences-publications/fastqToMat0</ext-link>) and is released under the MIT license. Genome sequence and annotations are included as <xref ref-type="supplementary-material" rid="scode4">Source code 4</xref>.</p></sec><sec id="s4-5-3"><title>Network inference</title><p>The Inferelator is implemented in Python, with dependencies on the widely-distributed scientific packages Numpy (<xref ref-type="bibr" rid="bib113">van der Walt et al., 2011</xref>), Scipy (<xref ref-type="bibr" rid="bib115">Virtanen et al., 2020</xref>), Pandas (<xref ref-type="bibr" rid="bib81">McKinney, 2010</xref>), and Scikit-learn (<xref ref-type="bibr" rid="bib91">Pedregosa et al., 2011</xref>). Scaling to a high-performance computing cluster is implemented with dask (<xref ref-type="bibr" rid="bib95">Rocklin, 2015</xref>). All network inference in this work was performed with the inferelator v0.3.0, using Python v3.7.3, Numpy v1.16.2, Pandas v0.24.2, Scikit-learn v0.20.3, Scipy v1.2.1, and dask v1.1.4. The inferelator package is available under the Simplified BSD licence and can be installed from PyPI (<ext-link ext-link-type="uri" xlink:href="https://pypi.org/project/inferelator/">https://pypi.org/project/inferelator/</ext-link>) or cloned from GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/flatironinstitute/inferelator">https://github.com/flatironinstitute/inferelator</ext-link>; <xref ref-type="bibr" rid="bib56">Jackson and Gibbs, 2020</xref>; copy archived at <ext-link ext-link-type="uri" xlink:href="https://github.com/elifesciences-publications/inferelator">https://github.com/elifesciences-publications/inferelator</ext-link>).</p></sec><sec id="s4-5-4"><title>Figure construction</title><p><xref ref-type="fig" rid="fig1">Figure 1</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref> are constructed using Adobe Illustrator. <xref ref-type="fig" rid="fig2">Figures 2</xref>–<xref ref-type="fig" rid="fig7">7</xref> and accompanying supplementary figures are constructed with R. The R (v3.5.1) (<xref ref-type="bibr" rid="bib94">R Development Core Team, 2018</xref>) packages used are as follows: for plotting, ggplot2 (v3.1.0) (<xref ref-type="bibr" rid="bib118">Wickham, 2016</xref>), cowplot (v0.9.4) (<xref ref-type="bibr" rid="bib123">Wilke, 2019</xref>), ggridges (v0.5.1) (<xref ref-type="bibr" rid="bib122">Wilke, 2018</xref>), ggrastr (v0.1.7) (<xref ref-type="bibr" rid="bib93">Petukhov, 2019</xref>), GGally (v1.4.0) (<xref ref-type="bibr" rid="bib100">Schloerke et al., 2018</xref>), viridis (v0.5.1) (<xref ref-type="bibr" rid="bib36">Garnier, 2018</xref>), RColorBrewer (v1.1–2) (<xref ref-type="bibr" rid="bib90">Neuwirth, 2014</xref>), and scales (v1.0.0) (<xref ref-type="bibr" rid="bib120">Wickham, 2018a</xref>); for data manipulation, dplyr (v0.7.8) (<xref ref-type="bibr" rid="bib119">Wickham et al., 2018</xref>), data.table (v1.12.0) (<xref ref-type="bibr" rid="bib32">Dowle and Srinivasan, 2019</xref>), reshape2 (v1.4.3) (<xref ref-type="bibr" rid="bib117">Wickham, 2007</xref>), and stringr (v1.3.1) (<xref ref-type="bibr" rid="bib121">Wickham, 2018b</xref>); and for single-cell analysis, scater (v1.10.1) (<xref ref-type="bibr" rid="bib78">McCarthy et al., 2017</xref>), scran (v1.10.2) (<xref ref-type="bibr" rid="bib74">Lun et al., 2016</xref>), umap (R: v0.2.0.0, python: v0.3.6) (<xref ref-type="bibr" rid="bib64">Konopka, 2018</xref>), igraph (v1.2.2) (<xref ref-type="bibr" rid="bib26">Csardi and Nepusz, 2006</xref>), DESeq2 (1.22.2) (<xref ref-type="bibr" rid="bib73">Love et al., 2014</xref>), corpcor (v1.6.9) (<xref ref-type="bibr" rid="bib99">Schafer et al., 2017</xref>), and sctransform (v0.2.0) (<xref ref-type="bibr" rid="bib48">Hafemeister and Satija, 2019</xref>). The R scripts to generate these figures and all required data are included with <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>. Network illustrations in <xref ref-type="fig" rid="fig6">Figures 6</xref> and <xref ref-type="fig" rid="fig7">7</xref> were generated using Gephi 0.9.2 from the inferelator output network (gefx formatted); the layouts used are Force Atlas 2, Noverlap and Label Adjust. Figures were minimally modified from R outputs to enhance layout and aesthetics using Adobe Illustrator.</p></sec><sec id="s4-5-5"><title>Interactive figures</title><p>Interactive versions of several panels from <xref ref-type="fig" rid="fig1">Figures 1</xref>–<xref ref-type="fig" rid="fig4">4</xref> are available as Shiny (<xref ref-type="bibr" rid="bib18">Chang et al., 2018</xref>) apps online at <ext-link ext-link-type="uri" xlink:href="http://shiny.bio.nyu.edu/YeastSingleCell2019/">http://shiny.bio.nyu.edu/YeastSingleCell2019/</ext-link>. Source code for the Shiny app is available upon request under the MIT license.</p></sec></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We would like to thank past and present members of the Gresham and Bonneau labs, as well as Christine Vogel’s lab, for discussions and feedback. We thank our undergraduate researchers, especially Juli Miller, for help constructing strains. We thank Tara Rock, Olivia Micci-Smith, and Hana Husic from the NYU Genomics Core facility for troubleshooting suggestions and DNA sequencing services. RB acknowledges support from the Flatiron Institute, the Simons Foundation, the NIH (R01DK103358, R01HD096770, and R01CA229235) and the NSF (IOS1546218). DG is funded by the NIH (R01GM107466) and NSF (MCB1818234).</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Resources, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Software, Formal analysis, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Software, Formal analysis</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Resources, Supervision, Funding acquisition, Project administration, Writing - review and editing</p></fn><fn fn-type="con" id="con5"><p>Conceptualization, Resources, Supervision, Funding acquisition, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="scode1"><label>Source code 1.</label><caption><title>A ‘tar.gz’ archive containing R scripts used to generate <xref ref-type="fig" rid="fig2">Figures 2</xref>–<xref ref-type="fig" rid="fig7">7</xref> and accompanying supplementary figures with a README detailing the necessary R environment to run them locally.</title><p>It also contains a data folder with the raw count matrix as a TSV file (103118_SS_Data.tsv.gz), the simulated negative data count matrix as a TSV file (110518_SS_NEG_Data.tsv.gz), a gene name metadata TSV file (yeast_gene_names.tsv), supplemental tables 5 (STable5.tsv) and 6 (STable6.tsv) as TSV files, and the yeast gene ontology slim mapping as a TAB file (go_slim_mapping.tab). <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref> also contains a priors folder with the Gold Standard, the three sets of priors data tested in this work, and the YEASTRACT comparison data, all as TSV files. <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref> also contains a network folder with the network learned in this paper (signed_network.tsv) as a TSV file, and the networks for each experimental condition (COND_signed_network.tsv) as 11 separate TSV files. <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref> also contains an inferelator folder with the python scripts used to generate the networks for <xref ref-type="fig" rid="fig5">Figures 5</xref>, <xref ref-type="fig" rid="fig6">6</xref>, <xref ref-type="fig" rid="fig7">7</xref>.</p></caption><media mime-subtype="x-gzip" mimetype="application" xlink:href="elife-51254-code1-v3.tar.gz"/></supplementary-material><supplementary-material id="scode2"><label>Source code 2.</label><caption><title>The raw count matrix as a gzipped TSV file.</title><p>This file contains 38,225 observations (cells). Doublets and low-count cells have already been removed; gene expression values are unmodified transcript counts after deartifacting using UMIs (these values are directly produced by the cellranger count pipeline)</p></caption><media mime-subtype="x-gzip" mimetype="application" xlink:href="elife-51254-code2-v3.tsv.gz"/></supplementary-material><supplementary-material id="scode3"><label>Source code 3.</label><caption><title>The network learned in this paper as a TSV file.</title></caption><media mime-subtype="tab-separated-values" mimetype="text" xlink:href="elife-51254-code3-v3.tsv"/></supplementary-material><supplementary-material id="scode4"><label>Source code 4.</label><caption><title>A ‘.tar.gz’ archive containing the sequences used for mapping reads.</title><p>It also contains a FASTA file containing the genotype-specific barcodes (bcdel_1_barcodes.fasta), a FASTA file containing the yeast S288C genome modified with markers (Saccharomyces_cerevisiae.R64-1-1.dna.toplevel.Marker.fa), and a GTF file containing the yeast gene annotations modified to include untranslated regions at the 5’ and 3’ end, and with markers (Saccharomyces_cerevisiae.R64-1-1.Marker.UTR.notRNA.gtf).</p></caption><media mime-subtype="x-gzip" mimetype="application" xlink:href="elife-51254-code4-v3.tar.gz"/></supplementary-material><supplementary-material id="scode5"><label>Source code 5.</label><caption><title>A zipped HTML document containing the raw R output figures for <xref ref-type="fig" rid="fig2">Figures 2</xref>–<xref ref-type="fig" rid="fig7">7</xref> and accompanying supplementary Figures.</title><p>The R markdown file to create this document is contained in <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>.</p></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-51254-code5-v3.zip"/></supplementary-material><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>An excel file containing Supplemental Tables 1-6.</title><p>Supplemental Table 1 contains all primer sequences used in this work. Supplemental Table 2 contains all <italic>Saccharomyces cerevisiae</italic> strains used in this work. Supplemental Table 3 contains all plasmids used in this work. Supplemental Table 4 contains all media formulations used in this work. Supplemental Table 5 contains the source data for modeling performance (as AUPR) that is reported graphically in <xref ref-type="fig" rid="fig5">Figure 5</xref>. Supplemental Table 6 contains the gene categorizations (cell cycle stage, RP, RiBi, etc) used in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-51254-supp1-v3.xlsx"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="pdf" mimetype="application" xlink:href="elife-51254-transrepform-v3.pdf"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>Sequencing data has been deposited in GEO: GSE125162. Figures 2-7 (&amp; supplementary figures) are generated from a single R markdown document. The scripts and all data necessary to do this analysis are provided as Source code 1. The raw output (knit HTML file) is provided as Source code 5. Interactive versions of several figures are available have been made available with the Shiny library in R: <ext-link ext-link-type="uri" xlink:href="http://shiny.bio.nyu.edu/cj59/YeastSingleCell2019/">http://shiny.bio.nyu.edu/cj59/YeastSingleCell2019/</ext-link>. The Inferelator package is available on GitHub and through python package managers (i.e. pip) under an open source license (BSD).</p><p>The following dataset was generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Jackson</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Gene regulatory network reconstruction using single-cell RNA sequencing of barcoded genotypes in diverse environments</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE125162">GSE125162</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation id="dataset2" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Nadal-Ribelles</surname><given-names>M</given-names></name><name><surname>Islam</surname><given-names>S</given-names></name><name><surname>Wei</surname><given-names>W</given-names></name><name><surname>Latorre</surname><given-names>P</given-names></name><name><surname>Steinmetz</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Sensitive, high-throughput single-cell RNA-Seq reveals within-clonal transcript-correlations in yeast populations</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE122392">GSE122392</pub-id></element-citation></p><p><element-citation id="dataset3" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Gasch</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><data-title>Single-cell RNA-seq reveals intrinsic and extrinsic regulatory heterogeneity in yeast responding to stress</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE102475">GSE102475</pub-id></element-citation></p><p><element-citation id="dataset4" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Scholes</surname><given-names>AN</given-names></name><name><surname>Lewis</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Comparison of RNA Isolation Methods in Yeast on RNA-Seq: Implications for Differential Expression and Meta-Analyses</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE135430">GSE135430</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adamson</surname> <given-names>B</given-names></name><name><surname>Norman</surname> <given-names>TM</given-names></name><name><surname>Jost</surname> <given-names>M</given-names></name><name><surname>Cho</surname> <given-names>MY</given-names></name><name><surname>Nuñez</surname> <given-names>JK</given-names></name><name><surname>Chen</surname> <given-names>Y</given-names></name><name><surname>Villalta</surname> <given-names>JE</given-names></name><name><surname>Gilbert</surname> <given-names>LA</given-names></name><name><surname>Horlbeck</surname> <given-names>MA</given-names></name><name><surname>Hein</surname> <given-names>MY</given-names></name><name><surname>Pak</surname> <given-names>RA</given-names></name><name><surname>Gray</surname> <given-names>AN</given-names></name><name><surname>Gross</surname> <given-names>CA</given-names></name><name><surname>Dixit</surname> <given-names>A</given-names></name><name><surname>Parnas</surname> <given-names>O</given-names></name><name><surname>Regev</surname> <given-names>A</given-names></name><name><surname>Weissman</surname> <given-names>JS</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A multiplexed Single-Cell CRISPR screening platform enables systematic dissection of the unfolded protein response</article-title><source>Cell</source><volume>167</volume><fpage>1867</fpage><lpage>1882</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.11.048</pub-id><pub-id pub-id-type="pmid">27984733</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aibar</surname> <given-names>S</given-names></name><name><surname>González-Blas</surname> <given-names>CB</given-names></name><name><surname>Moerman</surname> <given-names>T</given-names></name><name><surname>Huynh-Thu</surname> <given-names>VA</given-names></name><name><surname>Imrichova</surname> <given-names>H</given-names></name><name><surname>Hulselmans</surname> <given-names>G</given-names></name><name><surname>Rambow</surname> <given-names>F</given-names></name><name><surname>Marine</surname> <given-names>JC</given-names></name><name><surname>Geurts</surname> <given-names>P</given-names></name><name><surname>Aerts</surname> <given-names>J</given-names></name><name><surname>van den Oord</surname> <given-names>J</given-names></name><name><surname>Atak</surname> <given-names>ZK</given-names></name><name><surname>Wouters</surname> <given-names>J</given-names></name><name><surname>Aerts</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>SCENIC: single-cell regulatory network inference and clustering</article-title><source>Nature Methods</source><volume>14</volume><fpage>1083</fpage><lpage>1086</lpage><pub-id pub-id-type="doi">10.1038/nmeth.4463</pub-id><pub-id pub-id-type="pmid">28991892</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Airoldi</surname> <given-names>EM</given-names></name><name><surname>Miller</surname> <given-names>D</given-names></name><name><surname>Athanasiadou</surname> <given-names>R</given-names></name><name><surname>Brandt</surname> <given-names>N</given-names></name><name><surname>Abdul-Rahman</surname> <given-names>F</given-names></name><name><surname>Neymotin</surname> <given-names>B</given-names></name><name><surname>Hashimoto</surname> <given-names>T</given-names></name><name><surname>Bahmani</surname> <given-names>T</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Steady-state and dynamic gene expression programs in <italic>Saccharomyces cerevisiae</italic> in response to variation in environmental nitrogen</article-title><source>Molecular Biology of the Cell</source><volume>27</volume><fpage>1383</fpage><lpage>1396</lpage><pub-id pub-id-type="doi">10.1091/mbc.E14-05-1013</pub-id><pub-id pub-id-type="pmid">26941329</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Andréasson</surname> <given-names>C</given-names></name><name><surname>Ljungdahl</surname> <given-names>PO</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Receptor-mediated endoproteolytic activation of two transcription factors in yeast</article-title><source>Genes &amp; Development</source><volume>16</volume><fpage>3158</fpage><lpage>3172</lpage><pub-id pub-id-type="doi">10.1101/gad.239202</pub-id><pub-id pub-id-type="pmid">12502738</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arrieta-Ortiz</surname> <given-names>ML</given-names></name><name><surname>Hafemeister</surname> <given-names>C</given-names></name><name><surname>Bate</surname> <given-names>AR</given-names></name><name><surname>Chu</surname> <given-names>T</given-names></name><name><surname>Greenfield</surname> <given-names>A</given-names></name><name><surname>Shuster</surname> <given-names>B</given-names></name><name><surname>Barry</surname> <given-names>SN</given-names></name><name><surname>Gallitto</surname> <given-names>M</given-names></name><name><surname>Liu</surname> <given-names>B</given-names></name><name><surname>Kacmarczyk</surname> <given-names>T</given-names></name><name><surname>Santoriello</surname> <given-names>F</given-names></name><name><surname>Chen</surname> <given-names>J</given-names></name><name><surname>Rodrigues</surname> <given-names>CD</given-names></name><name><surname>Sato</surname> <given-names>T</given-names></name><name><surname>Rudner</surname> <given-names>DZ</given-names></name><name><surname>Driks</surname> <given-names>A</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name><name><surname>Eichenberger</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>An experimentally supported model of the <italic>Bacillus subtilis</italic> global transcriptional regulatory network</article-title><source>Molecular Systems Biology</source><volume>11</volume><elocation-id>839</elocation-id><pub-id pub-id-type="doi">10.15252/msb.20156236</pub-id><pub-id pub-id-type="pmid">26577401</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Athanasiadou</surname> <given-names>R</given-names></name><name><surname>Neymotin</surname> <given-names>B</given-names></name><name><surname>Brandt</surname> <given-names>N</given-names></name><name><surname>Wang</surname> <given-names>W</given-names></name><name><surname>Christiaen</surname> <given-names>L</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name><name><surname>Tranchina</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A complete statistical model for calibration of RNA-seq counts using external spike-ins and maximum likelihood theory</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1006794</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006794</pub-id><pub-id pub-id-type="pmid">30856174</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Azizi</surname> <given-names>E</given-names></name><name><surname>Carr</surname> <given-names>AJ</given-names></name><name><surname>Plitas</surname> <given-names>G</given-names></name><name><surname>Cornish</surname> <given-names>AE</given-names></name><name><surname>Konopacki</surname> <given-names>C</given-names></name><name><surname>Prabhakaran</surname> <given-names>S</given-names></name><name><surname>Nainys</surname> <given-names>J</given-names></name><name><surname>Wu</surname> <given-names>K</given-names></name><name><surname>Kiseliovas</surname> <given-names>V</given-names></name><name><surname>Setty</surname> <given-names>M</given-names></name><name><surname>Choi</surname> <given-names>K</given-names></name><name><surname>Fromme</surname> <given-names>RM</given-names></name><name><surname>Dao</surname> <given-names>P</given-names></name><name><surname>McKenney</surname> <given-names>PT</given-names></name><name><surname>Wasti</surname> <given-names>RC</given-names></name><name><surname>Kadaveru</surname> <given-names>K</given-names></name><name><surname>Mazutis</surname> <given-names>L</given-names></name><name><surname>Rudensky</surname> <given-names>AY</given-names></name><name><surname>Pe'er</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Single-Cell map of diverse immune phenotypes in the breast tumor microenvironment</article-title><source>Cell</source><volume>174</volume><fpage>1293</fpage><lpage>1308</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.05.060</pub-id><pub-id pub-id-type="pmid">29961579</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barabási</surname> <given-names>AL</given-names></name><name><surname>Gulbahce</surname> <given-names>N</given-names></name><name><surname>Loscalzo</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Network medicine: a network-based approach to human disease</article-title><source>Nature Reviews Genetics</source><volume>12</volume><fpage>56</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1038/nrg2918</pub-id><pub-id pub-id-type="pmid">21164525</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blondel</surname> <given-names>VD</given-names></name><name><surname>Guillaume</surname> <given-names>J-L</given-names></name><name><surname>Lambiotte</surname> <given-names>R</given-names></name><name><surname>Lefebvre</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Fast unfolding of communities in large networks</article-title><source>Journal of Statistical Mechanics: Theory and Experiment</source><volume>2008</volume><elocation-id>P10008</elocation-id><pub-id pub-id-type="doi">10.1088/1742-5468/2008/10/P10008</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bonneau</surname> <given-names>R</given-names></name><name><surname>Reiss</surname> <given-names>DJ</given-names></name><name><surname>Shannon</surname> <given-names>P</given-names></name><name><surname>Facciotti</surname> <given-names>M</given-names></name><name><surname>Hood</surname> <given-names>L</given-names></name><name><surname>Baliga</surname> <given-names>NS</given-names></name><name><surname>Thorsson</surname> <given-names>V</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The inferelator: an algorithm for learning parsimonious regulatory networks from systems-biology data sets de novo</article-title><source>Genome Biology</source><volume>7</volume><elocation-id>R36</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2006-7-5-r36</pub-id><pub-id pub-id-type="pmid">16686963</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brauer</surname> <given-names>MJ</given-names></name><name><surname>Huttenhower</surname> <given-names>C</given-names></name><name><surname>Airoldi</surname> <given-names>EM</given-names></name><name><surname>Rosenstein</surname> <given-names>R</given-names></name><name><surname>Matese</surname> <given-names>JC</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name><name><surname>Boer</surname> <given-names>VM</given-names></name><name><surname>Troyanskaya</surname> <given-names>OG</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Coordination of growth rate, cell cycle, stress response, and metabolic activity in yeast</article-title><source>Molecular Biology of the Cell</source><volume>19</volume><fpage>352</fpage><lpage>367</lpage><pub-id pub-id-type="doi">10.1091/mbc.e07-08-0779</pub-id><pub-id pub-id-type="pmid">17959824</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brennecke</surname> <given-names>P</given-names></name><name><surname>Anders</surname> <given-names>S</given-names></name><name><surname>Kim</surname> <given-names>JK</given-names></name><name><surname>Kołodziejczyk</surname> <given-names>AA</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Proserpio</surname> <given-names>V</given-names></name><name><surname>Baying</surname> <given-names>B</given-names></name><name><surname>Benes</surname> <given-names>V</given-names></name><name><surname>Teichmann</surname> <given-names>SA</given-names></name><name><surname>Marioni</surname> <given-names>JC</given-names></name><name><surname>Heisler</surname> <given-names>MG</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Accounting for technical noise in single-cell RNA-seq experiments</article-title><source>Nature Methods</source><volume>10</volume><fpage>1093</fpage><lpage>1095</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2645</pub-id><pub-id pub-id-type="pmid">24056876</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burnetti</surname> <given-names>AJ</given-names></name><name><surname>Aydin</surname> <given-names>M</given-names></name><name><surname>Buchler</surname> <given-names>NE</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Cell cycle start is coupled to entry into the yeast metabolic cycle across diverse strains and growth rates</article-title><source>Molecular Biology of the Cell</source><volume>27</volume><fpage>64</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1091/mbc.E15-07-0454</pub-id><pub-id pub-id-type="pmid">26538026</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Carmona-Gutierrez</surname> <given-names>D</given-names></name><name><surname>Ruckenstuhl</surname> <given-names>C</given-names></name><name><surname>Bauer</surname> <given-names>MA</given-names></name><name><surname>Eisenberg</surname> <given-names>T</given-names></name><name><surname>Büttner</surname> <given-names>S</given-names></name><name><surname>Madeo</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Cell death in yeast: growing applications of a dying buddy</article-title><source>Cell Death &amp; Differentiation</source><volume>17</volume><fpage>733</fpage><lpage>734</lpage><pub-id pub-id-type="doi">10.1038/cdd.2010.10</pub-id><pub-id pub-id-type="pmid">20383156</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Caruana</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="1998">1998</year><chapter-title>Multitask Learning</chapter-title><person-group person-group-type="editor"><name><surname>Thrun</surname> <given-names>S</given-names></name><name><surname>Pratt</surname> <given-names>L</given-names></name></person-group><source>Learning to Learn</source><publisher-loc>Boston</publisher-loc><publisher-name>Springer</publisher-name><fpage>95</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1007/978-1-4615-5529-2_5</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Castro</surname> <given-names>DM</given-names></name><name><surname>de Veaux</surname> <given-names>NR</given-names></name><name><surname>Miraldi</surname> <given-names>ER</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Multi-study inference of regulatory networks for more accurate models of gene regulation</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1006591</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006591</pub-id><pub-id pub-id-type="pmid">30677040</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>TE</given-names></name><name><surname>Stumpf</surname> <given-names>MPH</given-names></name><name><surname>Babtie</surname> <given-names>AC</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Gene regulatory network inference from Single-Cell data using multivariate information measures</article-title><source>Cell Systems</source><volume>5</volume><fpage>251</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2017.08.014</pub-id><pub-id pub-id-type="pmid">28957658</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>W</given-names></name><name><surname>Cheng</surname> <given-names>J</given-names></name><name><surname>Allaire</surname> <given-names>JJ</given-names></name><name><surname>Xie</surname> <given-names>Y</given-names></name><name><surname>McPherson</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>shiny: Web Application Framework for R</data-title><source>shiny</source><ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/shiny/index.html">https://cran.r-project.org/web/packages/shiny/index.html</ext-link></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J</given-names></name><name><surname>Chen</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Extended bayesian information criteria for model selection with large model spaces</article-title><source>Biometrika</source><volume>95</volume><fpage>759</fpage><lpage>771</lpage><pub-id pub-id-type="doi">10.1093/biomet/asn034</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S</given-names></name><name><surname>Mar</surname> <given-names>JC</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Evaluating methods of inferring gene regulatory networks highlights their lack of performance for single cell gene expression data</article-title><source>BMC Bioinformatics</source><volume>19</volume><elocation-id>232</elocation-id><pub-id pub-id-type="doi">10.1186/s12859-018-2217-z</pub-id><pub-id pub-id-type="pmid">29914350</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>M</given-names></name><name><surname>Zhou</surname> <given-names>X</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>VIPER: variability-preserving imputation for accurate gene expression recovery in single-cell RNA sequencing studies</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>196</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1575-1</pub-id><pub-id pub-id-type="pmid">30419955</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chomczynski</surname> <given-names>P</given-names></name><name><surname>Sacchi</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The single-step method of RNA isolation by acid guanidinium thiocyanate-phenol-chloroform extraction: twenty-something years on</article-title><source>Nature Protocols</source><volume>1</volume><fpage>581</fpage><lpage>585</lpage><pub-id pub-id-type="doi">10.1038/nprot.2006.83</pub-id><pub-id pub-id-type="pmid">17406285</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ciofani</surname> <given-names>M</given-names></name><name><surname>Madar</surname> <given-names>A</given-names></name><name><surname>Galan</surname> <given-names>C</given-names></name><name><surname>Sellars</surname> <given-names>M</given-names></name><name><surname>Mace</surname> <given-names>K</given-names></name><name><surname>Pauli</surname> <given-names>F</given-names></name><name><surname>Agarwal</surname> <given-names>A</given-names></name><name><surname>Huang</surname> <given-names>W</given-names></name><name><surname>Parkhurst</surname> <given-names>CN</given-names></name><name><surname>Muratet</surname> <given-names>M</given-names></name><name><surname>Newberry</surname> <given-names>KM</given-names></name><name><surname>Meadows</surname> <given-names>S</given-names></name><name><surname>Greenfield</surname> <given-names>A</given-names></name><name><surname>Yang</surname> <given-names>Y</given-names></name><name><surname>Jain</surname> <given-names>P</given-names></name><name><surname>Kirigin</surname> <given-names>FK</given-names></name><name><surname>Birchmeier</surname> <given-names>C</given-names></name><name><surname>Wagner</surname> <given-names>EF</given-names></name><name><surname>Murphy</surname> <given-names>KM</given-names></name><name><surname>Myers</surname> <given-names>RM</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name><name><surname>Littman</surname> <given-names>DR</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A validated regulatory network for Th17 cell specification</article-title><source>Cell</source><volume>151</volume><fpage>289</fpage><lpage>303</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2012.09.016</pub-id><pub-id pub-id-type="pmid">23021777</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Colman-Lerner</surname> <given-names>A</given-names></name><name><surname>Chin</surname> <given-names>TE</given-names></name><name><surname>Brent</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Yeast Cbk1 and Mob2 activate daughter-specific genetic programs to induce asymmetric cell fates</article-title><source>Cell</source><volume>107</volume><fpage>739</fpage><lpage>750</lpage><pub-id pub-id-type="doi">10.1016/S0092-8674(01)00596-7</pub-id><pub-id pub-id-type="pmid">11747810</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cox</surname> <given-names>KH</given-names></name><name><surname>Rai</surname> <given-names>R</given-names></name><name><surname>Distler</surname> <given-names>M</given-names></name><name><surname>Daugherty</surname> <given-names>JR</given-names></name><name><surname>Coffman</surname> <given-names>JA</given-names></name><name><surname>Cooper</surname> <given-names>TG</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title><italic>Saccharomyces cerevisiae</italic> GATA sequences function as TATA elements during nitrogen catabolite repression and when Gln3p is excluded from the nucleus by overproduction of Ure2p</article-title><source>Journal of Biological Chemistry</source><volume>275</volume><fpage>17611</fpage><lpage>17618</lpage><pub-id pub-id-type="doi">10.1074/jbc.M001648200</pub-id><pub-id pub-id-type="pmid">10748041</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Csardi</surname> <given-names>G</given-names></name><name><surname>Nepusz</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The igraph software package for complex network research <italic>InterJournal</italic></article-title><source>Complex Systems</source><volume>1695</volume></element-citation></ref><ref id="bib27"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Davidson</surname> <given-names>EH</given-names></name></person-group><year iso-8601-date="2012">2012</year><source>Gene Activity in Early Development</source><publisher-name>Elsevier</publisher-name><pub-id pub-id-type="doi">10.1016/B978-0-12-205160-9.X5001-5</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>de Boer</surname> <given-names>CG</given-names></name><name><surname>Hughes</surname> <given-names>TR</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>YeTFaSCo: a database of evaluated yeast transcription factor sequence specificities</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>D169</fpage><lpage>D179</lpage><pub-id pub-id-type="doi">10.1093/nar/gkr993</pub-id><pub-id pub-id-type="pmid">22102575</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>DeRisi</surname> <given-names>JL</given-names></name><name><surname>Iyer</surname> <given-names>VR</given-names></name><name><surname>Brown</surname> <given-names>PO</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Exploring the metabolic and genetic control of gene expression on a genomic scale</article-title><source>Science</source><volume>278</volume><fpage>680</fpage><lpage>686</lpage><pub-id pub-id-type="doi">10.1126/science.278.5338.680</pub-id><pub-id pub-id-type="pmid">9381177</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Didion</surname> <given-names>T</given-names></name><name><surname>Regenberg</surname> <given-names>B</given-names></name><name><surname>Jørgensen</surname> <given-names>MU</given-names></name><name><surname>Kielland-Brandt</surname> <given-names>MC</given-names></name><name><surname>Andersen</surname> <given-names>HA</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>The permease homologue Ssy1p controls the expression of amino acid and peptide transporter genes in <italic>Saccharomyces cerevisiae</italic></article-title><source>Molecular Microbiology</source><volume>27</volume><fpage>643</fpage><lpage>650</lpage><pub-id pub-id-type="doi">10.1046/j.1365-2958.1998.00714.x</pub-id><pub-id pub-id-type="pmid">9489675</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dixit</surname> <given-names>A</given-names></name><name><surname>Parnas</surname> <given-names>O</given-names></name><name><surname>Li</surname> <given-names>B</given-names></name><name><surname>Chen</surname> <given-names>J</given-names></name><name><surname>Fulco</surname> <given-names>CP</given-names></name><name><surname>Jerby-Arnon</surname> <given-names>L</given-names></name><name><surname>Marjanovic</surname> <given-names>ND</given-names></name><name><surname>Dionne</surname> <given-names>D</given-names></name><name><surname>Burks</surname> <given-names>T</given-names></name><name><surname>Raychowdhury</surname> <given-names>R</given-names></name><name><surname>Adamson</surname> <given-names>B</given-names></name><name><surname>Norman</surname> <given-names>TM</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Weissman</surname> <given-names>JS</given-names></name><name><surname>Friedman</surname> <given-names>N</given-names></name><name><surname>Regev</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Perturb-Seq: dissecting molecular circuits with scalable Single-Cell RNA profiling of pooled genetic screens</article-title><source>Cell</source><volume>167</volume><fpage>1853</fpage><lpage>1866</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.11.038</pub-id><pub-id pub-id-type="pmid">27984732</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Dowle</surname> <given-names>M</given-names></name><name><surname>Srinivasan</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>data.table: Extension of `data.frame`</data-title></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eriksson</surname> <given-names>PR</given-names></name><name><surname>Ganguli</surname> <given-names>D</given-names></name><name><surname>Nagarajavel</surname> <given-names>V</given-names></name><name><surname>Clark</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Regulation of histone gene expression in budding yeast</article-title><source>Genetics</source><volume>191</volume><fpage>7</fpage><lpage>20</lpage><pub-id pub-id-type="doi">10.1534/genetics.112.140145</pub-id><pub-id pub-id-type="pmid">22555441</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Filtz</surname> <given-names>TM</given-names></name><name><surname>Vogel</surname> <given-names>WK</given-names></name><name><surname>Leid</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Regulation of transcription factor activity by interconnected post-translational modifications</article-title><source>Trends in Pharmacological Sciences</source><volume>35</volume><fpage>76</fpage><lpage>85</lpage><pub-id pub-id-type="doi">10.1016/j.tips.2013.11.005</pub-id><pub-id pub-id-type="pmid">24388790</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>Y</given-names></name><name><surname>Jarboe</surname> <given-names>LR</given-names></name><name><surname>Dickerson</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Reconstructing genome-wide regulatory network of <italic>E. coli</italic> using transcriptome data and predicted transcription factor activities</article-title><source>BMC Bioinformatics</source><volume>12</volume><elocation-id>233</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-12-233</pub-id><pub-id pub-id-type="pmid">21668997</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Garnier</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>viridis: Default Color Maps from matplotlib</data-title></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gasch</surname> <given-names>AP</given-names></name><name><surname>Spellman</surname> <given-names>PT</given-names></name><name><surname>Kao</surname> <given-names>CM</given-names></name><name><surname>Carmel-Harel</surname> <given-names>O</given-names></name><name><surname>Eisen</surname> <given-names>MB</given-names></name><name><surname>Storz</surname> <given-names>G</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name><name><surname>Brown</surname> <given-names>PO</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Genomic expression programs in the response of yeast cells to environmental changes</article-title><source>Molecular Biology of the Cell</source><volume>11</volume><fpage>4241</fpage><lpage>4257</lpage><pub-id pub-id-type="doi">10.1091/mbc.11.12.4241</pub-id><pub-id pub-id-type="pmid">11102521</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gasch</surname> <given-names>AP</given-names></name><name><surname>Yu</surname> <given-names>FB</given-names></name><name><surname>Hose</surname> <given-names>J</given-names></name><name><surname>Escalante</surname> <given-names>LE</given-names></name><name><surname>Place</surname> <given-names>M</given-names></name><name><surname>Bacher</surname> <given-names>R</given-names></name><name><surname>Kanbar</surname> <given-names>J</given-names></name><name><surname>Ciobanu</surname> <given-names>D</given-names></name><name><surname>Sandor</surname> <given-names>L</given-names></name><name><surname>Grigoriev</surname> <given-names>IV</given-names></name><name><surname>Kendziorski</surname> <given-names>C</given-names></name><name><surname>Quake</surname> <given-names>SR</given-names></name><name><surname>McClean</surname> <given-names>MN</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Single-cell RNA sequencing reveals intrinsic and extrinsic regulatory heterogeneity in yeast responding to stress</article-title><source>PLOS Biology</source><volume>15</volume><elocation-id>e2004050</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.2004050</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Georis</surname> <given-names>I</given-names></name><name><surname>Feller</surname> <given-names>A</given-names></name><name><surname>Vierendeels</surname> <given-names>F</given-names></name><name><surname>Dubois</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The yeast GATA factor Gat1 occupies a central position in nitrogen catabolite repression-sensitive gene activation</article-title><source>Molecular and Cellular Biology</source><volume>29</volume><fpage>3803</fpage><lpage>3815</lpage><pub-id pub-id-type="doi">10.1128/MCB.00399-09</pub-id><pub-id pub-id-type="pmid">19380492</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giaever</surname> <given-names>G</given-names></name><name><surname>Chu</surname> <given-names>AM</given-names></name><name><surname>Ni</surname> <given-names>L</given-names></name><name><surname>Connelly</surname> <given-names>C</given-names></name><name><surname>Riles</surname> <given-names>L</given-names></name><name><surname>Véronneau</surname> <given-names>S</given-names></name><name><surname>Dow</surname> <given-names>S</given-names></name><name><surname>Lucau-Danila</surname> <given-names>A</given-names></name><name><surname>Anderson</surname> <given-names>K</given-names></name><name><surname>André</surname> <given-names>B</given-names></name><name><surname>Arkin</surname> <given-names>AP</given-names></name><name><surname>Astromoff</surname> <given-names>A</given-names></name><name><surname>El-Bakkoury</surname> <given-names>M</given-names></name><name><surname>Bangham</surname> <given-names>R</given-names></name><name><surname>Benito</surname> <given-names>R</given-names></name><name><surname>Brachat</surname> <given-names>S</given-names></name><name><surname>Campanaro</surname> <given-names>S</given-names></name><name><surname>Curtiss</surname> <given-names>M</given-names></name><name><surname>Davis</surname> <given-names>K</given-names></name><name><surname>Deutschbauer</surname> <given-names>A</given-names></name><name><surname>Entian</surname> <given-names>KD</given-names></name><name><surname>Flaherty</surname> <given-names>P</given-names></name><name><surname>Foury</surname> <given-names>F</given-names></name><name><surname>Garfinkel</surname> <given-names>DJ</given-names></name><name><surname>Gerstein</surname> <given-names>M</given-names></name><name><surname>Gotte</surname> <given-names>D</given-names></name><name><surname>Güldener</surname> <given-names>U</given-names></name><name><surname>Hegemann</surname> <given-names>JH</given-names></name><name><surname>Hempel</surname> <given-names>S</given-names></name><name><surname>Herman</surname> <given-names>Z</given-names></name><name><surname>Jaramillo</surname> <given-names>DF</given-names></name><name><surname>Kelly</surname> <given-names>DE</given-names></name><name><surname>Kelly</surname> <given-names>SL</given-names></name><name><surname>Kötter</surname> <given-names>P</given-names></name><name><surname>LaBonte</surname> <given-names>D</given-names></name><name><surname>Lamb</surname> <given-names>DC</given-names></name><name><surname>Lan</surname> <given-names>N</given-names></name><name><surname>Liang</surname> <given-names>H</given-names></name><name><surname>Liao</surname> <given-names>H</given-names></name><name><surname>Liu</surname> <given-names>L</given-names></name><name><surname>Luo</surname> <given-names>C</given-names></name><name><surname>Lussier</surname> <given-names>M</given-names></name><name><surname>Mao</surname> <given-names>R</given-names></name><name><surname>Menard</surname> <given-names>P</given-names></name><name><surname>Ooi</surname> <given-names>SL</given-names></name><name><surname>Revuelta</surname> <given-names>JL</given-names></name><name><surname>Roberts</surname> <given-names>CJ</given-names></name><name><surname>Rose</surname> <given-names>M</given-names></name><name><surname>Ross-Macdonald</surname> <given-names>P</given-names></name><name><surname>Scherens</surname> <given-names>B</given-names></name><name><surname>Schimmack</surname> <given-names>G</given-names></name><name><surname>Shafer</surname> <given-names>B</given-names></name><name><surname>Shoemaker</surname> <given-names>DD</given-names></name><name><surname>Sookhai-Mahadeo</surname> <given-names>S</given-names></name><name><surname>Storms</surname> <given-names>RK</given-names></name><name><surname>Strathern</surname> <given-names>JN</given-names></name><name><surname>Valle</surname> <given-names>G</given-names></name><name><surname>Voet</surname> <given-names>M</given-names></name><name><surname>Volckaert</surname> <given-names>G</given-names></name><name><surname>Wang</surname> <given-names>CY</given-names></name><name><surname>Ward</surname> <given-names>TR</given-names></name><name><surname>Wilhelmy</surname> <given-names>J</given-names></name><name><surname>Winzeler</surname> <given-names>EA</given-names></name><name><surname>Yang</surname> <given-names>Y</given-names></name><name><surname>Yen</surname> <given-names>G</given-names></name><name><surname>Youngman</surname> <given-names>E</given-names></name><name><surname>Yu</surname> <given-names>K</given-names></name><name><surname>Bussey</surname> <given-names>H</given-names></name><name><surname>Boeke</surname> <given-names>JD</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name><name><surname>Philippsen</surname> <given-names>P</given-names></name><name><surname>Davis</surname> <given-names>RW</given-names></name><name><surname>Johnston</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Functional profiling of the <italic>Saccharomyces cerevisiae</italic> genome</article-title><source>Nature</source><volume>418</volume><fpage>387</fpage><lpage>391</lpage><pub-id pub-id-type="doi">10.1038/nature00935</pub-id><pub-id pub-id-type="pmid">12140549</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gietz</surname> <given-names>RD</given-names></name><name><surname>Schiestl</surname> <given-names>RH</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>High-efficiency yeast transformation using the LiAc/SS carrier DNA/PEG method</article-title><source>Nature Protocols</source><volume>2</volume><fpage>31</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1038/nprot.2007.13</pub-id><pub-id pub-id-type="pmid">17401334</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Godard</surname> <given-names>P</given-names></name><name><surname>Urrestarazu</surname> <given-names>A</given-names></name><name><surname>Vissers</surname> <given-names>S</given-names></name><name><surname>Kontos</surname> <given-names>K</given-names></name><name><surname>Bontempi</surname> <given-names>G</given-names></name><name><surname>van Helden</surname> <given-names>J</given-names></name><name><surname>Andre</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Effect of 21 different nitrogen sources on global gene expression in the yeast <italic>Saccharomyces cerevisiae</italic></article-title><source>Molecular and Cellular Biology</source><volume>27</volume><fpage>3065</fpage><lpage>3086</lpage><pub-id pub-id-type="doi">10.1128/MCB.01084-06</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>González</surname> <given-names>A</given-names></name><name><surname>Hall</surname> <given-names>MN</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Nutrient sensing and TOR signaling in yeast and mammals</article-title><source>The EMBO Journal</source><volume>36</volume><fpage>397</fpage><lpage>408</lpage><pub-id pub-id-type="doi">10.15252/embj.201696010</pub-id><pub-id pub-id-type="pmid">28096180</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gray</surname> <given-names>JV</given-names></name><name><surname>Petsko</surname> <given-names>GA</given-names></name><name><surname>Johnston</surname> <given-names>GC</given-names></name><name><surname>Ringe</surname> <given-names>D</given-names></name><name><surname>Singer</surname> <given-names>RA</given-names></name><name><surname>Werner-Washburne</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>&quot;Sleeping Beauty&quot;: Quiescence in <italic>Saccharomyces cerevisiae</italic></article-title><source>Microbiology and Molecular Biology Reviews</source><volume>68</volume><fpage>187</fpage><lpage>206</lpage><pub-id pub-id-type="doi">10.1128/MMBR.68.2.187-206.2004</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greenfield</surname> <given-names>A</given-names></name><name><surname>Madar</surname> <given-names>A</given-names></name><name><surname>Ostrer</surname> <given-names>H</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>DREAM4: combining genetic and dynamic information to identify biological networks and dynamical models</article-title><source>PLOS ONE</source><volume>5</volume><elocation-id>e13397</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0013397</pub-id><pub-id pub-id-type="pmid">21049040</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greenfield</surname> <given-names>A</given-names></name><name><surname>Hafemeister</surname> <given-names>C</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Robust data-driven incorporation of prior knowledge into the inference of dynamic regulatory networks</article-title><source>Bioinformatics</source><volume>29</volume><fpage>1060</fpage><lpage>1067</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btt099</pub-id><pub-id pub-id-type="pmid">23525069</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grün</surname> <given-names>D</given-names></name><name><surname>Kester</surname> <given-names>L</given-names></name><name><surname>van Oudenaarden</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Validation of noise models for single-cell transcriptomics</article-title><source>Nature Methods</source><volume>11</volume><fpage>637</fpage><lpage>640</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2930</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hafemeister</surname> <given-names>C</given-names></name><name><surname>Satija</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Normalization and variance stabilization of single-cell RNA-seq data using regularized negative binomial regression</article-title><source>Genome Biology</source><volume>20</volume><elocation-id>296</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-019-1874-1</pub-id><pub-id pub-id-type="pmid">31870423</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Heitman</surname> <given-names>J</given-names></name><name><surname>Movva</surname> <given-names>NR</given-names></name><name><surname>Hall</surname> <given-names>MN</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Targets for cell cycle arrest by the immunosuppressant rapamycin in yeast</article-title><source>Science</source><volume>253</volume><fpage>905</fpage><lpage>909</lpage><pub-id pub-id-type="doi">10.1126/science.1715094</pub-id><pub-id pub-id-type="pmid">1715094</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hicks</surname> <given-names>SC</given-names></name><name><surname>Townes</surname> <given-names>FW</given-names></name><name><surname>Teng</surname> <given-names>M</given-names></name><name><surname>Irizarry</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Missing data and technical variability in single-cell RNA-sequencing experiments</article-title><source>Biostatistics</source><volume>19</volume><fpage>562</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.1093/biostatistics/kxx053</pub-id><pub-id pub-id-type="pmid">29121214</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hinnebusch</surname> <given-names>AG</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Translational regulation of <italic>GCN4</italic> and the general amino acid control of yeast</article-title><source>Annual Review of Microbiology</source><volume>59</volume><fpage>407</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1146/annurev.micro.59.031805.133833</pub-id><pub-id pub-id-type="pmid">16153175</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hofman-Bang</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Nitrogen catabolite repression in <italic>Saccharomyces cerevisiae</italic></article-title><source>Molecular Biotechnology</source><volume>12</volume><fpage>35</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1385/MB:12:1:35</pub-id><pub-id pub-id-type="pmid">10554772</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>JX</given-names></name><name><surname>Thomas</surname> <given-names>CE</given-names></name><name><surname>Brunak</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Network biology concepts in complex disease comorbidities</article-title><source>Nature Reviews. Genetics</source><volume>17</volume><fpage>615</fpage><lpage>629</lpage><pub-id pub-id-type="doi">10.1038/nrg.2016.87</pub-id><pub-id pub-id-type="pmid">27498692</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hwang</surname> <given-names>B</given-names></name><name><surname>Lee</surname> <given-names>JH</given-names></name><name><surname>Bang</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Single-cell RNA sequencing technologies and bioinformatics pipelines</article-title><source>Experimental &amp; Molecular Medicine</source><volume>50</volume><elocation-id>96</elocation-id><pub-id pub-id-type="doi">10.1038/s12276-018-0071-8</pub-id><pub-id pub-id-type="pmid">30089861</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Iraqui</surname> <given-names>I</given-names></name><name><surname>Vissers</surname> <given-names>S</given-names></name><name><surname>Bernard</surname> <given-names>F</given-names></name><name><surname>de Craene</surname> <given-names>JO</given-names></name><name><surname>Boles</surname> <given-names>E</given-names></name><name><surname>Urrestarazu</surname> <given-names>A</given-names></name><name><surname>André</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Amino acid signaling in <italic>Saccharomyces cerevisiae</italic>: a permease-like sensor of external amino acids and F-Box protein Grr1p are required for transcriptional induction of the <italic>AGP1</italic> gene, which encodes a broad-specificity amino acid permease</article-title><source>Molecular and Cellular Biology</source><volume>19</volume><fpage>989</fpage><lpage>1001</lpage><pub-id pub-id-type="doi">10.1128/MCB.19.2.989</pub-id><pub-id pub-id-type="pmid">9891035</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Jackson</surname> <given-names>CA</given-names></name><name><surname>Gibbs</surname> <given-names>CS</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Inferelator</source><version designator="0.3.0">v0.3.0</version><publisher-name>PyPi</publisher-name><ext-link ext-link-type="uri" xlink:href="https://pypi.org/project/inferelator/">https://pypi.org/project/inferelator/</ext-link></element-citation></ref><ref id="bib57"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Jackson</surname> <given-names>CA</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>Tools for extracting and mapping transcriptional barcodes from single-cell sequencing reads</data-title><source>fastqToMat0</source><version designator="923037c">923037c</version><publisher-name>GitHub</publisher-name><ext-link ext-link-type="uri" xlink:href="https://github.com/flatironinstitute/fastqToMat0">https://github.com/flatironinstitute/fastqToMat0</ext-link></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jaitin</surname> <given-names>DA</given-names></name><name><surname>Weiner</surname> <given-names>A</given-names></name><name><surname>Yofe</surname> <given-names>I</given-names></name><name><surname>Lara-Astiaso</surname> <given-names>D</given-names></name><name><surname>Keren-Shaul</surname> <given-names>H</given-names></name><name><surname>David</surname> <given-names>E</given-names></name><name><surname>Salame</surname> <given-names>TM</given-names></name><name><surname>Tanay</surname> <given-names>A</given-names></name><name><surname>van Oudenaarden</surname> <given-names>A</given-names></name><name><surname>Amit</surname> <given-names>I</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Dissecting immune circuits by linking CRISPR-Pooled screens with Single-Cell RNA-Seq</article-title><source>Cell</source><volume>167</volume><fpage>1883</fpage><lpage>1896</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.11.039</pub-id><pub-id pub-id-type="pmid">27984734</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Jalali</surname> <given-names>A</given-names></name><name><surname>Sanghavi</surname> <given-names>S</given-names></name><name><surname>Ruan</surname> <given-names>C</given-names></name><name><surname>Ravikumar</surname> <given-names>PK</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A dirty model for Multi-task learning</article-title><conf-name>Advances in Neural Information Processing Systems</conf-name><fpage>964</fpage><lpage>972</lpage><ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/paper/4125-a-dirty-model-for-multi-task-learning">https://papers.nips.cc/paper/4125-a-dirty-model-for-multi-task-learning</ext-link></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Y</given-names></name><name><surname>Rothermel</surname> <given-names>B</given-names></name><name><surname>Thornton</surname> <given-names>J</given-names></name><name><surname>Butow</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>A basic helix-loop-helix-leucine zipper transcription complex in yeast functions in a signaling pathway from mitochondria to the nucleus</article-title><source>Molecular and Cellular Biology</source><volume>17</volume><fpage>1110</fpage><lpage>1117</lpage><pub-id pub-id-type="doi">10.1128/MCB.17.3.1110</pub-id><pub-id pub-id-type="pmid">9032238</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johnston</surname> <given-names>G</given-names></name><name><surname>Pringle</surname> <given-names>J</given-names></name><name><surname>Hartwell</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="1977">1977</year><article-title>Coordination of growth with cell division in the yeast</article-title><source>Experimental Cell Research</source><volume>105</volume><fpage>79</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1016/0014-4827(77)90154-9</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kivioja</surname> <given-names>T</given-names></name><name><surname>Vähärautio</surname> <given-names>A</given-names></name><name><surname>Karlsson</surname> <given-names>K</given-names></name><name><surname>Bonke</surname> <given-names>M</given-names></name><name><surname>Enge</surname> <given-names>M</given-names></name><name><surname>Linnarsson</surname> <given-names>S</given-names></name><name><surname>Taipale</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Counting absolute numbers of molecules using unique molecular identifiers</article-title><source>Nature Methods</source><volume>9</volume><fpage>72</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1778</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Komeili</surname> <given-names>A</given-names></name><name><surname>Wedaman</surname> <given-names>KP</given-names></name><name><surname>O'Shea</surname> <given-names>EK</given-names></name><name><surname>Powers</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Mechanism of metabolic control. target of rapamycin signaling links nitrogen quality to the activity of the Rtg1 and Rtg3 transcription factors</article-title><source>The Journal of Cell Biology</source><volume>151</volume><fpage>863</fpage><lpage>878</lpage><pub-id pub-id-type="doi">10.1083/jcb.151.4.863</pub-id><pub-id pub-id-type="pmid">11076970</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Konopka</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>umap: Uniform Manifold Approximation and Projection</data-title></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lahtvee</surname> <given-names>PJ</given-names></name><name><surname>Sánchez</surname> <given-names>BJ</given-names></name><name><surname>Smialowska</surname> <given-names>A</given-names></name><name><surname>Kasvandik</surname> <given-names>S</given-names></name><name><surname>Elsemman</surname> <given-names>IE</given-names></name><name><surname>Gatto</surname> <given-names>F</given-names></name><name><surname>Nielsen</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Absolute quantification of protein and mRNA abundances demonstrate variability in Gene-Specific translation efficiency in yeast</article-title><source>Cell Systems</source><volume>4</volume><fpage>495</fpage><lpage>504</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2017.03.003</pub-id><pub-id pub-id-type="pmid">28365149</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lam</surname> <given-names>KY</given-names></name><name><surname>Westrick</surname> <given-names>ZM</given-names></name><name><surname>Müller</surname> <given-names>CL</given-names></name><name><surname>Christiaen</surname> <given-names>L</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Fused regression for Multi-source gene regulatory network inference</article-title><source>PLOS Computational Biology</source><volume>12</volume><elocation-id>e1005157</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1005157</pub-id><pub-id pub-id-type="pmid">27923054</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leek</surname> <given-names>JT</given-names></name><name><surname>Scharpf</surname> <given-names>RB</given-names></name><name><surname>Bravo</surname> <given-names>HC</given-names></name><name><surname>Simcha</surname> <given-names>D</given-names></name><name><surname>Langmead</surname> <given-names>B</given-names></name><name><surname>Johnson</surname> <given-names>WE</given-names></name><name><surname>Geman</surname> <given-names>D</given-names></name><name><surname>Baggerly</surname> <given-names>K</given-names></name><name><surname>Irizarry</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Tackling the widespread and critical impact of batch effects in high-throughput data</article-title><source>Nature Reviews Genetics</source><volume>11</volume><fpage>733</fpage><lpage>739</lpage><pub-id pub-id-type="doi">10.1038/nrg2825</pub-id><pub-id pub-id-type="pmid">20838408</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Levy</surname> <given-names>SF</given-names></name><name><surname>Ziv</surname> <given-names>N</given-names></name><name><surname>Siegal</surname> <given-names>ML</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Bet hedging in yeast by heterogeneous, age-correlated expression of a stress protectant</article-title><source>PLOS Biology</source><volume>10</volume><elocation-id>e1001325</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.1001325</pub-id><pub-id pub-id-type="pmid">22589700</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>WV</given-names></name><name><surname>Li</surname> <given-names>JJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>An accurate and robust imputation method scImpute for single-cell RNA-seq data</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>997</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-03405-7</pub-id><pub-id pub-id-type="pmid">29520097</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>X</given-names></name><name><surname>Butow</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="1993">1993</year><article-title>RTG1 and RTG2: two yeast genes required for a novel path of communication from mitochondria to the nucleus</article-title><source>Cell</source><volume>72</volume><fpage>61</fpage><lpage>71</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(93)90050-Z</pub-id><pub-id pub-id-type="pmid">8422683</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ljungdahl</surname> <given-names>PO</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Amino-acid-induced signalling via the SPS-sensing pathway in yeast</article-title><source>Biochemical Society Transactions</source><volume>37</volume><fpage>242</fpage><lpage>247</lpage><pub-id pub-id-type="doi">10.1042/BST0370242</pub-id><pub-id pub-id-type="pmid">19143640</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Loewith</surname> <given-names>R</given-names></name><name><surname>Hall</surname> <given-names>MN</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Target of rapamycin (TOR) in nutrient signaling and growth control</article-title><source>Genetics</source><volume>189</volume><fpage>1177</fpage><lpage>1201</lpage><pub-id pub-id-type="doi">10.1534/genetics.111.133363</pub-id><pub-id pub-id-type="pmid">22174183</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Love</surname> <given-names>MI</given-names></name><name><surname>Huber</surname> <given-names>W</given-names></name><name><surname>Anders</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title><source>Genome Biology</source><volume>15</volume><elocation-id>550</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id><pub-id pub-id-type="pmid">25516281</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lun</surname> <given-names>AT</given-names></name><name><surname>McCarthy</surname> <given-names>DJ</given-names></name><name><surname>Marioni</surname> <given-names>JC</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A step-by-step workflow for low-level analysis of single-cell RNA-seq data with bioconductor</article-title><source>F1000Research</source><volume>5</volume><elocation-id>2122</elocation-id><pub-id pub-id-type="doi">10.12688/f1000research.9501.2</pub-id><pub-id pub-id-type="pmid">27909575</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>S</given-names></name><name><surname>Kemmeren</surname> <given-names>P</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name><name><surname>Statnikov</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>De-novo learning of genome-scale regulatory networks in <italic>S. cerevisiae</italic></article-title><source>PLOS ONE</source><volume>9</volume><elocation-id>e106479</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0106479</pub-id><pub-id pub-id-type="pmid">25215507</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Macosko</surname> <given-names>EZ</given-names></name><name><surname>Basu</surname> <given-names>A</given-names></name><name><surname>Satija</surname> <given-names>R</given-names></name><name><surname>Nemesh</surname> <given-names>J</given-names></name><name><surname>Shekhar</surname> <given-names>K</given-names></name><name><surname>Goldman</surname> <given-names>M</given-names></name><name><surname>Tirosh</surname> <given-names>I</given-names></name><name><surname>Bialas</surname> <given-names>AR</given-names></name><name><surname>Kamitaki</surname> <given-names>N</given-names></name><name><surname>Martersteck</surname> <given-names>EM</given-names></name><name><surname>Trombetta</surname> <given-names>JJ</given-names></name><name><surname>Weitz</surname> <given-names>DA</given-names></name><name><surname>Sanes</surname> <given-names>JR</given-names></name><name><surname>Shalek</surname> <given-names>AK</given-names></name><name><surname>Regev</surname> <given-names>A</given-names></name><name><surname>McCarroll</surname> <given-names>SA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Highly parallel Genome-wide expression profiling of individual cells using nanoliter droplets</article-title><source>Cell</source><volume>161</volume><fpage>1202</fpage><lpage>1214</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2015.05.002</pub-id><pub-id pub-id-type="pmid">26000488</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Madar</surname> <given-names>A</given-names></name><name><surname>Greenfield</surname> <given-names>A</given-names></name><name><surname>Vanden-Eijnden</surname> <given-names>E</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>DREAM3: network inference using dynamic context likelihood of relatedness and the inferelator</article-title><source>PLOS ONE</source><volume>5</volume><elocation-id>e9803</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0009803</pub-id><pub-id pub-id-type="pmid">20339551</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McCarthy</surname> <given-names>DJ</given-names></name><name><surname>Campbell</surname> <given-names>KR</given-names></name><name><surname>Lun</surname> <given-names>ATL</given-names></name><name><surname>Wills</surname> <given-names>QF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Scater: pre-processing, quality control, normalization and visualization of single-cell RNA-seq data in R</article-title><source>Bioinformatics</source><volume>247</volume><fpage>777</fpage><lpage>1186</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btw777</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>McInnes</surname> <given-names>L</given-names></name><name><surname>Healy</surname> <given-names>J</given-names></name><name><surname>Melville</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>UMAP: uniform manifold approximation and projection for dimension reduction</article-title><source>arXiv</source><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1802.03426">https://arxiv.org/abs/1802.03426</ext-link></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McIsaac</surname> <given-names>RS</given-names></name><name><surname>Oakes</surname> <given-names>BL</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Dummit</surname> <given-names>KA</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name><name><surname>Noyes</surname> <given-names>MB</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Synthetic gene expression perturbation systems with rapid, tunable, single-gene specificity in yeast</article-title><source>Nucleic Acids Research</source><volume>41</volume><elocation-id>e57</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gks1313</pub-id><pub-id pub-id-type="pmid">23275543</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>McKinney</surname> <given-names>WO</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Data structures for statistical computing in Python</article-title><conf-name>Proceedings of the 9th Python in Science Conference</conf-name><conf-loc>Austin</conf-loc><fpage>51</fpage><lpage>56</lpage><ext-link ext-link-type="uri" xlink:href="https://conference.scipy.org/proceedings/scipy2010/mckinney.html">https://conference.scipy.org/proceedings/scipy2010/mckinney.html</ext-link></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Milias-Argeitis</surname> <given-names>A</given-names></name><name><surname>Oliveira</surname> <given-names>AP</given-names></name><name><surname>Gerosa</surname> <given-names>L</given-names></name><name><surname>Falter</surname> <given-names>L</given-names></name><name><surname>Sauer</surname> <given-names>U</given-names></name><name><surname>Lygeros</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Elucidation of genetic interactions in the yeast GATA-Factor network using bayesian model selection</article-title><source>PLOS Computational Biology</source><volume>12</volume><elocation-id>e1004784</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1004784</pub-id><pub-id pub-id-type="pmid">26967983</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>D</given-names></name><name><surname>Brandt</surname> <given-names>N</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Systematic identification of factors mediating accelerated mRNA degradation in response to changes in environmental nitrogen</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007406</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007406</pub-id><pub-id pub-id-type="pmid">29782489</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miraldi</surname> <given-names>ER</given-names></name><name><surname>Pokrovskii</surname> <given-names>M</given-names></name><name><surname>Watters</surname> <given-names>A</given-names></name><name><surname>Castro</surname> <given-names>DM</given-names></name><name><surname>De Veaux</surname> <given-names>N</given-names></name><name><surname>Hall</surname> <given-names>JA</given-names></name><name><surname>Lee</surname> <given-names>JY</given-names></name><name><surname>Ciofani</surname> <given-names>M</given-names></name><name><surname>Madar</surname> <given-names>A</given-names></name><name><surname>Carriero</surname> <given-names>N</given-names></name><name><surname>Littman</surname> <given-names>DR</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Leveraging chromatin accessibility for transcriptional regulatory network inference in T helper 17 cells</article-title><source>Genome Research</source><volume>29</volume><fpage>449</fpage><lpage>463</lpage><pub-id pub-id-type="doi">10.1101/gr.238253.118</pub-id><pub-id pub-id-type="pmid">30696696</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mittal</surname> <given-names>N</given-names></name><name><surname>Guimaraes</surname> <given-names>JC</given-names></name><name><surname>Gross</surname> <given-names>T</given-names></name><name><surname>Schmidt</surname> <given-names>A</given-names></name><name><surname>Vina-Vilaseca</surname> <given-names>A</given-names></name><name><surname>Nedialkova</surname> <given-names>DD</given-names></name><name><surname>Aeschimann</surname> <given-names>F</given-names></name><name><surname>Leidel</surname> <given-names>SA</given-names></name><name><surname>Spang</surname> <given-names>A</given-names></name><name><surname>Zavolan</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The Gcn4 transcription factor reduces protein synthesis capacity and extends yeast lifespan</article-title><source>Nature Communications</source><volume>8</volume><elocation-id>457</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-017-00539-y</pub-id><pub-id pub-id-type="pmid">28878244</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mueller</surname> <given-names>PP</given-names></name><name><surname>Hinnebusch</surname> <given-names>AG</given-names></name></person-group><year iso-8601-date="1986">1986</year><article-title>Multiple upstream AUG codons mediate translational control of GCN4</article-title><source>Cell</source><volume>45</volume><fpage>201</fpage><lpage>207</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(86)90384-3</pub-id><pub-id pub-id-type="pmid">3516411</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nadal-Ribelles</surname> <given-names>M</given-names></name><name><surname>Islam</surname> <given-names>S</given-names></name><name><surname>Wei</surname> <given-names>W</given-names></name><name><surname>Latorre</surname> <given-names>P</given-names></name><name><surname>Nguyen</surname> <given-names>M</given-names></name><name><surname>de Nadal</surname> <given-names>E</given-names></name><name><surname>Posas</surname> <given-names>F</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Sensitive high-throughput single-cell RNA-seq reveals within-clonal transcript correlations in yeast populations</article-title><source>Nature Microbiology</source><volume>4</volume><fpage>683</fpage><lpage>692</lpage><pub-id pub-id-type="doi">10.1038/s41564-018-0346-9</pub-id><pub-id pub-id-type="pmid">30718850</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nagarajan</surname> <given-names>S</given-names></name><name><surname>Kruckeberg</surname> <given-names>AL</given-names></name><name><surname>Schmidt</surname> <given-names>KH</given-names></name><name><surname>Kroll</surname> <given-names>E</given-names></name><name><surname>Hamilton</surname> <given-names>M</given-names></name><name><surname>McInnerney</surname> <given-names>K</given-names></name><name><surname>Summers</surname> <given-names>R</given-names></name><name><surname>Taylor</surname> <given-names>T</given-names></name><name><surname>Rosenzweig</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Uncoupling reproduction from metabolism extends chronological lifespan in yeast</article-title><source>PNAS</source><volume>111</volume><fpage>E1538</fpage><lpage>E1547</lpage><pub-id pub-id-type="doi">10.1073/pnas.1323918111</pub-id><pub-id pub-id-type="pmid">24706810</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Natarajan</surname> <given-names>K</given-names></name><name><surname>Meyer</surname> <given-names>MR</given-names></name><name><surname>Jackson</surname> <given-names>BM</given-names></name><name><surname>Slade</surname> <given-names>D</given-names></name><name><surname>Roberts</surname> <given-names>C</given-names></name><name><surname>Hinnebusch</surname> <given-names>AG</given-names></name><name><surname>Marton</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Transcriptional profiling shows that Gcn4p is a master regulator of gene expression during amino acid starvation in yeast</article-title><source>Molecular and Cellular Biology</source><volume>21</volume><fpage>4347</fpage><lpage>4368</lpage><pub-id pub-id-type="doi">10.1128/MCB.21.13.4347-4368.2001</pub-id><pub-id pub-id-type="pmid">11390663</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Neuwirth</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2014">2014</year><data-title>RColorBrewer: ColorBrewer Palettes</data-title></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F</given-names></name><name><surname>Varoquaux</surname> <given-names>G</given-names></name><name><surname>Gramfort</surname> <given-names>A</given-names></name><name><surname>Michel</surname> <given-names>V</given-names></name><name><surname>Thirion</surname> <given-names>B</given-names></name><name><surname>Grisel</surname> <given-names>O</given-names></name><name><surname>Blondel</surname> <given-names>M</given-names></name><name><surname>Prettenhofer</surname> <given-names>P</given-names></name><name><surname>Weiss</surname> <given-names>R</given-names></name><name><surname>Dubourg</surname> <given-names>V</given-names></name><name><surname>Vanderplas</surname> <given-names>J</given-names></name><name><surname>Passos</surname> <given-names>A</given-names></name><name><surname>Cournapeau</surname> <given-names>D</given-names></name><name><surname>Brucher</surname> <given-names>M</given-names></name><name><surname>Perrot</surname> <given-names>M</given-names></name><name><surname>É</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Scikit-learn: machine learning in Python</article-title><source>Journal of Machine Learning Research : JMLR</source><volume>12</volume><fpage>2825</fpage><lpage>2830</lpage></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pelechano</surname> <given-names>V</given-names></name><name><surname>Wei</surname> <given-names>W</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Extensive transcriptional heterogeneity revealed by isoform profiling</article-title><source>Nature</source><volume>497</volume><fpage>127</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1038/nature12121</pub-id><pub-id pub-id-type="pmid">23615609</pub-id></element-citation></ref><ref id="bib93"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Petukhov</surname> <given-names>V</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>ggrastr: Raster layers for ggplot2</data-title></element-citation></ref><ref id="bib94"><element-citation publication-type="software"><person-group person-group-type="author"><collab>R Development Core Team</collab></person-group><year iso-8601-date="2018">2018</year><data-title><italic>R: A Language and Environment for Statistical Computing</italic></data-title><publisher-loc>Vienna, Austria</publisher-loc><ext-link ext-link-type="uri" xlink:href="http://www.r-project.org">http://www.r-project.org</ext-link></element-citation></ref><ref id="bib95"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Rocklin</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Dask: parallel computation with blocked algorithms and task scheduling</article-title><conf-name>Proceedings of the 14th Python in Science Conference</conf-name><fpage>126</fpage><lpage>132</lpage><ext-link ext-link-type="uri" xlink:href="http://conference.scipy.org/proceedings/scipy2015/matthew_rocklin.html">http://conference.scipy.org/proceedings/scipy2015/matthew_rocklin.html</ext-link></element-citation></ref><ref id="bib96"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rødkaer</surname> <given-names>SV</given-names></name><name><surname>Faergeman</surname> <given-names>NJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Glucose- and nitrogen sensing and regulatory mechanisms in <italic>Saccharomyces cerevisiae</italic></article-title><source>FEMS Yeast Research</source><volume>14</volume><fpage>683</fpage><lpage>696</lpage><pub-id pub-id-type="doi">10.1111/1567-1364.12157</pub-id><pub-id pub-id-type="pmid">24738657</pub-id></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ruiz-Roig</surname> <given-names>C</given-names></name><name><surname>Noriega</surname> <given-names>N</given-names></name><name><surname>Duch</surname> <given-names>A</given-names></name><name><surname>Posas</surname> <given-names>F</given-names></name><name><surname>de Nadal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The Hog1 SAPK controls the Rtg1/Rtg3 transcriptional complex activity by multiple regulatory mechanisms</article-title><source>Molecular Biology of the Cell</source><volume>23</volume><fpage>4286</fpage><lpage>4296</lpage><pub-id pub-id-type="doi">10.1091/mbc.e12-04-0289</pub-id><pub-id pub-id-type="pmid">22956768</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Saint</surname> <given-names>M</given-names></name><name><surname>Bertaux</surname> <given-names>F</given-names></name><name><surname>Tang</surname> <given-names>W</given-names></name><name><surname>Sun</surname> <given-names>X-M</given-names></name><name><surname>Game</surname> <given-names>L</given-names></name><name><surname>Köferle</surname> <given-names>A</given-names></name><name><surname>Bähler</surname> <given-names>J</given-names></name><name><surname>Shahrezaei</surname> <given-names>V</given-names></name><name><surname>Marguerat</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Single-cell imaging and RNA sequencing reveal patterns of gene expression heterogeneity during fission yeast growth and adaptation</article-title><source>Nature Microbiology</source><volume>4</volume><fpage>480</fpage><lpage>491</lpage><pub-id pub-id-type="doi">10.1038/s41564-018-0330-4</pub-id></element-citation></ref><ref id="bib99"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Schafer</surname> <given-names>J</given-names></name><name><surname>Opgen-Rhein</surname> <given-names>R</given-names></name><name><surname>Zuber</surname> <given-names>V</given-names></name><name><surname>Ahdesmaki</surname> <given-names>M</given-names></name><name><surname>Silva</surname> <given-names>APD</given-names></name><name><surname>Strimmer</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2017">2017</year><data-title>corpcor: Efficient Estimation of Covariance and (Partial) Correlation</data-title></element-citation></ref><ref id="bib100"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Schloerke</surname> <given-names>B</given-names></name><name><surname>Crowley</surname> <given-names>J</given-names></name><name><surname>Cook</surname> <given-names>D</given-names></name><name><surname>Briatte</surname> <given-names>F</given-names></name><name><surname>Marbach</surname> <given-names>M</given-names></name><name><surname>Thoen</surname> <given-names>E</given-names></name><name><surname>Elberg</surname> <given-names>A</given-names></name><name><surname>Larmarange</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>GGally: Extension to “ggplot2.</data-title></element-citation></ref><ref id="bib101"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Scholes</surname> <given-names>AN</given-names></name><name><surname>Lewis</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Comparison of RNA isolation methods on RNA-Seq: implications for differential expression and Meta-Analyses</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/728014</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siahpirani</surname> <given-names>AF</given-names></name><name><surname>Roy</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A prior-based integrative framework for functional transcriptional regulatory network inference</article-title><source>Nucleic Acids Research</source><volume>45</volume><elocation-id>2221</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkw1160</pub-id><pub-id pub-id-type="pmid">27794550</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Silverman</surname> <given-names>SJ</given-names></name><name><surname>Petti</surname> <given-names>AA</given-names></name><name><surname>Slavov</surname> <given-names>N</given-names></name><name><surname>Parsons</surname> <given-names>L</given-names></name><name><surname>Briehof</surname> <given-names>R</given-names></name><name><surname>Thiberge</surname> <given-names>SY</given-names></name><name><surname>Zenklusen</surname> <given-names>D</given-names></name><name><surname>Gandhi</surname> <given-names>SJ</given-names></name><name><surname>Larson</surname> <given-names>DR</given-names></name><name><surname>Singer</surname> <given-names>RH</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Metabolic cycling in single yeast cells from unsynchronized steady-state populations limited on glucose or phosphate</article-title><source>PNAS</source><volume>107</volume><fpage>6946</fpage><lpage>6951</lpage><pub-id pub-id-type="doi">10.1073/pnas.1002422107</pub-id><pub-id pub-id-type="pmid">20335538</pub-id></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Slavov</surname> <given-names>N</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Coupling among growth rate response, metabolic cycle, and cell division cycle in yeast</article-title><source>Molecular Biology of the Cell</source><volume>22</volume><fpage>1997</fpage><lpage>2009</lpage><pub-id pub-id-type="doi">10.1091/mbc.e11-02-0132</pub-id><pub-id pub-id-type="pmid">21525243</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soneson</surname> <given-names>C</given-names></name><name><surname>Robinson</surname> <given-names>MD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Bias, robustness and scalability in single-cell differential expression analysis</article-title><source>Nature Methods</source><volume>15</volume><fpage>255</fpage><lpage>261</lpage><pub-id pub-id-type="doi">10.1038/nmeth.4612</pub-id><pub-id pub-id-type="pmid">29481549</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spellman</surname> <given-names>PT</given-names></name><name><surname>Sherlock</surname> <given-names>G</given-names></name><name><surname>Zhang</surname> <given-names>MQ</given-names></name><name><surname>Iyer</surname> <given-names>VR</given-names></name><name><surname>Anders</surname> <given-names>K</given-names></name><name><surname>Eisen</surname> <given-names>MB</given-names></name><name><surname>Brown</surname> <given-names>PO</given-names></name><name><surname>Botstein</surname> <given-names>D</given-names></name><name><surname>Futcher</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Comprehensive identification of cell cycle-regulated genes of the yeast <italic>Saccharomyces cerevisiae</italic> by microarray hybridization</article-title><source>Molecular Biology of the Cell</source><volume>9</volume><fpage>3273</fpage><lpage>3297</lpage><pub-id pub-id-type="doi">10.1091/mbc.9.12.3273</pub-id><pub-id pub-id-type="pmid">9843569</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stanbrough</surname> <given-names>M</given-names></name><name><surname>Magasanik</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Transcriptional and posttranslational regulation of the general amino acid permease of <italic>Saccharomyces cerevisiae</italic></article-title><source>Journal of Bacteriology</source><volume>177</volume><fpage>94</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1128/JB.177.1.94-102.1995</pub-id><pub-id pub-id-type="pmid">7798155</pub-id></element-citation></ref><ref id="bib108"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Talarek</surname> <given-names>N</given-names></name><name><surname>Gueydon</surname> <given-names>E</given-names></name><name><surname>Schwob</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Homeostatic control of START through negative feedback between Cln3-Cdk1 and Rim15/Greatwall kinase in budding yeast</article-title><source>eLife</source><volume>6</volume><elocation-id>e26233</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.26233</pub-id><pub-id pub-id-type="pmid">28600888</pub-id></element-citation></ref><ref id="bib109"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>F</given-names></name><name><surname>Barbacioru</surname> <given-names>C</given-names></name><name><surname>Wang</surname> <given-names>Y</given-names></name><name><surname>Nordman</surname> <given-names>E</given-names></name><name><surname>Lee</surname> <given-names>C</given-names></name><name><surname>Xu</surname> <given-names>N</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Bodeau</surname> <given-names>J</given-names></name><name><surname>Tuch</surname> <given-names>BB</given-names></name><name><surname>Siddiqui</surname> <given-names>A</given-names></name><name><surname>Lao</surname> <given-names>K</given-names></name><name><surname>Surani</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>mRNA-Seq whole-transcriptome analysis of a single cell</article-title><source>Nature Methods</source><volume>6</volume><fpage>377</fpage><lpage>382</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1315</pub-id><pub-id pub-id-type="pmid">19349980</pub-id></element-citation></ref><ref id="bib110"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tchourine</surname> <given-names>K</given-names></name><name><surname>Vogel</surname> <given-names>C</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Condition-Specific modeling of biophysical parameters advances inference of regulatory networks</article-title><source>Cell Reports</source><volume>23</volume><fpage>376</fpage><lpage>388</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2018.03.048</pub-id><pub-id pub-id-type="pmid">29641998</pub-id></element-citation></ref><ref id="bib111"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teixeira</surname> <given-names>MC</given-names></name><name><surname>Monteiro</surname> <given-names>PT</given-names></name><name><surname>Palma</surname> <given-names>M</given-names></name><name><surname>Costa</surname> <given-names>C</given-names></name><name><surname>Godinho</surname> <given-names>CP</given-names></name><name><surname>Pais</surname> <given-names>P</given-names></name><name><surname>Cavalheiro</surname> <given-names>M</given-names></name><name><surname>Antunes</surname> <given-names>M</given-names></name><name><surname>Lemos</surname> <given-names>A</given-names></name><name><surname>Pedreira</surname> <given-names>T</given-names></name><name><surname>Sá-Correia</surname> <given-names>I</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>YEASTRACT: an upgraded database for the analysis of transcription regulatory networks in <italic>Saccharomyces cerevisiae</italic></article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>D348</fpage><lpage>D353</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx842</pub-id><pub-id pub-id-type="pmid">29036684</pub-id></element-citation></ref><ref id="bib112"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>BP</given-names></name><name><surname>Kudlicki</surname> <given-names>A</given-names></name><name><surname>Rowicka</surname> <given-names>M</given-names></name><name><surname>McKnight</surname> <given-names>SL</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Logic of the yeast metabolic cycle: temporal compartmentalization of cellular processes</article-title><source>Science</source><volume>310</volume><fpage>1152</fpage><lpage>1158</lpage><pub-id pub-id-type="doi">10.1126/science.1120499</pub-id><pub-id pub-id-type="pmid">16254148</pub-id></element-citation></ref><ref id="bib113"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van der Walt</surname> <given-names>S</given-names></name><name><surname>Colbert</surname> <given-names>SC</given-names></name><name><surname>Varoquaux</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The NumPy array: a structure for efficient numerical computation</article-title><source>Computing in Science &amp; Engineering</source><volume>13</volume><fpage>22</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2011.37</pub-id></element-citation></ref><ref id="bib114"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Dijk</surname> <given-names>D</given-names></name><name><surname>Sharma</surname> <given-names>R</given-names></name><name><surname>Nainys</surname> <given-names>J</given-names></name><name><surname>Yim</surname> <given-names>K</given-names></name><name><surname>Kathail</surname> <given-names>P</given-names></name><name><surname>Carr</surname> <given-names>AJ</given-names></name><name><surname>Burdziak</surname> <given-names>C</given-names></name><name><surname>Moon</surname> <given-names>KR</given-names></name><name><surname>Chaffer</surname> <given-names>CL</given-names></name><name><surname>Pattabiraman</surname> <given-names>D</given-names></name><name><surname>Bierie</surname> <given-names>B</given-names></name><name><surname>Mazutis</surname> <given-names>L</given-names></name><name><surname>Wolf</surname> <given-names>G</given-names></name><name><surname>Krishnaswamy</surname> <given-names>S</given-names></name><name><surname>Pe'er</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Recovering gene interactions from Single-Cell data using data diffusion</article-title><source>Cell</source><volume>174</volume><fpage>716</fpage><lpage>729</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.05.061</pub-id><pub-id pub-id-type="pmid">29961576</pub-id></element-citation></ref><ref id="bib115"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname> <given-names>P</given-names></name><name><surname>Gommers</surname> <given-names>R</given-names></name><name><surname>Oliphant</surname> <given-names>TE</given-names></name><name><surname>Haberland</surname> <given-names>M</given-names></name><name><surname>Reddy</surname> <given-names>T</given-names></name><name><surname>Cournapeau</surname> <given-names>D</given-names></name><name><surname>Burovski</surname> <given-names>E</given-names></name><name><surname>Peterson</surname> <given-names>P</given-names></name><name><surname>Weckesser</surname> <given-names>W</given-names></name><name><surname>Bright</surname> <given-names>J</given-names></name><name><surname>van der Walt</surname> <given-names>SJ</given-names></name><name><surname>Brett</surname> <given-names>M</given-names></name><name><surname>Wilson</surname> <given-names>J</given-names></name><name><surname>Millman</surname> <given-names>KJ</given-names></name><name><surname>Mayorov</surname> <given-names>N</given-names></name><name><surname>Nelson</surname> <given-names>ARJ</given-names></name><name><surname>Jones</surname> <given-names>E</given-names></name><name><surname>Kern</surname> <given-names>R</given-names></name><name><surname>Larson</surname> <given-names>E</given-names></name><name><surname>Carey</surname> <given-names>CJ</given-names></name><name><surname>Polat</surname> <given-names>İlhan</given-names></name><name><surname>Feng</surname> <given-names>Y</given-names></name><name><surname>Moore</surname> <given-names>EW</given-names></name><name><surname>VanderPlas</surname> <given-names>J</given-names></name><name><surname>Laxalde</surname> <given-names>D</given-names></name><name><surname>Perktold</surname> <given-names>J</given-names></name><name><surname>Cimrman</surname> <given-names>R</given-names></name><name><surname>Henriksen</surname> <given-names>I</given-names></name><name><surname>Quintero</surname> <given-names>EA</given-names></name><name><surname>Harris</surname> <given-names>CR</given-names></name><name><surname>Archibald</surname> <given-names>AM</given-names></name><name><surname>Ribeiro</surname> <given-names>AH</given-names></name><name><surname>Pedregosa</surname> <given-names>F</given-names></name><name><surname>van Mulbregt</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SciPy 1.0: fundamental algorithms for scientific computing in Python</article-title><source>Nature Methods</source><volume>13</volume><pub-id pub-id-type="doi">10.1038/s41592-019-0686-2</pub-id></element-citation></ref><ref id="bib116"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ward</surname> <given-names>LD</given-names></name><name><surname>Bussemaker</surname> <given-names>HJ</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Predicting functional transcription factor binding through alignment-free and affinity-based analysis of orthologous promoter sequences</article-title><source>Bioinformatics</source><volume>24</volume><fpage>i165</fpage><lpage>i171</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btn154</pub-id><pub-id pub-id-type="pmid">18586710</pub-id></element-citation></ref><ref id="bib117"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Reshaping data with the reshape package</article-title><source>Journal of Statistical Software</source><volume>21</volume><fpage>1</fpage><lpage>20</lpage><pub-id pub-id-type="doi">10.18637/jss.v021.i12</pub-id></element-citation></ref><ref id="bib118"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><source>Ggplot2: Elegant Graphics for Data Analysis</source><publisher-name>Springer-Verlag</publisher-name></element-citation></ref><ref id="bib119"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name><name><surname>François</surname> <given-names>R</given-names></name><name><surname>Henry</surname> <given-names>L</given-names></name><name><surname>Müller</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>dplyr: A Grammar of Data Manipulation</data-title></element-citation></ref><ref id="bib120"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2018">2018a</year><data-title>scales: Scale Functions for Visualization</data-title></element-citation></ref><ref id="bib121"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2018">2018b</year><data-title>stringr: Simple Consistent Wrappers for Common String Operations</data-title></element-citation></ref><ref id="bib122"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wilke</surname> <given-names>CO</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>ggridges: Ridgeline Plots in ggplot2</data-title></element-citation></ref><ref id="bib123"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wilke</surname> <given-names>CO</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>cowplot: Streamlined Plot Theme and Plot Annotations for ggplot2</data-title></element-citation></ref><ref id="bib124"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wilkins</surname> <given-names>O</given-names></name><name><surname>Hafemeister</surname> <given-names>C</given-names></name><name><surname>Plessis</surname> <given-names>A</given-names></name><name><surname>Holloway-Phillips</surname> <given-names>MM</given-names></name><name><surname>Pham</surname> <given-names>GM</given-names></name><name><surname>Nicotra</surname> <given-names>AB</given-names></name><name><surname>Gregorio</surname> <given-names>GB</given-names></name><name><surname>Jagadish</surname> <given-names>SV</given-names></name><name><surname>Septiningsih</surname> <given-names>EM</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name><name><surname>Purugganan</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>EGRINs (Environmental gene regulatory influence networks) in rice that function in the response to water deficit, high temperature, and agricultural environments</article-title><source>The Plant Cell</source><volume>28</volume><fpage>2365</fpage><lpage>2384</lpage><pub-id pub-id-type="doi">10.1105/tpc.16.00158</pub-id><pub-id pub-id-type="pmid">27655842</pub-id></element-citation></ref><ref id="bib125"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>C</given-names></name><name><surname>Su</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Identification of cell types from single-cell transcriptomes using a novel clustering method</article-title><source>Bioinformatics</source><volume>31</volume><fpage>1974</fpage><lpage>1980</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv088</pub-id></element-citation></ref><ref id="bib126"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>GX</given-names></name><name><surname>Terry</surname> <given-names>JM</given-names></name><name><surname>Belgrader</surname> <given-names>P</given-names></name><name><surname>Ryvkin</surname> <given-names>P</given-names></name><name><surname>Bent</surname> <given-names>ZW</given-names></name><name><surname>Wilson</surname> <given-names>R</given-names></name><name><surname>Ziraldo</surname> <given-names>SB</given-names></name><name><surname>Wheeler</surname> <given-names>TD</given-names></name><name><surname>McDermott</surname> <given-names>GP</given-names></name><name><surname>Zhu</surname> <given-names>J</given-names></name><name><surname>Gregory</surname> <given-names>MT</given-names></name><name><surname>Shuga</surname> <given-names>J</given-names></name><name><surname>Montesclaros</surname> <given-names>L</given-names></name><name><surname>Underwood</surname> <given-names>JG</given-names></name><name><surname>Masquelier</surname> <given-names>DA</given-names></name><name><surname>Nishimura</surname> <given-names>SY</given-names></name><name><surname>Schnall-Levin</surname> <given-names>M</given-names></name><name><surname>Wyatt</surname> <given-names>PW</given-names></name><name><surname>Hindson</surname> <given-names>CM</given-names></name><name><surname>Bharadwaj</surname> <given-names>R</given-names></name><name><surname>Wong</surname> <given-names>A</given-names></name><name><surname>Ness</surname> <given-names>KD</given-names></name><name><surname>Beppu</surname> <given-names>LW</given-names></name><name><surname>Deeg</surname> <given-names>HJ</given-names></name><name><surname>McFarland</surname> <given-names>C</given-names></name><name><surname>Loeb</surname> <given-names>KR</given-names></name><name><surname>Valente</surname> <given-names>WJ</given-names></name><name><surname>Ericson</surname> <given-names>NG</given-names></name><name><surname>Stevens</surname> <given-names>EA</given-names></name><name><surname>Radich</surname> <given-names>JP</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Hindson</surname> <given-names>BJ</given-names></name><name><surname>Bielas</surname> <given-names>JH</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Massively parallel digital transcriptional profiling of single cells</article-title><source>Nature Communications</source><volume>8</volume><elocation-id>14049</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms14049</pub-id><pub-id pub-id-type="pmid">28091601</pub-id></element-citation></ref><ref id="bib127"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zilionis</surname> <given-names>R</given-names></name><name><surname>Nainys</surname> <given-names>J</given-names></name><name><surname>Veres</surname> <given-names>A</given-names></name><name><surname>Savova</surname> <given-names>V</given-names></name><name><surname>Zemmour</surname> <given-names>D</given-names></name><name><surname>Klein</surname> <given-names>AM</given-names></name><name><surname>Mazutis</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Single-cell barcoding and sequencing using droplet microfluidics</article-title><source>Nature Protocols</source><volume>12</volume><fpage>44</fpage><lpage>73</lpage><pub-id pub-id-type="doi">10.1038/nprot.2016.154</pub-id><pub-id pub-id-type="pmid">27929523</pub-id></element-citation></ref><ref id="bib128"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zinzalla</surname> <given-names>V</given-names></name><name><surname>Graziola</surname> <given-names>M</given-names></name><name><surname>Mastriani</surname> <given-names>A</given-names></name><name><surname>Vanoni</surname> <given-names>M</given-names></name><name><surname>Alberghina</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Rapamycin-mediated G1 arrest involves regulation of the cdk inhibitor Sic1 in <italic>Saccharomyces cerevisiae</italic></article-title><source>Molecular Microbiology</source><volume>63</volume><fpage>1482</fpage><lpage>1494</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2958.2007.05599.x</pub-id><pub-id pub-id-type="pmid">17302822</pub-id></element-citation></ref><ref id="bib129"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The adaptive lasso and its oracle properties</article-title><source>Journal of the American Statistical Association</source><volume>101</volume><fpage>1418</fpage><lpage>1429</lpage><pub-id pub-id-type="doi">10.1198/016214506000000735</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.51254.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Barkai</surname><given-names>Naama</given-names></name><role>Reviewing Editor</role><aff><institution>Weizmann Institute of Science</institution><country>Israel</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>In your article, you present individual transcriptomes of diploid yeast cells, applying your method to monitor transcriptional changes across an array of newly generated barcoded deletion mutants across a panel of stresses. Beyond the development of the method, the main and most original point of the manuscript is that you infer gene regulatory networks based on the transcriptomes retrieved from barcoded genotypes in diverse conditions. The manuscript is well written and together with related reports will become one of the golden standards for yeast single cells. You are commended for providing a complete and user-friendly dataset (deposited and interactive through a shiny app), which will be a valuable resource for the yeast community.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your work entitled &quot;Gene regulatory network reconstruction using single-cell RNA sequencing of barcoded genotypes in diverse environments&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by three peer reviewers, and the evaluation has been overseen by a Reviewing Editor and a Senior Editor. The reviewers have opted to remain anonymous.</p><p>Our decision has been reached after consultation between the reviewers. Based on these discussions and the individual reviews below. As you will see, the reviewers found the paper to be a bit preliminary in the interpretation and the meaningfulness of the presented data. The reviewers did find the work to be potentially suitable for <italic>eLife</italic> and will be happy to look at a revised version, provided that you can fully address all comments raised in the individual reviews.</p><p><italic>Reviewer #1:</italic></p><p>Here, Jackson et al. apply for the first time a 10x Genomics protocol for scRNAseq in barcoded <italic>S. cerevisiae</italic> deletion strains. The work analyzed 11 transcription-factor deletion strains pooled together and responding to 11 different conditions related to nitrogen metabolism. The authors then applied a gene-regulatory network (GRN) inference method to predict a nitrogen metabolism regulatory network.</p><p>The pluses of the work are that it is a hot topic and, although not the first scRNAseq in fungi, the first to look very broadly at tens of thousands of yeast cells. The barcoding method is interesting, although unclear how it's different from the barcode sequencing that is part of the standard 10x genomics pipeline and has been used before in other systems.</p><p>The weaker points for me were in the analysis. The authors have expertise in GRN inference, but I had a hard time understanding from the main text what was novel here and specific for single cell data versus previously published methods applied to data pooled across cells/conditions. I will leave it to network inference modelers to dissect those details. But in a broader sense, I was left wanting more follow-up to show that the methods produced new insights. The authors predicted a network but as there is no biological validation and little computational validation it's unclear how big of an advance this is. I was also left wondering about the biological insights that can be gleaned from having single cell data (beyond variation in cell-cycle stage, which is readily identifiable in all scRNAseq studies). I suspect there is interesting biology in the heterogeneity in the response data but there was not much addressed on that topic. In my opinion, this is a great new method with a potentially powerful dataset. But since there are many GRN methods and this overall approach seems similar to Perturb-seq and other methods, the results and impact for me fall below the bar of <italic>eLife</italic>.</p><p>Some specific points are outlined below.</p><p>1) The authors report reads from 38,000 cells, which to date is the most cells studied in fungi. But the median number of genes covered is only &lt;700. Since the paper focuses on re-bulked data (to call differentially expressed genes by DESeq2 and, I think, for their main GRN network inference?), I was left wondering how many genes are measured in the re-bulked data per condition. I was surprised how few genes were called by DESeq (Figure 4B), but it's unclear how many genes are actually measured in &gt;1 cell condition/mutant. I was also curious what fraction of known targets (e.g. based on prior studies or ChIP-seq datasets) were called for measured genes.</p><p>2) It would be useful to know how well the 10x protocol works for cells and if the aggregated wild-type data recapitulates bulk profiles in conditions that have been previously measured. I was a little concerned at how the cells were collected, which appeared to take live cells and wash them several times in RNALater buffer – does that immediately kill cells? If not, I was wondering if that is inducing a response. I was also left wondering if the protocol captures only the most abundant transcripts. Perhaps I missed this on the supplement, but I was wondering how the% cells in which a transcript was measured compared to RPKM from bulk measurements. Clearly more abundant transcripts will be more easily captured, but some more analysis here would be useful for a new method.</p><p>3) The GRN modeling was not clear to me from the main text. It appears that the authors are using their published Inferelator method that takes priors based on ChIP-seq data, and re-bulks the scRNAseq data (at least for the multi-task inference). Perhaps their point is that the 10x approach allows pooling of many genotypes and conditions, but for me the analysis missed the potential power of having single cell data. The authors make statements on the networks in Figure 6 and Figure 7 about the number of &quot;novel&quot; regulatory connections – but I saw no validation of those predictions, including by computation. How do we know that any of these are real and that the method is producing new insights? AUROCs comparing to known data is not enough to say that new regulatory connections were discovered. This was especially true for Figure 7 – how do we know these new predictions mean anything about a connection to cell cycle without some validation?</p><p>4) I had some quibbles with part of the Discussion.</p><p>i) First, while the authors cite several recent <italic>S. cerevisiae</italic> scRNAseq datasets, saying that this is the &quot;first report of large-scale scRNAseq&quot; in yeast is not accurate – it's true they measured 38,000 cells but at a depth of only &lt;700 genes per cell, which is far fewer than 2500-3000 of other studies in several hundred cells. A fairer sentence is required, also citations of recent <italic>Sz. pombe</italic> scRNAseq (Saint et al., 2019) should be included.</p><p>ii) &quot;We observe significant heterogeneity in individual cells.… Much of this variation can be explained by the mitotic cell cycle&quot; – that statement is not true, there does not seem to be an attempt to quantify heterogeneity over most genes. That they see heterogeneity in cell cycle stage as expected does not mean that cell cycle stage explains heterogeneity in the rest of the response, which was not reported on here.</p><p><italic>Reviewer #2:</italic></p><p>In the manuscript submitted by Jackson et al., the authors proposed to profile individual transcriptomes of diploid yeast cells optimizing the 10x genomics pipeline. The manuscript demonstrates the feasibility and robustness of the method and the authors applied it to monitor the transcriptional networks across an array of newly generated barcoded deletion mutants across a panel of stresses. The manuscript is scientifically sound and of outstanding quality, it's well written and together with previous reports this year will become one of the golden standards for yeast single cells.</p><p>The authors provide a complete and user-friendly dataset (deposited and interactive through a shiny app) that will be a valuable resource for the yeast community. All in all, I think this is a very solid manuscript that with minor corrections I would strongly support for publication in <italic>eLife</italic>.</p><p>Beyond the development of the method, the main and most original point of the manuscript is that the authors aim to generate infer gene regulatory networks based on the transcriptomes retrieved from barcoded genotypes in diverse conditions. I have some questions and comments (of varying levels of concern) that I feel should be addressed in the current version of the manuscript.</p><p>The authors leverage in their previous experience in GRN reconstruction and propose this approach has allowed them to discover novel regulatory relationships between cell cycle-regulated gene expression in response to changes in nitrogen source. In my opinion, this part of the manuscript needs to be reinforced and some of the conclusions driven from the GRN should be and some representative novel regulatory relationships experimentally demonstrated. As well the authors should provide context to their findings as they often read a bit disconnected.</p><p>In terms of data quality, the number of mitochondrial and ribosomal reads per genotype and condition should be plotted, as a quality metric and given that a lot of ribosomal gene expression is cell cycle regulated. This could be relevant to understand separate clusters within conditions which the authors do not mention in the results and/or discussion and given the fact that different zymolyase concentrations were used for cell lysis.</p><p>The optical density of the cells at the harvesting (besides the total cell number) should be provided to ease the reproducibility for other labs.</p><p>If I understand correctly, cell cycle clusters within conditions in Figure 3 is confusing. For example expression of DSE2 or PIR1 seem to be highest in the green and grey clusters respectively (Figure 3A) however in Figure 3B panel I the highest expression is assigned to the grey-yellow for DSE2 and PIR1 to the green cluster.</p><p>As well, the UMAPS from Figure 3A, the authors claim the clustering is mainly condition-dependent and genotype-independent. However, conditions like MMEtOH, CSTARVE, NLIM-PRO have clear clusters that do not seem cell cycle-dependent and these might be biologically relevant. The authors should at least comment or those or run a DE analysis to see what these are.</p><p>How do the newly generated data compare to Gasch et al., 2017 and Nadal-Ribelles et al., 2019?</p><p>The number of differentially expressed genes even in the YPD condition seems a bit low. How do these pseudobulk compare to the deletome data (Holstege lab) or other published datasets?</p><p>As well, the authors use DESeq in Figure 4B, but these results are contradictory with what is shown in Figure S4Bi which is done by Welch testing. This is a bit confusing and does not add much to the reader. I would suggest to run DE between conditions using DESeq or provide the reasoning as to why these two different approaches are used and done?</p><p>Why do transcription factor activities FKH1 and FKH2 and SWI4 SWI5 do not overlap (they almost seem mutually exclusive), one would expect them to have similar profiles. Similarly, NDD1 regulates S-phase genes but the TFA does not overlap with the HTB expression shown in Figure 3.</p><p><italic>Reviewer #3:</italic></p><p>This paper by Gresham, Bonneau and colleagues presents to date the largest single cell RNA-seq datasets in yeast using a novel deletion and barcoding strategy that enables them to measure individual cells under different genetic perturbations of transcription factors and environmental conditions. The study is focused on better understanding the regulatory network in nitrogen starvation however it could be broadly applied across multiple conditions. Some findings are: the genotypes tend to be generally uniformly present in all conditions except RTG1 and 3 and GLN3; there is a co-regulated set of genes regulated by cell cycle and nitrogen TFs; multi-task learning is a viable approach for network inference in scRNAseq data. The paper is a significant contribution to the field providing a novel dataset to yeast and general gene regulation community.</p><p>I have some comments, which I think are minor and can strengthen the messages of the paper.</p><p>1) In Figure 5C, is the single task network inferred by merging all the data and learning a single network or by learning separate networks with single tasking and aggregating the results? If not, how does the single task per condition followed by aggregation perform?</p><p>2) The authors don't get much into the context-specificity of the inferred networks. They interpret only the final aggregated network. It would be useful to know how similar the individual condition-specific networks are and if there is a conserved core used by multiple conditions. The only comparison of context specificity is being done at the level of AUPR, it might be helpful to do this comparison just by comparing the inferred networks.</p><p>3) Some discussion about the variation in the AUPRs would be helpful. Is it because the gold standard is biased towards the conditions on which the AUPR is high. It seems the AUPR for the MMEtOH network is close to what is inferred by the multi-task learning, and some explanation of why this might be is helpful.</p><p>4) A comparison of the AUPR of the single task condition-specific networks and the multi-task condition specific network could further show the advantage of the multi-task learning framework.</p><p>5) It would be helpful to emphasize if and how the multi-task learning approach used here was from extended from the Castro, 2019 paper.</p><p>6) The Discussion could be strengthened. The authors present some results about the interplay between cell cycle and nitrogen response. It was not clear why this is interesting to study beyond that there is a shared regulatory program. This might be worth bringing up in the Discussion to tie back to the initial goal of inferring a network for nitrogen metabolism and the TOR signaling pathway and the general role of cell cycle and stress response.</p><p>7) The authors don't find a substantial impact of TF knockout on gene expression under different conditions (I assume the comparisons were done while controlling for the conditions). How does this compare to bulk data? How much of this observation could be due to the sparsity of scRNAseq data versus the redundancy of TFs.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.51254.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><p>We have revised the manuscript based on the feedback that we have received from reviewers. To address the reviewer concerns we have undertaken additional experimentation and analyses. Below, we summarize the major changes made to the text.</p><p>1) A common concern of the three reviewers was the quality and utility of the single-cell RNAseq data. To address this concern we performed bulk RNAseq on wild-type cells from the YPD rich media condition. Our analysis of this experiment used the same computational pipeline as used for our single-cell reads and indicates that scRNAseq in yeast produces data that are very comparable to that produced from bulk RNAseq. In addition, we have performed comparisons to other published single-cell data and to another bulk sequencing experiment from GEO, which support this conclusion. This analysis is included as Figure 2—figure supplement 2. We have added additional QC metrics suggested by the reviewers as Figure 2—figure supplement 3. We believe that these results validate the quality of data produced using this method and obviate any concerns that our sample processing introduces biases in the data.</p><p>2) A second common concern was validation of the network predictions made by our inference technique. To address this concern, we have compared the novel predictions made in this work to the YEASTRACT database. More than half of the predictions we make are supported by some type of evidence (either changes in target gene expression or physical localization by ChIP). As this evidence was not included in the modeling priors we believe that this provides orthogonal support for at least 50% of the interactions that we discovered in our study.</p><p>3) To address concerns regarding the computational approaches, we have regenerated the gene regulatory network results reported in Figures 5-7 incorporating the following changes:</p><p>a) We updated the Inferelator package for python to version 0.3.0 (some minor software bugs were fixed and the package is now compatible with Python 3; this upgrade did not change any modeling results).</p><p>b) We updated the YEASTRACT prior data to match the 2019 release of the YEASTRACT database; as a result, ~2000 TF-gene interactions were removed as being poorly supported, and ~3500 TF-gene interactions were added relative to the 2018 release used in our initial submission. Updating the modeling priors with these changes improved many of our modeling results.</p><p>c) While preparing this revision, we identified an issue that was causing our network inference to make spurious predictions due to genes that have a mean count value in at least one task group that is very close to zero (but not zero). We have mitigated this problem by filtering out genes that don’t have a minimum mean count of 0.05 (at least 1 read per 20 cells). For multi-task learning, this filter is applied task-wise; genes filtered from one task are still modeled in other tasks. This approach results in removal of the spurious interactions.</p><p>d)We have performed additional cross-validation for Figure 5B (now 20 runs).</p><p>4) The Materials and methods section has been substantially updated to incorporate these additions to the manuscript and several minor typos were corrected.</p><p>5) One reviewer identified a figure which had a labeling error introduced during panelling with Adobe Illustrator. This motivated us to include as a supplemental data file (Source code 5) an HTML document generated with Rmarkdown that contains Figures 2-7 (including all figure supplements). The figures produced in this way have not been subject to any subsequent adjustments in Illustrator and thus provide a reproducible record of the data analysis and presentation that can be regenerated by running the Rmarkdown script (which is included in Source code 1).</p><disp-quote content-type="editor-comment"><p>Reviewer #1:</p><p>[…] The pluses of the work are that it is a hot topic and, although not the first scRNAseq in fungi, the first to look very broadly at tens of thousands of yeast cells. The barcoding method is interesting, although unclear how it's different from the barcode sequencing that is part of the standard 10x genomics pipeline and has been used before in other systems.</p><p>The weaker points for me were in the analysis. The authors have expertise in GRN inference, but I had a hard time understanding from the main text what was novel here and specific for single cell data versus previously published methods applied to data pooled across cells/conditions. I will leave it to network inference modelers to dissect those details. But in a broader sense, I was left wanting more follow-up to show that the methods produced new insights. The authors predicted a network but as there is no biological validation and little computational validation it's unclear how big of an advance this is. I was also left wondering about the biological insights that can be gleaned from having single cell data (beyond variation in cell-cycle stage, which is readily identifiable in all scRNAseq studies). I suspect there is interesting biology in the heterogeneity in the response data but there was not much addressed on that topic. In my opinion, this is a great new method with a potentially powerful dataset. But since there are many GRN methods and this overall approach seems similar to Perturb-seq and other methods, the results and impact for me fall below the bar of eLife.</p></disp-quote><p>We thank the reviewer for recognizing the novelty of our study. As we emphasize in our manuscript, adopting the 10x Genomics systems for budding yeast (and microbes in general) is non-trivial and we have developed a robust methodology that we are convinced will be widely adopted. We are aware that the Perturb-seq method does express a barcode that is informative about the gRNA for CRISPR/Cas9 screens; however, our method expresses a barcode from the drug resistant cassette that is used for deleting the gene of interest and thus provides a more direct readout of the cells’ genotypes. The successful application of 10x Genomics to microbial cells and the method of multiplexing genotypes make this paper of broad interest to the readers of <italic>eLife</italic>.</p><p>With respect to the novelty of the computational approaches, we concur that there are indeed many network inference methods. However, we are convinced that there is a great deal of novelty in applying GRN inference techniques to real-world single-cell data in a testable way. We are aware of only one study that is similar in scope and goals to ours; it was released as a preprint around the same time as this work, and it has been recently accepted for publication in Science. (Norman, T.M., et al. (2019). Exploring genetic interaction manifolds constructed from rich single-cell phenotypes). The rigorous testing of unsolved aspects of GRN using scRNAseq data, such as imputation, and the application of multitask learning are key features of our manuscript that we are convinced will be of great value to the larger biological research community that aims to use scRNAseq for GRN reconstruction.</p><disp-quote content-type="editor-comment"><p>Some specific points are outlined below.</p><p>1) The authors report reads from 38,000 cells, which to date is the most cells studied in fungi. But the median number of genes covered is only &lt;700. Since the paper focuses on re-bulked data (to call differentially expressed genes by DESeq2 and, I think, for their main GRN network inference?), I was left wondering how many genes are measured in the re-bulked data per condition. I was surprised how few genes were called by DESeq (Figure 4B), but it's unclear how many genes are actually measured in &gt;1 cell condition/mutant. I was also curious what fraction of known targets (e.g. based on prior studies or ChIP-seq datasets) were called for measured genes.</p></disp-quote><p>It is critical to note that only Figure 4B, Figure 4C, and Figure 4—figure supplement 1B rely on bulked data. All other figures and analyses in our paper are based on single-cell data. The huge increase in the number of individual measurements that scRNAseq provides is particularly important for the GRN inference. We have clarified this point in the manuscript.</p><p>Cells cultured in YPD, the most commonly used rich media, are the best baseline for comparison to bulk methods. We have 11,037 cells present in this data set. A median of 684 genes are detected per cell, and 5,533 genes have at least one read across all cells (of 5,773 protein-coding genes). 5,403 genes are counted in more than 1 cell in this condition. As additional points of comparisons, 299 genes average at least one count/cell, and 1,831 genes average at least one count in every 10 cells. We believe that the low count/cell is the result of overall low sampling rates for transcripts (some of which may be technically addressable in the future), and is not the result of specific gene bias, which is supported by the high correlation to our new data using trizol extracted bulk RNAseq, which is included as Figure 2—figure supplement 2.</p><p>With respect to the fraction of known targets that are measured we compared the novel interactions identified in our study and find that about 50% of them are supported by prior studies included in the YEASTRACT database.</p><disp-quote content-type="editor-comment"><p>2) It would be useful to know how well the 10x protocol works for cells and if the aggregated wild-type data recapitulates bulk profiles in conditions that have been previously measured. I was a little concerned at how the cells were collected, which appeared to take live cells and wash them several times in RNALater buffer – does that immediately kill cells? If not, I was wondering if that is inducing a response. I was also left wondering if the protocol captures only the most abundant transcripts. Perhaps I missed this on the supplement, but I was wondering how the% cells in which a transcript was measured compared to RPKM from bulk measurements. Clearly more abundant transcripts will be more easily captured, but some more analysis here would be useful for a new method.</p></disp-quote><p>We thank the reviewer for raising this point; this is a very important methodological detail. It is necessary to fix cells prior to processing to prevent changes in the transcriptome. We chose to use RNAlater (saturated ammonium sulfate) in the initial development of our protocol as it is a widely used approach that has been successfully applied to yeast in the past. Snap freezing in liquid nitrogen is not a fixative but a storage solution, and it is not feasible with the single-cell workflow anyway, and we found in preliminary testing that methanol fixation introduced obvious bias after reverse transcription.</p><p>To specifically address the reviewer’s concern about transcript capture bias, we performed an RNA sequencing experiment using bulk RNA extracted from wildtype cells in YPD using a trizol-based protocol. For this bulk experiment, we used a 3’ end barcoding with UMI strategy that is directly comparable to the 10x genomics method for cDNA synthesis and library preparation, and which can be analyzed through the same computational pipeline as our single-cell data. The results of this experiment are presented in Figure 2—figure supplement 2. We find that, in aggregate, the single-cell expression data from WT cells in YPD correlates very well with the expression data from bulk RNA (spearman correlation of 0.94). In addition, despite considerable differences in technical protocols, as well as using different strains (auxotrophic haploids vs prototrophic diploids in our study), we find that the single-cell expression data from our 10x Genomics protocol also correlates well with the expression data from the yscRNAseq protocol reported by Nadal-Ribelles et al, 2019 (spearman correlation of 0.83). We believe that this convincingly shows that the 10x Genomics-based method we developed for scRNAseq does not introduce any biases in the data.</p><disp-quote content-type="editor-comment"><p>3) The GRN modeling was not clear to me from the main text. It appears that the authors are using their published Inferelator method that takes priors based on ChIP-seq data, and re-bulks the scRNAseq data (at least for the multi-task inference). Perhaps their point is that the 10x approach allows pooling of many genotypes and conditions, but for me the analysis missed the potential power of having single cell data. The authors make statements on the networks in Figure 6 and Figure 7 about the number of &quot;novel&quot; regulatory connections – but I saw no validation of those predictions, including by computation. How do we know that any of these are real and that the method is producing new insights? AUROCs comparing to known data is not enough to say that new regulatory connections were discovered. This was especially true for Figure 7 – how do we know these new predictions mean anything about a connection to cell cycle without some validation?</p></disp-quote><p>We have clarified in the text that all network modeling is based on single-cell data and does not use pseudo-bulked data. We now emphasize that a novel interaction in this context is defined as interactions we discover using our network inference method without prior knowledge. To validate these interactions we now include a comparison to data in the YEASTRACT database that was not included in our network inference method as priors. Of the 6114 novel interactions discovered in our network, YEASTRACT provides evidence of an existing regulatory relationship for more than half. We believe that this provides orthogonal validation of our newly discovered interactions.</p><p>As our data allows us to determine the expression of individual cells the true advantage is accessing expression at the single-cell level. While some high-quality work has been done to study the intersection of cell-cycle regulation and metabolic regulation, we believe that studying asynchronous cultures at the single-cell level will be immensely valuable. Studying these novel regulatory relationships will be the subject of future work.</p><disp-quote content-type="editor-comment"><p>4) I had some quibbles with part of the Discussion.</p><p>i) First, while the authors cite several recent <italic>S. cerevisiae</italic> scRNAseq datasets, saying that this is the &quot;first report of large-scale scRNAseq&quot; in yeast is not accurate – it's true they measured 38,000 cells but at a depth of only &lt;700 genes per cell, which is far fewer than 2500-3000 of other studies in several hundred cells. A fairer sentence is required, also citations of recent Sz. pombe scRNAseq (Saint et al., 2019) should be included.</p></disp-quote><p>In response to this comment we have changed the phrase ‘large-scale’ to ‘droplet-based’ and changed instances of ‘yeast’ in this context to ‘budding yeast’. We have also noted that other work has higher read depth per cell than this work and cite the recent study by Saint et al., 2019.</p><disp-quote content-type="editor-comment"><p>ii) &quot;We observe significant heterogeneity in individual cells.… Much of this variation can be explained by the mitotic cell cycle&quot; – that statement is not true, there does not seem to be an attempt to quantify heterogeneity over most genes. That they see heterogeneity in cell cycle stage as expected does not mean that cell cycle stage explains heterogeneity in the rest of the response, which was not reported on here.</p></disp-quote><p>We provide plots of descriptive measures of variability as Figure 2—figure supplement 4. Our experimental design does not allow for a biological interpretation of these results as we are currently not able to distinguish technical variability from biological variability on a per-gene basis. We have included additional discussion about the potential for heterogeneity in some growth conditions; Figure 3—figure supplement 2 suggests that there are some cells undergoing different responses to stressful conditions. Although we are cautious about the capacity for our experimental to distinguish between biological heterogeneity and technical heterogeneity on a per-gene basis we provide these results as an additional example of the potential power of our approach.</p><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>[…] The authors provide a complete and user-friendly dataset (deposited and interactive through a shiny app) that will be a valuable resource for the yeast community. All in all, I think this is a very solid manuscript that with minor corrections I would strongly support for publication in eLife.</p><p>Beyond the development of the method, the main and most original point of the manuscript is that the authors aim to generate infer gene regulatory networks based on the transcriptomes retrieved from barcoded genotypes in diverse conditions. I have some questions and comments (of varying levels of concern) that I feel should be addressed in the current version of the manuscript.</p><p>The authors leverage in their previous experience in GRN reconstruction and propose this approach has allowed them to discover novel regulatory relationships between cell cycle-regulated gene expression in response to changes in nitrogen source. In my opinion, this part of the manuscript needs to be reinforced and some of the conclusions driven from the GRN should be and some representative novel regulatory relationships experimentally demonstrated. As well the authors should provide context to their findings as they often read a bit disconnected.</p></disp-quote><p>We now include a comparison of the network which we have learned to regulatory interactions defined in the YEASTRACT database that we did not provide to our network inference method as priors. Of the 6,114 new interactions identified in our network, more than half have evidence of an existing regulatory relationship in YEASTRACT. While additional experimentation would certainly provide additional evidence for individual interactions, we feel that this result validates the overall network. Dissection of individual interactions and further functional characterization of novel regulatory relationships between the cell cycle and nitrogen metabolism is the subject of future work.</p><disp-quote content-type="editor-comment"><p>In terms of data quality, the number of mitochondrial and ribosomal reads per genotype and condition should be plotted, as a quality metric and given that a lot of ribosomal gene expression is cell cycle regulated. This could be relevant to understand separate clusters within conditions which the authors do not mention in the results and/or discussion and given the fact that different zymolyase concentrations were used for cell lysis.</p></disp-quote><p>This is an excellent suggestion. We performed this analysis and have added Figure 2—figure supplement 3 which contains total counts, ribosomal genes, ribosomal biogenesis genes, induced environmental stress response genes, and mitochondrial genome genes overlaid over the UMAP plots.</p><disp-quote content-type="editor-comment"><p>The optical density of the cells at the harvesting (besides the total cell number) should be provided to ease the reproducibility for other labs.</p></disp-quote><p>Cell concentrations were determined by cell counting (not OD); we have revised the Materials and methods section to include the concentrations at harvest (the DIAUXY cell density is not available; cells were harvested based on glucose concentration in media and the cell density was only measured after resuspension in RNAlater. Back calculation from this value suggests a culture density of ~1.0 x 10<sup>8</sup> cells/mL).</p><disp-quote content-type="editor-comment"><p>If I understand correctly, cell cycle clusters within conditions in Figure 3 is confusing. For example expression of DSE2 or PIR1 seem to be highest in the green and grey clusters respectively (Figure 3A) however in Figure 3B panel I the highest expression is assigned to the grey-yellow for DSE2 and PIR1 to the green cluster.</p></disp-quote><p>We thank the reviewer for pointing out this mistake. The labeling on Figure 3B was incorrect; the DSE2 and PIR1 labels were swapped (they were correct in Figure 3—figure supplement 1B). This error has been fixed.</p><disp-quote content-type="editor-comment"><p>As well, the UMAPS from Figure 3A, the authors claim the clustering is mainly condition-dependent and genotype-independent. However, conditions like MMEtOH, CSTARVE, NLIM-PRO have clear clusters that do not seem cell cycle-dependent and these might be biologically relevant. The authors should at least comment or those or run a DE analysis to see what these are.</p></disp-quote><p>This is an excellent suggestion. We have relabeled Figure 3A with expression of gene categories and included it as Figure 3—figure supplement 2. It appears that several stressful conditions have clusters that are upregulated for environmental stress response genes and downregulated for ribosomal and cell-cycle genes. The most likely explanation is that in higher-stress conditions, we are observing some cells that are in a quiescent state. We propose this possibility in the main text while cautioning that the static nature of our experimental design limits our ability to conclusively demonstrate this.</p><disp-quote content-type="editor-comment"><p>How do the newly generated data compare to Gasch et al., 2017 and Nadal-Ribelles et al., 2019?</p></disp-quote><p>We have compared our scRNAseq data to several published expression datasets in Figure 2—figure supplement 2. We find reasonable agreement with Nadal-Ribelles et al., 2019 despite differences in strain and experimental technique. The agreement with data form Gasch et al., 2017 is less similar, likely reflecting differences in genetic background, growth conditions, and experimental technique. These comparisons highlight the challenge of integrating single-cell data from separate experiments. However, several recent papers have explored integration of single-cell data (e.g. Butler et al., 2018, Stuart et al., 2019). One of the main long-term advantages of our work (that we will explore in the future) is that it could be applied to integrate multiple data sets that have been prepared differently into unified network models.</p><disp-quote content-type="editor-comment"><p>The number of differentially expressed genes even in the YPD condition seems a bit low. How do these pseudobulk compare to the deletome data (Holstege lab) or other published datasets?</p></disp-quote><p>The deletome data from Holstege 2014 was generated in synthetic complete media (the closest comparison in our data set is minimal media). The TFs which we have focused on are in pathways known to be dysregulated in auxotrophic strains like the BY4741 family used as a basis for the yeast deletion collection (the auxotrophies are all deletions in nitrogen anabolic pathways; our work does not use these strains, which makes direct comparisons to the standard yeast deletion collection more challenging).</p><disp-quote content-type="editor-comment"><p>As well, the authors use DESeq in Figure 4B, but these results are contradictory with what is shown in Figure S4Bi which is done by Welch testing. This is a bit confusing and does not add much to the reader. I would suggest to run DE between conditions using DESeq or provide the reasoning as to why these two different approaches are used and done?</p></disp-quote><p>We agree with the reviewer that this analysis adds nothing to this work and is a source of confusion. Therefore, we have removed the t-test based figures.</p><disp-quote content-type="editor-comment"><p>Why do transcription factor activities FKH1 and FKH2 and SWI4 SWI5 do not overlap (they almost seem mutually exclusive), one would expect them to have similar profiles. Similarly, NDD1 regulates S-phase genes but the TFA does not overlap with the HTB expression shown in Figure 3.</p></disp-quote><p>We thank the reviewer for identifying this inconsistency. While looking into this, we identified an issue with the modeling that disproportionately affected genes with extremely low average values in one modeling task resulting in them highly influencing certain TF activities. We have corrected this problem by adding a filter to remove genes with low average expression, and we have replaced NDD1 in this figure with MBP1, another cell-cycle TF.</p><p>SWI4 and SWI5 are not currently annotated on SGD as acting in the same stage of the cell cycle (SWI4 is annotated as late G1/S and SWI5 is annotated as early G1/M), and so we don’t expect them to overlap. TFs with functional redundancy like FHK1 and FHK2 would be expected to overlap, but the redundancy makes modeling regulatory relationships more challenging (this problem is one of the main reasons we have chosen nitrogen metabolism as our model, as it is also full of functional redundancies and cross-regulatory loops). Methodological approaches to functional redundancies is the subject of ongoing work.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>[…] I have some comments, which I think can strengthen the messages of the paper.</p><p>1) In Figure 5C, is the single task network inferred by merging all the data and learning a single network or by learning separate networks with single tasking and aggregating the results? If not, how does the single task per condition followed by aggregation perform?</p><p>4) A comparison of the AUPR of the single task condition-specific networks and the multi-task condition specific network could further show the advantage of the multi-task learning framework.</p></disp-quote><p>Both of these (point 1 and point 4) are excellent suggestions. Initially, Figure 5D was comparing merged data learning a single network to AMuSR. Figure 5D now compares all data combined to learn a single-task network [BBSR (ALL)], all data learned separately with single-tasking and then aggregated into a single network [BBSR (BY TASK)], and all data which is learned together with multitask learning [AMuSR (MTL)]. In addition, the second panel of 5D has been replaced with a comparison of the task-specific networks learned individually with BBSR (labeled as [BBSR (BY TASK)]), and the task-specific networks learned jointly with AMuSR (labeled as [AMuSR (MTL)]). We find that the majority of the performance gained by MTL is due to separating individual conditions; sharing information during regression improves performance on some task-specific networks, but decreases performance on others (overall the effect balances out on the final aggregate network).</p><disp-quote content-type="editor-comment"><p>2) The authors don't get much into the context-specificity of the inferred networks. They interpret only the final aggregated network. It would be useful to know how similar the individual condition-specific networks are and if there is a conserved core used by multiple conditions. The only comparison of context specificity is being done at the level of AUPR, it might be helpful to do this comparison just by comparing the inferred networks.</p></disp-quote><p>This is an excellent point. The composition of the final, aggregate network is examined in Figure 6—figure supplement 1C-E. We find that there does exist a common core of interaction which we recover in every task, most of which are in the prior. Much of the learned network corresponds to interactions found in multiple condition-specific networks, but we also find a large number of interactions that are unique to a single condition-specific network (when we examine individual condition networks we find a large number of TF-gene interactions which are present in only one network; most of these are not included in the final, aggregate network, but some are). We think that this observation provides a strong justification for learning networks from separate growth conditions with distinct transcriptional profiles and have emphasized this point in our revised manuscript.</p><disp-quote content-type="editor-comment"><p>3) Some discussion about the variation in the AUPRs would be helpful. Is it because the gold standard is biased towards the conditions on which the AUPR is high. It seems the AUPR for the MMEtOH network is close to what is inferred by the multi-task learning, and some explanation of why this might be is helpful.</p></disp-quote><p>This is a great point. We think that the variation is mainly due to cells in certain conditions requiring more of their transcriptome to be active. Cells in rich YPD media have mostly glycolytic, cell growth, and cell cycle expression programs active, whereas cells in minimal MMD media require all of these processes plus anabolic pathways for a number of essential compounds. Cells in MMEtOH can turn off glycolysis but need respiration and redox-related pathways to be active. We have added this interpretation to the main text.</p><disp-quote content-type="editor-comment"><p>5) It would be helpful to emphasize if and how the multi-task learning approach used here was from extended from the Castro, 2019 paper.</p></disp-quote><p>Although there are some changes in implementation, the multi-task learning approach here is the same conceptually as the technique introduced in Castro et al., 2019. The most important difference is that we’ve chosen not to weight prior interactions differently than interactions that are not in the prior during modeling; priors in this work are only used to calculate transcription factor activity. We have clarified this in the text.</p><disp-quote content-type="editor-comment"><p>6) The Discussion could be strengthened. The authors present some results about the interplay between cell cycle and nitrogen response. It was not clear why this is interesting to study beyond that there is a shared regulatory program. This might be worth bringing up in the Discussion to tie back to the initial goal of inferring a network for nitrogen metabolism and the TOR signaling pathway and the general role of cell cycle and stress response.</p></disp-quote><p>We have added a paragraph to our Discussion to more explicitly discuss the relationship between nitrogen metabolism, TOR signaling and the cell cycle.</p><disp-quote content-type="editor-comment"><p>7) The authors don't find a substantial impact of TF knockout on gene expression under different conditions (I assume the comparisons were done while controlling for the conditions). How does this compare to bulk data? How much of this observation could be due to the sparsity of scRNAseq data versus the redundancy of TFs.</p></disp-quote><p>Our new experiments show scRNAseq data is in good agreement with bulk RNAseq data. We think that the limited impact of TF knockouts on expression is consistent with earlier published studied (e.g. the Holstege deletome study). Therefore, we think that the primary limitations of TF deletions are 1) functional redundancy and 2) studying expression in static rather than dynamic conditions. We now include these points in our Discussion and propose alternative approaches that might elicit stronger transcriptional responses (e.g. inducible overexpression of TFs).</p></body></sub-article></article>