<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">85537</article-id><article-id pub-id-type="doi">10.7554/eLife.85537</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Advance</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Structural screens identify candidate human homologs of insect chemoreceptors and cryptic <italic>Drosophila</italic> gustatory receptor-like proteins</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name><surname>Benton</surname><given-names>Richard</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4305-8301</contrib-id><email>Richard.Benton@unil.ch</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes" id="author-300950"><name><surname>Himmel</surname><given-names>Nathaniel J</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7876-6960</contrib-id><email>nathanieljohn.himmel@unil.ch</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/019whta54</institution-id><institution>Center for Integrative Genomics, Faculty of Biology and Medicine, University of Lausanne</institution></institution-wrap><addr-line><named-content content-type="city">Lausanne</named-content></addr-line><country>Switzerland</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Desplan</surname><given-names>Claude</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0190ak572</institution-id><institution>New York University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Desplan</surname><given-names>Claude</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0190ak572</institution-id><institution>New York University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>20</day><month>02</month><year>2023</year></pub-date><pub-date pub-type="collection"><year>2023</year></pub-date><volume>12</volume><elocation-id>e85537</elocation-id><history><date date-type="received" iso-8601-date="2022-12-15"><day>15</day><month>12</month><year>2022</year></date><date date-type="accepted" iso-8601-date="2023-02-16"><day>16</day><month>02</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at .</event-desc><date date-type="preprint" iso-8601-date="2022-12-15"><day>15</day><month>12</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.12.13.519744"/></event></pub-history><permissions><copyright-statement>© 2023, Benton and Himmel</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Benton and Himmel</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-85537-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-85537-figures-v2.pdf"/><related-article related-article-type="article-reference" ext-link-type="doi" xlink:href="10.7554/eLife.62507" id="ra1"/><abstract><p>Insect odorant receptors and gustatory receptors define a superfamily of seven transmembrane domain ion channels (referred to here as 7TMICs), with homologs identified across Animalia except Chordata. Previously, we used sequence-based screening methods to reveal conservation of this family in unicellular eukaryotes and plants (DUF3537 proteins) (Benton et al., 2020). Here, we combine three-dimensional structure-based screening, ab initio protein folding predictions, phylogenetics, and expression analyses to characterize additional candidate homologs with tertiary but little or no primary structural similarity to known 7TMICs, including proteins in disease-causing <italic>Trypanosoma</italic>. Unexpectedly, we identify structural similarity between 7TMICs and PHTF proteins, a deeply conserved family of unknown function, whose human orthologs display enriched expression in testis, cerebellum, and muscle. We also discover divergent groups of 7TMICs in insects, which we term the gustatory receptor-like (Grl) proteins. Several <italic>Drosophila melanogaster Grl</italic>s display selective expression in subsets of taste neurons, suggesting that they are previously unrecognized insect chemoreceptors. Although we cannot exclude the possibility of remarkable structural convergence, our findings support the origin of 7TMICs in a eukaryotic common ancestor, counter previous assumptions of complete loss of 7TMICs in Chordata, and highlight the extreme evolvability of this protein fold, which likely underlies its functional diversification in different cellular contexts.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>chemosensory receptor</kwd><kwd>ion channel</kwd><kwd>protein structure</kwd><kwd>phylogenetics</kwd><kwd>comparative genomics</kwd><kwd>insect</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>D. melanogaster</italic></kwd><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100010663</institution-id><institution>H2020 European Research Council</institution></institution-wrap></funding-source><award-id>833548</award-id><principal-award-recipient><name><surname>Benton</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100001711</institution-id><institution>Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung</institution></institution-wrap></funding-source><award-id>310030B-185377</award-id><principal-award-recipient><name><surname>Benton</surname><given-names>Richard</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100000854</institution-id><institution>Human Frontier Science Program</institution></institution-wrap></funding-source><award-id>LT-0003/2022-L</award-id><principal-award-recipient><name><surname>Himmel</surname><given-names>Nathaniel J</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>A new screening strategy for divergent homologs of insect odorant and gustatory receptors, based upon predicted three-dimensional structural similarity, unexpectedly identifies candidates in humans.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The insect chemosensory receptor repertoires of odorant receptors (Ors) and gustatory receptors (Grs) define a highly divergent family of ligand-gated ion channels, which underlie these animals’ ability to respond to chemical cues in the external world (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib32">Joseph and Carlson, 2015</xref>; <xref ref-type="bibr" rid="bib66">Robertson, 2019</xref>). Despite its vast size and functional importance, this family has long been an evolutionary enigma, displaying no resemblance to other classes of ion channels. Indeed, for many years, insect Ors and Grs were thought to be an invertebrate-specific protein class (<xref ref-type="bibr" rid="bib3">Benton, 2006</xref>; <xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>). This view changed in the past decade, with the sequencing of a large number of genomes enabling the identification of homologs across animals (generally termed Gr-like [GRL] proteins), including non-Bilateria (e.g., the sea anemone <italic>Nematostella vectensis</italic>), Hemichordata (e.g., the sea acorn <italic>Saccoglossus kowalevskii</italic>), various unicellular eukaryotes (e.g., the chytrid fungus <italic>Spizellomyces punctatus</italic> and the alga <italic>Vitrella brassicaformis</italic>) and Plantae (known as Domain of Unknown Function [DUF] 3537 proteins) (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>; <xref ref-type="bibr" rid="bib65">Robertson, 2015</xref>; <xref ref-type="bibr" rid="bib70">Saina et al., 2015</xref>). For simplicity in nomenclature, we term here this broader superfamily (i.e., Ors, Grs, GRLs, and DUF3537 proteins) as ‘seven transmembrane domain ion channels’ (7TMICs), to distinguish them from unrelated 7TM G protein-coupled receptors. (We acknowledge that in most cases we do not know yet whether they are ion channels, and leave open the possibility for future updates to nomenclature.) Despite extensive searching, 7TMIC homologs have not been identified in Chordata, leading to proposals that these proteins were lost at or near the base of the chordate lineage (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib65">Robertson, 2015</xref>; <xref ref-type="bibr" rid="bib70">Saina et al., 2015</xref>).</p><p>A substantial challenge in identifying 7TMIC homologs is their extreme sequence divergence (as little as 8% amino acid identity). The inclusion of proteins in this family relies primarily on the presence of topological features, notably seven TM domains and an intracellular N-terminus (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>; <xref ref-type="bibr" rid="bib4">Benton et al., 2006</xref>). Although insect Grs were originally recognized as possessing a short, conserved motif in transmembrane domain 7 (TM7) (described below) (<xref ref-type="bibr" rid="bib66">Robertson, 2019</xref>; <xref ref-type="bibr" rid="bib73">Scott et al., 2001</xref>), this motif is only partially or not at all conserved outside insects (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>). For many protein families, the tertiary (three-dimensional) structure is generally more conserved than primary structure (<xref ref-type="bibr" rid="bib30">Illergård et al., 2009</xref>; <xref ref-type="bibr" rid="bib53">Murzin et al., 1995</xref>), and this property can offer an orthogonal strategy to identify homologous proteins. For the 7TMIC superfamily, the recent cryo-electronic microscopic (cryo-EM) structures of homotetrameric complexes of insect Ors (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>) provide important experimental insight into the tertiary structure of these proteins (as well as mechanistic insights into how these ion channels function). In our previous study (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>), we used ab initio structural predictions of candidate 7TMIC sequences to reinforce our proposals of homology despite extremely low amino acid identity.</p><p>The recent breakthroughs in accuracy (to atomic level) and speed (seconds-to-minutes per sequence) of protein structure predictions, notably by AlphaFold2 (<xref ref-type="bibr" rid="bib33">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib77">Varadi et al., 2022</xref>), have now enabled millions of protein models to be generated. Here, we have exploited the unprecedented resource of the AlphaFold Protein Structure Database (<xref ref-type="bibr" rid="bib33">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib77">Varadi et al., 2022</xref>) and the Dali protein structure comparison algorithm (<xref ref-type="bibr" rid="bib28">Holm, 2022</xref>), to screen for additional 7TMIC homologs by virtue of their tertiary structural similarity to experimentally determined insect Or structures.</p></sec><sec id="s2" sec-type="results|discussion"><title>Results and discussion</title><sec id="s2-1"><title>Tertiary structure-based screening for candidate 7TMIC homologs</title><p>Cryo-EM structures of two insect Ors have been obtained: the fig wasp (<italic>Apocrypta bakeri</italic>) Or co-receptor (Orco) (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="fig" rid="fig1">Figure 1A–B</xref>), which is a highly conserved member of the repertoire across most insect species (<xref ref-type="bibr" rid="bib4">Benton et al., 2006</xref>; <xref ref-type="bibr" rid="bib31">Jones et al., 2005</xref>; <xref ref-type="bibr" rid="bib39">Larsson et al., 2004</xref>) and MhOr5 from the jumping bristletail (<italic>Machilis hrabei</italic>), a broadly tuned volatile sensor (<xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>). Despite sharing only 18% amino acid identity, these proteins adopt a highly similar fold (<xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>). As Orco shows higher sequence similarity to Grs – the ancestral family of insect chemosensory receptors from which Ors derived (<xref ref-type="bibr" rid="bib8">Brand et al., 2018</xref>; <xref ref-type="bibr" rid="bib18">Dunipace et al., 2001</xref>; <xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>) – we used <italic>A. bakeri</italic> Orco as the query structure in our analysis.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Structure-based screening for seven transmembrane domain ion channel (7TMIC) homologs.</title><p>(<bold>A</bold>) Top view of a cryo-electronic microscopic (cryo-EM) structure of the homotetramer of Or co-receptor (Orco) from <italic>A. bakeri</italic> (derived from PDB 6C70; <xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>), in which one subunit has a spectrum coloration (N-terminus [blue] to C-terminus [red]). The ion channel pore is formed at the interface of the four subunits. A side view is shown below. The anchor domain, comprising the cytoplasmic projections of TM4-6 and TM7a, forms most of the inter-subunit interactions in odorant receptors (Ors) (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>). (<bold>B</bold>) Top: output of transmembrane topology predictions of DeepTMHMM (<xref ref-type="bibr" rid="bib25">Hallgren et al., 2022</xref>) for <italic>A. bakeri</italic> Orco. Bottom: schematic of the membrane topology of an Orco monomer, with the same spectrum coloration as in (<bold>A</bold>), reproduced from Figure 1a from <xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>. Note that the seventh predicted helical region is divided into two in the cryo-EM structure: TM7a (located in the cytosol) and TM7b (located in the membrane). (<bold>C</bold>) Comparisons of side and top views of the cryo-EM structure of an <italic>A. bakeri</italic> Orco subunit (6C70-A) (left) and an AlphaFold2 protein structure prediction of <italic>A. bakeri</italic> Orco. Helical regions are numbered in the top views. Note the model contains the extracellular loop 2 (EL2) and intracellular loop 2 (IL2) regions that were not able to be accurately visualized in the cryo-EM structure (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>). Quantitative comparisons of structures are provided in <xref ref-type="table" rid="table1">Table 1</xref>. (<bold>D</bold>) Summary of the results of the screen for Orco/Or-like protein folds in the AlphaFold Protein Structure Database for the indicated species using Dali (<xref ref-type="bibr" rid="bib28">Holm, 2022</xref>). The threshold of Dali Z-score &gt;10 was informed by inspection of the results of the screen (see Results). Raw outputs of the screen are provided in <xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>. (<bold>E</bold>) Top: transmembrane topology predictions of the single screen hits from the <italic>Trypanosoma</italic> species <italic>Leishmania infantum</italic> and <italic>Trypanosoma brucei brucei</italic>. Bottom: AlphaFold2 structural models of these proteins, displayed as in (<bold>C</bold>). The long N-terminal region contains tandem Membrane Occupation and Recognition Nexus (MORN) repeats and sequence of unknown structure (gray); these are masked in the top view of the models. (<bold>F</bold>) Visual comparison of the <italic>L. infantum</italic> GRL1 AlphaFold2 model (the N-terminal region is masked) with the <italic>A. bakeri</italic> Orco structure, aligned with Coot (<xref ref-type="bibr" rid="bib21">Emsley et al., 2010</xref>). Quantitative comparisons of structures are provided in <xref ref-type="table" rid="table1">Table 1</xref>. (<bold>G</bold>) Consensus phylogeny of putative trypanosome homologs. The primary sequence database was assembled using <italic>L. infantum</italic> GRL1 (XP_001464500.1) and <italic>T. brucei brucei</italic> GRL1 (XP_845058.1) as query sequences (highlighted in bold). Branch support values refer to maximum likelihood UFboot/Bayesian posterior probabilities. Note that although the <italic>Trypanosoma cruzi</italic> homolog (XP_803355.1) was not identified in the original Dali screen, visual inspection of the corresponding AlphaFold2 model (A0A2V2WL40) revealed the same global fold.</p><p><supplementary-material id="fig1sdata1"><label>Figure 1—source data 1.</label><caption><title>FASTA file containing the amino acid sequences for validated trypanosome GRLs used in phylogenetic analyses.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig1-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig1sdata2"><label>Figure 1—source data 2.</label><caption><title>FASTA file containing the multiple sequence alignment of trypanosome GRLs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig1-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig1sdata3"><label>Figure 1—source data 3.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of trypanosome GRLs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig1-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig1sdata4"><label>Figure 1—source data 4.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of trypanosome GRLs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig1-data4-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig1-v2.tif"/></fig><p>In our previous work (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>), we generated ab initio protein models of Orco and candidate homologs in various unicellular eukaryotes using trRosetta (<xref ref-type="bibr" rid="bib79">Yang et al., 2020</xref>) and RaptorX (<xref ref-type="bibr" rid="bib35">Källberg et al., 2012</xref>). We therefore first examined the AlphaFold2 structural model of <italic>A. bakeri</italic> Orco (<xref ref-type="fig" rid="fig1">Figure 1C</xref>; <xref ref-type="bibr" rid="bib33">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib77">Varadi et al., 2022</xref>). This model displays striking qualitative similarity to the experimental structure (PDB 6C70 chain A) (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). We assessed structural similarity quantitatively using two algorithms: first, using pairwise structural alignment in Dali (<xref ref-type="bibr" rid="bib28">Holm, 2022</xref>), we extracted the resultant Z-score (the sum of equivalent residue-wise C<sub>α</sub>-C<sub>α</sub> distances between two proteins); second, we determined the template modeling (TM)-score from TM-align (<xref ref-type="bibr" rid="bib80">Zhang and Skolnick, 2004</xref>; <xref ref-type="bibr" rid="bib81">Zhang and Skolnick, 2005</xref>) (a measure of the global similarity of full-length proteins) (<xref ref-type="table" rid="table1">Table 1</xref>). These measures confirmed the visual impression that the modeled and experimental structures are almost identical (e.g., TM-score=0.96, where 1 would be a perfect match). We extended our assessment of available (or newly generated) AlphaFold2 models to other well-established members of the 7TMIC family from animals as well as much more divergent unicellular 7TMIC homologs previously identified (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>; <xref ref-type="supplementary-material" rid="sdata1">Source data 1</xref>). Using the same quantitative assessments, these all displayed substantial tertiary structural similarity to <italic>A. bakeri</italic> Orco (<xref ref-type="table" rid="table1">Table 1</xref>), reinforcing our previous conclusions that these proteins form part of the same superfamily. Moreover, the observation that multiple distinct algorithms (AlphaFold2, trRosetta, and RaptorX) predict the same global fold of these proteins strengthens confidence in the validity of ab initio structural models.</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Quantitative structural comparisons of candidate seven transmembrane domain ion channel (7TMIC) homologs.</title><p>Summary of amino acid identity (%), Dali Z-score, and TM-align TM-score of the indicated experimentally determined or ab initio-predicted structures of 7TMIC homologs (or negative-control, unrelated proteins) compared to <italic>A. bakeri</italic> Or co-receptor (Orco). The Orco cryo-electronic microscopic (cryo-EM) structure chain A (6C70-A) (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>) was used as the query in all comparisons. Protein models are provided in <xref ref-type="supplementary-material" rid="sdata1">Source data 1</xref>. Note the nomenclature of unicellular eukaryotic 7TMICs is tentative; identical names (e.g., GRL1) do not imply orthology. Typically, a Z-score &gt;20 indicates that the two proteins being compared are definitely homologous, 8–20 that they are probably homologous, and 2–8 is a ‘gray area’ influenced by protein size and fold (<xref ref-type="bibr" rid="bib27">Holm, 2020</xref>). TM-scores of 0.5–1 indicate that the two proteins being compared adopt generally the same fold, while TM-scores of 0–0.3 indicate random structural similarity (<xref ref-type="bibr" rid="bib80">Zhang and Skolnick, 2004</xref>; <xref ref-type="bibr" rid="bib81">Zhang and Skolnick, 2005</xref>). For the negative controls, the amino acid identity differs slightly between the experimentally determined and ab initio<italic>-</italic>predicted proteins because of small differences in sequence coverage.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom" rowspan="2">Category</th><th align="left" valign="bottom" rowspan="2">Protein</th><th align="left" valign="bottom" rowspan="2">Model or PDB</th><th align="left" valign="bottom" rowspan="2">Method or algorithm</th><th align="left" valign="bottom" colspan="3">Comparison to <italic>A. bakeri</italic> Orco (6C70-A)</th></tr><tr><th align="left" valign="bottom">Amino acid identity (%)</th><th align="left" valign="bottom">DaliZ-score</th><th align="left" valign="bottom">TM-alignTM-score</th></tr></thead><tbody><tr><td align="left" valign="bottom" rowspan="4">Positive<break/>controls (known 7TMIC)</td><td align="left" valign="bottom"><italic>A. bakeri</italic> Orco</td><td align="left" valign="bottom">61b81_unrelaxed_rank_1_model_2</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">100</td><td align="char" char="." valign="bottom">50.7</td><td align="char" char="." valign="bottom">0.96</td></tr><tr><td align="left" valign="bottom"><italic>M. hrabei</italic> Or5</td><td align="char" char="hyphen" valign="bottom">7LIC-A</td><td align="left" valign="bottom">Cryo-EM</td><td align="char" char="." valign="bottom">19</td><td align="char" char="." valign="bottom">36.3</td><td align="char" char="." valign="bottom">0.81</td></tr><tr><td align="left" valign="bottom"><italic>Drosophila melanogaster</italic> Gr64a</td><td align="left" valign="bottom">AF-P83293-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">13</td><td align="char" char="." valign="bottom">29.6</td><td align="char" char="." valign="bottom">0.79</td></tr><tr><td align="left" valign="bottom"><italic>N. vectensis</italic> GRL1</td><td align="left" valign="bottom">AF-A7S7G0-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">31.3</td><td align="char" char="." valign="bottom">0.78</td></tr><tr><td align="left" valign="bottom" rowspan="16">Unicellular eukaryotic 7TMIC</td><td align="left" valign="bottom"><italic>Thecamonas trahens</italic> GRL1</td><td align="left" valign="bottom">AF-A0A0L0DUY0-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">23.2</td><td align="char" char="." valign="bottom">0.71</td></tr><tr><td align="left" valign="bottom"><italic>T. trahens</italic> GRL2</td><td align="left" valign="bottom">AF-A0A0L0DQC1-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">12</td><td align="char" char="." valign="bottom">25.3</td><td align="char" char="." valign="bottom">0.70</td></tr><tr><td align="left" valign="bottom"><italic>T. trahens</italic> GRL3</td><td align="left" valign="bottom">AF-A0A0L0D5B5-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">14</td><td align="char" char="." valign="bottom">13.1</td><td align="char" char="." valign="bottom">0.50</td></tr><tr><td align="left" valign="bottom"><italic>T. trahens</italic> GRL4</td><td align="left" valign="bottom">AF-A0A0L0D5H0-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">9.9</td><td align="char" char="." valign="bottom">0.53</td></tr><tr><td align="left" valign="bottom"><italic>T. trahens</italic> GRL5</td><td align="left" valign="bottom">AF-A0A0L0DD38-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">12.2</td><td align="char" char="." valign="bottom">0.56</td></tr><tr><td align="left" valign="bottom"><italic>T. trahens</italic> GRL6</td><td align="left" valign="bottom">AF-A0A0L0DJ52-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">15.6</td><td align="char" char="." valign="bottom">0.57</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL1</td><td align="left" valign="bottom">AF-A0A0G4FIT4-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">9.1</td><td align="char" char="." valign="bottom">0.47</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL2</td><td align="left" valign="bottom">AF-A0A0G4ECU2-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">11</td><td align="char" char="." valign="bottom">14.4</td><td align="char" char="." valign="bottom">0.57</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL3</td><td align="left" valign="bottom">AF-A0A0G4FWI7-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">14</td><td align="char" char="." valign="bottom">23.8</td><td align="char" char="." valign="bottom">0.74</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL4</td><td align="left" valign="bottom">AF-A0A0G4EU86-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">18.5</td><td align="char" char="." valign="bottom">0.70</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL5</td><td align="left" valign="bottom">AF-A0A0G4FBY6-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">18.5</td><td align="char" char="." valign="bottom">0.68</td></tr><tr><td align="left" valign="bottom"><italic>V. brassicaformis</italic> GRL6</td><td align="left" valign="bottom">AF-A0A0G4G8W6-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">21.4</td><td align="char" char="." valign="bottom">0.70</td></tr><tr><td align="left" valign="bottom"><italic>Micromonas pusilla</italic> GRL1</td><td align="left" valign="bottom">AF-C1MGH9-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">12</td><td align="char" char="." valign="bottom">11.3</td><td align="char" char="." valign="bottom">0.60</td></tr><tr><td align="left" valign="bottom"><italic>Chloropicon primus</italic> GRL1</td><td align="left" valign="bottom">AF-A0A5B8MFA4-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">18.1</td><td align="char" char="." valign="bottom">0.71</td></tr><tr><td align="left" valign="bottom"><italic>L. infantum</italic> GRL1</td><td align="left" valign="bottom">AF-A4HWQ9-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">6</td><td align="char" char="." valign="bottom">13.5</td><td align="char" char="." valign="bottom">0.64</td></tr><tr><td align="left" valign="bottom"><italic>T. brucei</italic> GRL1</td><td align="left" valign="bottom">AF-Q57U78-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">13.4</td><td align="char" char="." valign="bottom">0.62</td></tr><tr><td align="left" valign="bottom" rowspan="10">Fly Grl</td><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl36a</td><td align="left" valign="bottom">AF-Q8INZ1-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">19.5</td><td align="char" char="." valign="bottom">0.67</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl36b</td><td align="left" valign="bottom">AF-Q8INY2-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">15.2</td><td align="char" char="." valign="bottom">0.62</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl40a</td><td align="left" valign="bottom">AF-Q0E8M7-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">19.5</td><td align="char" char="." valign="bottom">0.66</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl43a</td><td align="left" valign="bottom">AF-Q9V4Q0-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">19.9</td><td align="char" char="." valign="bottom">0.69</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl58a</td><td align="left" valign="bottom">AF-Q9W2A4-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">15.0</td><td align="char" char="." valign="bottom">0.60</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl62a</td><td align="left" valign="bottom">AF-B7Z0I0-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">19.4</td><td align="char" char="." valign="bottom">0.69</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl62b</td><td align="left" valign="bottom">AF-B7Z0I1-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">11</td><td align="char" char="." valign="bottom">19.1</td><td align="char" char="." valign="bottom">0.66</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl62c</td><td align="left" valign="bottom">AF-Q6ILZ2-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">17.2</td><td align="char" char="." valign="bottom">0.63</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Grl65a</td><td align="left" valign="bottom">AF-Q8IQ72-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">11</td><td align="char" char="." valign="bottom">25.9</td><td align="char" char="." valign="bottom">0.74</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> GrlHz</td><td align="left" valign="bottom">AF-Q9W1W8-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">7</td><td align="char" char="." valign="bottom">22.5</td><td align="char" char="." valign="bottom">0.74</td></tr><tr><td align="left" valign="bottom" rowspan="3">PHTF</td><td align="left" valign="bottom"><italic>Homo sapiens</italic> PHTF1</td><td align="left" valign="bottom">AF-Q9UMS5-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">7</td><td align="char" char="." valign="bottom">12.9</td><td align="char" char="." valign="bottom">0.63</td></tr><tr><td align="left" valign="bottom"><italic>H. sapiens</italic> PHTF2</td><td align="left" valign="bottom">AF-Q8N3S3-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">12.0</td><td align="char" char="." valign="bottom">0.62</td></tr><tr><td align="left" valign="bottom"><italic>D. melanogaster</italic> Phtf</td><td align="left" valign="bottom">AF-Q9V9A8-F1-model_v3</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">5</td><td align="char" char="." valign="bottom">11.8</td><td align="char" char="." valign="bottom">0.63</td></tr><tr><td align="left" valign="bottom" rowspan="16">Negative controls<break/>(non-7TMIC)</td><td align="left" valign="bottom" rowspan="2"><italic>Bos taurus</italic> Rhodopsin</td><td align="char" char="hyphen" valign="bottom">1F88-A</td><td align="left" valign="bottom">X-ray crystal</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">2.1</td><td align="char" char="." valign="bottom">0.31</td></tr><tr><td align="left" valign="bottom">AF-P02699-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">&lt;2.0</td><td align="char" char="." valign="bottom">0.19</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>Chlamydomonas reinhardtii</italic> ChR2</td><td align="char" char="hyphen" valign="bottom">6EID-A</td><td align="left" valign="bottom">X-ray crystal</td><td align="char" char="." valign="bottom">7</td><td align="char" char="." valign="bottom">3.6</td><td align="char" char="." valign="bottom">0.27</td></tr><tr><td align="left" valign="bottom">AF-Q8RUT8-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">3.4</td><td align="char" char="." valign="bottom">0.10</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>H. sapiens</italic> Frizzled4</td><td align="char" char="." valign="bottom">6BD4</td><td align="left" valign="bottom">X-ray crystal</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">4.0</td><td align="char" char="." valign="bottom">0.34</td></tr><tr><td align="left" valign="bottom">AF-Q9ULV1-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">5</td><td align="char" char="." valign="bottom">2.9</td><td align="char" char="." valign="bottom">0.19</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>H. sapiens</italic> AdipR</td><td align="char" char="." valign="bottom">5LXG</td><td align="left" valign="bottom">X-ray crystal</td><td align="char" char="." valign="bottom">2</td><td align="char" char="." valign="bottom">3.6</td><td align="char" char="." valign="bottom">0.29</td></tr><tr><td align="left" valign="bottom">AF-Q96A54-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">2</td><td align="char" char="." valign="bottom">&lt;2.0</td><td align="char" char="." valign="bottom">0.14</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>Escherichia coli</italic> GlpG</td><td align="char" char="." valign="bottom">2XOV</td><td align="left" valign="bottom">X-ray crystal</td><td align="char" char="." valign="bottom">5</td><td align="char" char="." valign="bottom">3.5</td><td align="char" char="." valign="bottom">0.27</td></tr><tr><td align="left" valign="bottom">AF-P09391-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">6</td><td align="char" char="." valign="bottom">3.3</td><td align="char" char="." valign="bottom">0.13</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>Mus musculus</italic> TRPV3</td><td align="char" char="hyphen" valign="bottom">6LGP-D</td><td align="left" valign="bottom">Cryo-EM</td><td align="char" char="." valign="bottom">10</td><td align="char" char="." valign="bottom">2.7</td><td align="char" char="." valign="bottom">0.27</td></tr><tr><td align="left" valign="bottom">AF-Q8K424-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">14</td><td align="char" char="." valign="bottom">2.3</td><td align="char" char="." valign="bottom">0.08</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>M. musculus</italic> Piezo</td><td align="char" char="hyphen" valign="bottom">6BPZ-B</td><td align="left" valign="bottom">Cryo-EM</td><td align="char" char="." valign="bottom">5</td><td align="char" char="." valign="bottom">4.0</td><td align="char" char="." valign="bottom">0.27</td></tr><tr><td align="left" valign="bottom">AF-E2JF22-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">5</td><td align="char" char="." valign="bottom">2.3</td><td align="char" char="." valign="bottom">0.08</td></tr><tr><td align="left" valign="bottom" rowspan="2"><italic>B. taurus</italic> CNGA/CNGB</td><td align="char" char="hyphen" valign="bottom">7O4H-A</td><td align="left" valign="bottom">Cryo-EM</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">2.8</td><td align="char" char="." valign="bottom">0.24</td></tr><tr><td align="left" valign="bottom">AF-Q00194-F1-model_v4</td><td align="left" valign="bottom">AlphaFold2</td><td align="char" char="." valign="bottom">9</td><td align="char" char="." valign="bottom">3.3</td><td align="char" char="." valign="bottom">0.11</td></tr></tbody></table></table-wrap><p>We proceeded to screen the AlphaFold Protein Structure Database for other proteins that are structurally similar to <italic>A. bakeri</italic> Orco using the hierarchical search function in Dali (<xref ref-type="bibr" rid="bib28">Holm, 2022</xref>). This algorithm currently permits pairwise alignment of Orco to the complete predicted structural proteomes of 47 species – encompassing several vertebrates, invertebrates, plants, unicellular eukaryotes and prokaryotes – returning hits ordered by Z-score (<xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>). We focused on those hits with a Z-score of &gt;10 (<xref ref-type="fig" rid="fig1">Figure 1D</xref>). This threshold successfully captured known 7TMICs, while removing a large number of proteins (generally with a much lower Z-score) that did not fulfill other criteria for structural similarity, as described below. Of the expected hits, within the <italic>D. melanogaster</italic> structural proteome we recovered all models of the members of the Or and Gr repertoires. From <italic>Caenorhabditis elegans</italic>, we found all members of the gustatory receptor (GUR) family (<xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>) – including the photoreceptor LITE-1 (formerly GUR-2) (<xref ref-type="bibr" rid="bib20">Edwards et al., 2008</xref>; <xref ref-type="bibr" rid="bib24">Gong et al., 2016</xref>; <xref ref-type="bibr" rid="bib44">Liu et al., 2010</xref>) – and the serpentine receptor R (SRR) family (which are of unknown function, but display diverse neuronal and non-neuronal expression patterns <xref ref-type="bibr" rid="bib78">Vidal et al., 2018</xref>; <xref ref-type="fig" rid="fig1">Figure 1D</xref> and <xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>). From the four plant species screened, all members of the DUF3537 family were successfully identified (<xref ref-type="fig" rid="fig1">Figure 1</xref> and <xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>). Inspection of several models below our Z-score threshold indicated that the proteins (typically multipass membrane proteins) have likely spurious resemblance to subregions of Orco rather than displaying similarity in their overall fold.</p><p>As will be illustrated below for individual novel candidate 7TMIC homologs, other hits were subsequently analyzed for their fulfillment of several criteria: (i) the presence of seven predicted TM domains, (ii) a predicted intracellular location of the N-terminus, and (iii) longer intracellular than extracellular loops (like insect Ors [<xref ref-type="bibr" rid="bib55">Otaki and Yamamoto, 2003</xref>], while also recognizing that intracellular loops can vary enormously in length in homologs [<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>]). For hits that fulfilled these criteria, ‘reverse’ searching of the <italic>D. melanogaster</italic> structural proteome with Dali was performed to verify that Ors and Grs were structurally the most similar proteins in this species (<xref ref-type="supplementary-material" rid="sdata3">Source data 3</xref>). We next qualitatively assessed the predicted tertiary structural similarity to <italic>A. bakeri</italic> Orco (<xref ref-type="fig" rid="fig1">Figure 1A–C</xref>; <xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>), verifying: (i) the characteristic packing of the TMs, (ii) the projection of the long TM4, TM5, and TM6 below the main bundle of helices (forming the ‘anchor’ domain where most inter-subunit contacts occur in complexes; <xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>), and (iii) the exceptional splitting of TM7 into two subregions (TM7a, part of the anchor domain, and TM7b, which lines the ion conduction pathway; <xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>). Structures were also quantitatively compared to <italic>A. bakeri</italic> Orco, as described above (<xref ref-type="table" rid="table1">Table 1</xref>). As negative controls, we also performed comparisons with a variety of other multipass membrane proteins belonging to other superfamilies, including several with seven TMs (e.g., Rhodopsin, Frizzled, and the Adiponectin receptor) (<xref ref-type="table" rid="table1">Table 1</xref>). The new candidate homologs all displayed quantitative measures of similarity that were within the range of previously identified 7TMIC homologs, and clearly superior to the scores of negative control proteins (<xref ref-type="table" rid="table1">Table 1</xref>). We now present these candidate homologs from different species and the potential evolutionary and biological implications for the 7TMIC family, bearing in mind the caveat that some of these may represent cases of structural convergence (discussed below).</p><p>Extending our previous discovery of 7TMICs in various single-celled eukaryotes (informally grouped here under the term Protozoa) (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>), we identified single proteins in two species belonging to the Trypanosomatida order: <italic>L. infantum</italic> and <italic>T. brucei</italic>, the causal agents in humans of trypanosomiasis (sleeping sickness) and visceral leishmaniasis (black fever), respectively (<xref ref-type="fig" rid="fig1">Figure 1D–F</xref> and <xref ref-type="table" rid="table1">Table 1</xref>). Beyond the 7TMIC-like protein fold (<xref ref-type="fig" rid="fig1">Figure 1E–F</xref> and <xref ref-type="table" rid="table1">Table 1</xref>), these proteins are characterized in their N-terminal regions by a Membrane Occupation and Recognition Nexus (MORN)-repeat domain, which is implicated in protein-protein interaction and possibly lipid binding (<xref ref-type="bibr" rid="bib71">Sajko et al., 2020</xref>). BLAST searches identified homologous proteins only within trypanosomes (<xref ref-type="fig" rid="fig1">Figure 1G</xref>), consistent with our failure to recover these sequences in earlier primary structure-based screens for 7TMICs. We did not detect any structurally related proteins to Orco in Prokaryota or Fungi (previously, fungal GRLs were only identified in chytrids [<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>], which are not currently surveyed via Dali). Together, these results reinforce our previous conclusion (<xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>) that 7TMICs evolved in or prior to the last eukaryotic common ancestor, and provide a first example of fusion of this TM protein fold with a distinct, cytoplasmic protein domain.</p></sec><sec id="s2-2"><title>PHTF proteins are candidate vertebrate 7TMICs</title><p>Given previous lack of success in identifying homologs of 7TMICs within any chordate genome, we were intrigued that our screen recovered two hits from <italic>H. sapiens</italic> (and orthologous proteins of the three other vertebrate species screened) (<xref ref-type="fig" rid="fig1">Figure 1D</xref> and <xref ref-type="supplementary-material" rid="sdata1 sdata2 sdata3">Source data 1–3</xref>). The human proteins, PHTF1 and PHTF2, are very similar to each other (54.1% amino acid identity) and have the characteristic topology of 7TMICs (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). The next most similar vertebrate proteins to Orco had substantially lower Dali Z-scores than PHTFs and represented a variety of likely spurious matches (<xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>). The single <italic>D. melanogaster</italic> ortholog (Phtf) (<xref ref-type="bibr" rid="bib47">Manuel et al., 2000</xref>) displays a similar topology to the vertebrate proteins (<xref ref-type="fig" rid="fig2">Figure 2A</xref>), and is the next most similar protein model to <italic>A. bakeri</italic> Orco after the <italic>D. melanogaster</italic> Grs, Ors, and Grls (see next section) (<xref ref-type="supplementary-material" rid="sdata2">Source data 2</xref>). PHTF is an acronym of ‘Putative Homeodomain Transcription Factor’, a name originally proposed because of presumably artifactual sequence similarity of a short region around TM4 to homeodomain DNA-binding sequences (<xref ref-type="bibr" rid="bib61">Raich et al., 1999</xref>); subsequent histological and biochemical studies (discussed below) established that PHTF1 is an integral membrane protein (<xref ref-type="bibr" rid="bib56">Oyhenart et al., 2003</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>PHTF proteins are candidate vertebrate seven transmembrane domain ion channels (7TMICs).</title><p>(<bold>A</bold>) DeepTMHMM-predicted transmembrane topology of PHTF proteins. (<bold>B</bold>) Top: AlphaFold2 predicted structure of <italic>H. sapiens</italic> PHTF1; in the image on the right the long N-terminal region (NTR) and intracellular loop 1 (IL1) are highlighted in blue; these sequences contain a few predicted helical regions but are of largely unknown structure. Bottom: visual comparison of the <italic>H. sapiens</italic> PHTF1 AlphaFold2 structure (in which the NTR and IL1 are masked) with the <italic>A. bakeri</italic> Or co-receptor (Orco) structure. (<bold>C</bold>) AlphaFold2 structures of PHTF proteins in which the NTR and IL1 are masked. Quantitative comparisons of these structures to the cryo-electronic microscopic (cryo-EM) Orco structure are provided in <xref ref-type="table" rid="table1">Table 1</xref>. (<bold>D</bold>) Major taxa/species in which a PHTF homolog was identified (see sequence databases in <xref ref-type="supplementary-material" rid="fig2sdata1">Figure 2—source data 1</xref>). Silhouette images in this and other figures are from Phylopic (<ext-link ext-link-type="uri" xlink:href="https://www.phylopic.org/">https://www.phylopic.org/</ext-link>). (<bold>E</bold>) Phylogenies of a representative set of PHTF sequences. The sequence database was constructed using the <italic>D. melanogaster</italic> and <italic>H. sapiens</italic> PHTF query sequences. Top left: maximum likelihood phylogeny (JTT + R10 model) and Bayesian phylogeny. The scale bars represent the average number of substitutions per site. Bottom left: phylogenies where weakly supported branches (&lt;95/0.95) have been rearranged and polytomies resolved in a species tree-aware manner. Right: strict consensus of the species tree-aware phylogenies. There is a single eukaryotic PHTF clade and the PHTF1-2 split occurred in the jawed vertebrate lineage. However, this interpretation relies on the rearrangement of the weakly supported jawless vertebrate PHTF branch. Therefore, an alternative but weakly supported hypothesis is that the duplication occurred in a common vertebrate ancestor and a single PHTF copy was lost in jawless vertebrates. Select branch support values are present on key branches and refer to maximum likelihood UFboot/Bayesian posterior probabilities. Asterisks indicate that branch support was below the threshold for species-aware rearrangement. The fully annotated trees are available in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplements 1</xref>–<xref ref-type="fig" rid="fig2s3">3</xref>. (<bold>F</bold>) Summary of tissue-enriched RNA expression of <italic>H. sapiens PHTF1</italic> and <italic>PHTF2</italic> (data are from the GTex Portal; the fully annotated dataset is provided in <xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>) and <italic>D. melanogaster Phtf</italic> (data from the Fly Atlas 2.0; the fully annotated dataset is provided in <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>). (<bold>G</bold>) Left: Uniform Manifold Approximation and Projection (UMAP) representation of RNA-seq datasets from individual cells of the <italic>D. melanogaster</italic> testis and seminal vesicle generated as part of the Fly Cell Atlas (10× relaxed dataset) (<xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>) colored for expression of <italic>Phtf</italic>. Simplified annotations of cell clusters displaying the highest levels of <italic>Phtf</italic> expression are adapted from <xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>; unlabeled clusters represent non-germline cell types of the testis.</p><p><supplementary-material id="fig2sdata1"><label>Figure 2—source data 1.</label><caption><title>FASTA file containing the amino acid sequences of validated eukaryotic PHTFs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata2"><label>Figure 2—source data 2.</label><caption><title>FASTA file containing the representative amino acid sequences of eukaryotic PHTFs used in phylogenetic analyses.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata3"><label>Figure 2—source data 3.</label><caption><title>FASTA file containing the multiple sequence alignment of eukaryotic PHTFs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata4"><label>Figure 2—source data 4.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of eukaryotic PHTFs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data4-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata5"><label>Figure 2—source data 5.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of eukaryotic PHTFs, based on the maximum likelihood phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data5-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata6"><label>Figure 2—source data 6.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of eukaryotic PHTFs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data6-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata7"><label>Figure 2—source data 7.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of eukaryotic PHTFs, based on the Bayesian phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data7-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig2sdata8"><label>Figure 2—source data 8.</label><caption><title>Newick tree file containing the strict consensus of the species-aware phylogenies of eukaryotic PHTFs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig2-data8-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Fully annotated phylogenetic trees for PHTF homologs.</title><p>Sequences are from the protein sequence database generated using <italic>D. melanogaster</italic> Phtf and <italic>H. sapiens</italic> PHTF1/2, and are representatives of clusters of 90% sequence identity. For maximum likelihood, the tree was generated using a JTT + R10 substitution model. Branch support values for maximum likelihood (UFboot) and Bayesian analyses (posterior probability) are shown at the branches. The scale bars represent the average number of substitutions per site.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-figsupp1-v2.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Fully annotated species-aware trees for PHTF homologs.</title><p>Trees are based on the maximum likelihood (left) and Bayesian (right) trees. Branches without support values were eligible for rearrangement.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-figsupp2-v2.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Strict consensus of the species-aware trees for PHTF homologs.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-figsupp3-v2.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Tissue-specific RNA expression of <italic>H. sapiens PHTF1</italic> and <italic>PHTF2</italic>.</title><p>Plot of RNA expression levels (transcripts per million [TPM]) from the indicated tissues is from the GTEx Portal (GTEx Analysis Release V8 [dbGaP Accession phs000424.v8.p2]).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-figsupp4-v2.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Tissue-specific RNA expression of <italic>D. melanogaster Phtf</italic> and <italic>Grls</italic>.</title><p>Heatmap plot of the expression of <italic>D. melanogaster Phtf</italic> and <italic>Grl</italic>s in the indicated tissues/life stages/sexes determined by bulk RNA-seq; fragments per kilobase of exon per million mapped fragments (FPKM) values are shown; data are from the Fly Atlas 2.0 (<xref ref-type="bibr" rid="bib38">Krause et al., 2022</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig2-figsupp5-v2.tif"/></fig></fig-group><p>To visually compare AlphaFold2 models of PHTF orthologs with <italic>A. bakeri</italic> Orco, we masked the long (&gt;300 amino acid) first intracellular loop (<xref ref-type="fig" rid="fig2">Figure 2A</xref>), whose structure is mostly unpredicted but contains a few α-helical regions, as well as the ~100-residue N-terminus (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). This visualization revealed the clear similarity in the organization of the seven TM helical core of the protein, including the split TM7 (<xref ref-type="fig" rid="fig2">Figure 2B–C</xref>), which was verified by quantitative structural comparisons (<xref ref-type="table" rid="table1">Table 1</xref>).</p><p>In contrast to other, taxon-restricted members of the 7TMIC superfamily, highly conserved PHTF homologs were found across Eukaryota, including in Bilateria, Cnidaria, and several unicellular species (<xref ref-type="fig" rid="fig2">Figure 2D</xref>). Phylogenetic analyses of a representative PHTF protein sequence dataset revealed that there is a single eukaryotic PHTF clade (<xref ref-type="fig" rid="fig2">Figure 2E</xref> and <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplements 1</xref>–<xref ref-type="fig" rid="fig2s3">3</xref>). Bayesian and maximum likelihood phylogenetics largely agree on the topology of this tree and suggest that the PHTF1-PHTF2 duplication occurred specifically in the jawed vertebrate lineage (Gnathostomata) (<xref ref-type="fig" rid="fig2">Figure 2E</xref>).</p><p>Previous tissue-specific RNA expression analysis by northern blotting of <italic>H. sapiens PHTF1</italic> and <italic>PHTF2</italic> revealed enrichment in testis and muscle, respectively (<xref ref-type="bibr" rid="bib47">Manuel et al., 2000</xref>). We confirmed and extended these conclusions by analyzing publicly available bulk RNA-sequencing (RNA-seq) datasets: <italic>PHFT1</italic> is most abundantly detected in cerebellum and testis, and <italic>PHTF2</italic> in skeletal muscle and arteries (<xref ref-type="fig" rid="fig2">Figure 2F</xref> and <xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). <italic>D. melanogaster Phtf</italic> displays highly enriched expression in the testis, and much lower expression in neural tissues in the FlyAtlas 2.0 bulk RNA-seq datasets (<xref ref-type="fig" rid="fig2">Figure 2F</xref> and <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>; <xref ref-type="bibr" rid="bib38">Krause et al., 2022</xref>), potentially indicating a closer functional relationship to <italic>PHTF1</italic> than <italic>PHTF2</italic>. Higher resolution expression analysis of <italic>Phtf</italic> in male reproductive tissue, using the Fly Cell Atlas (<xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>), revealed the most prominent expression in developing spermatocytes and spermatids (<xref ref-type="fig" rid="fig2">Figure 2G</xref>). The transcript expression of <italic>D. melanogaster Phtf</italic> is concordant with detection of rat (<italic>Rattus norvegicus</italic>) PHTF1 protein from primary spermatocytes to the end of spermatogenesis, predominantly localized to the endoplasmic reticulum (<xref ref-type="bibr" rid="bib58">Oyhenart et al., 2005b</xref>; <xref ref-type="bibr" rid="bib56">Oyhenart et al., 2003</xref>). The N-terminal region of mouse (<italic>M. musculus</italic>) PHTF1 associates with the testis-enriched FEM1B E3 ubiquitin ligase and is suggested to recruit it to the endoplasmic reticulum (<xref ref-type="bibr" rid="bib57">Oyhenart et al., 2005a</xref>). Overexpression and/or knock-down studies of PHTF1 and PHTF2 in cell lines hint at roles in regulating cell proliferation and survival, and possible links to various cancers (<xref ref-type="bibr" rid="bib13">Chi et al., 2020</xref>; <xref ref-type="bibr" rid="bib29">Huang et al., 2015</xref>). However, the biological function of any PHTF protein in any organism is unclear. Nevertheless, PHTFs represent the first candidate homologs of insect Ors/Grs in chordates, indicating that they might not have been completely lost from this lineage, as previously thought (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib65">Robertson, 2015</xref>); we suggest they also act as ion channels.</p></sec><sec id="s2-3"><title>Novel sets of candidate insect chemoreceptors</title><p>Within the hits of our screen of <italic>D. melanogaster</italic> protein structures, we noticed 10 proteins that do not belong to the canonical Gr or Or families (<xref ref-type="supplementary-material" rid="sdata1 sdata2 sdata3">Source data 1–3</xref>). These proteins have a similar length and TM topology as Grs and Ors (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Visual inspection and quantitative analyses confirmed that their predicted fold is very similar to that of <italic>A. bakeri</italic> Orco (<xref ref-type="fig" rid="fig3">Figure 3A</xref> and <xref ref-type="table" rid="table1">Table 1</xref>). As they almost completely lack other defining sequence features of these families (see below), we named these Grl proteins, using the same cytogenetic-based gene nomenclature conventions of other chemosensory gene families (e.g., <xref ref-type="bibr" rid="bib17">Drosophila Odorant Receptor Nomenclature Committee, 2000</xref>), with one exception (GrlHolozoa [GrlHz], see below).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Insect Grls are highly divergent, candidate chemosensory receptors.</title><p>(<bold>A</bold>) Proposed nomenclature of <italic>D. melanogaster</italic> Grls (the original gene name and cytological location are in parentheses), with corresponding DeepTMHMM-predicted transmembrane topologies and AlphaFold2 structural models. Note that TM7 is not predicted for Grl36b and Grl58a by DeepTMHMM, but is predicted – with the characteristic TM7a/7b split – in the structural model (as well as predicted by Phobius [data not shown]). Quantitative comparisons of these structures to the cryo-electronic microscopic (cryo-EM) Or co-receptor (Orco) structure are provided in <xref ref-type="table" rid="table1">Table 1</xref>. (<bold>B</bold>) Sequence similarity network of Grls, Grs, and Ors (including Orco). The network was generated using an all-to-all comparison made by MMSeqs2 as implemented by gs2. The connections represent E-values where the weakest connections (arbitrarily defined as edge weights &gt;1) are colored in lighter gray. Lack of connection between two nodes indicates that those two sequences could not be identified as having any significant sequence similarity under the most sensitive MMSeqs2 settings. Nodes and edges are arranged in a prefuse force-directed layout. The graph splitting tree is visualized in <xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>; however, we do not place high confidence in the phylogenetic accuracy of the tree due to the likely effects of long branch attraction. The evolution of GrlHolozoa (GrlHz) is described in <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>, with detailed phylogenies in <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplements 2</xref>–<xref ref-type="fig" rid="fig3s4">4</xref>. (<bold>C</bold>) Schematic of the gene arrangement of <italic>Grl36a</italic> and <italic>Gr36</italic> homologs in drosophilids. Color coding reflects relatedness with respect to major speciation and gene duplication events; colors match the phylogenetic tree branches in <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplement 6B–C</xref>. The <italic>Drosophila</italic> subgenus entirely lacks Gr36 homologs (see <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplement 6</xref>). (<bold>D</bold>) Alignment of the C-terminal region of <italic>D. melanogaster</italic> Orco, Gr64a, select insect Gr36/Gr59 homologs, and <italic>D. melanogaster</italic> Grl36a and Grl43a, extracted from a larger alignment available in <xref ref-type="supplementary-material" rid="fig3sdata5">Figure 3—source data 5</xref>. The black bar shows the common location of a phase 0 intron, which is presumably homologous in different sequences. The canonical TM7 motif of the Gr family (represented as relative amino acid frequencies extracted from WebLogo) is shown above the sequence, and the variant motifs of different Gr or Grl ortholog groups are shown below. (<bold>E</bold>) Phylogenies of Gr36, Gr59c/d, Grl36a, Grl43a and homologous non-drosophilid sequences (color-coded as in (<bold>D</bold>)). The sequence database was assembled using <italic>D. melanogaster</italic> Gr36a, Grl36a, and Grl43a as the query sequences. Top left: maximum likelihood phylogeny (JTT + F + R7 model) and Bayesian phylogeny. The scale bars represent average number of substitutions per site. Bottom left: phylogenies where weakly supported branches (&lt;95/0.95) have been rearranged and polytomies resolved in a species tree-aware manner. Right: strict consensus of the species tree-aware phylogenies. These analyses support that Gr36 and Grl36a/43a are sister clades, which likely split after Gr59c/d diverged from the ancestral lineage. Sequences are colored as in (<bold>D</bold>). Select branch support values are present on key branches and refer to maximum likelihood UFboot and Bayesian posterior probabilities, in this order. Asterisks indicate that branch support was below the threshold for species-aware rearrangement. A simplified schematic of gene duplication and loss is illustrated in <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplement 6F</xref>. The fully annotated trees are available in <xref ref-type="fig" rid="fig3s7">Figure 3—figure supplements 7</xref>–<xref ref-type="fig" rid="fig3s9">9</xref>. (<bold>F</bold>) Histogram of <italic>Gr</italic> and <italic>Grl</italic> expression levels in adult proboscis and maxillary palps determined by bulk RNA-sequencing (RNA-seq). Mean values ± SD of fragments per kilobase of transcript per million mapped reads (FPKM) are plotted; n=3 biological replicates. Data is from <xref ref-type="bibr" rid="bib19">Dweck et al., 2021</xref>. (<bold>G</bold>) Left: t-distributed stochastic neighbor embedding (tSNE) representation of RNA-seq datasets from individual cells of the <italic>D. melanogaster</italic> proboscis and maxillary palp – generated as part of the Fly Cell Atlas (10× stringent dataset) (<xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>) – colored for expression of the indicated genes. <italic>Gr64f</italic> and <italic>Gr66a</italic> are broad markers of ‘sweet/appetitive’ and ‘bitter/aversive’ gustatory sensory neurons, respectively. Transcripts for three <italic>Grl</italic>s are detected in subsets of bitter/aversive neurons. Annotations of cell clusters are adapted from <xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>; unlabeled clusters represent other non-gustatory sensory neuron or non-neuronal cell types of this tissue.</p><p><supplementary-material id="fig3sdata1"><label>Figure 3—source data 1.</label><caption><title>FASTA file containing the amino acid sequences used in the network and graph splitting analysis of gustatory receptors (Grs), odorant receptors (Ors), and Grls.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata2"><label>Figure 3—source data 2.</label><caption><title>Tab delimited text file containing the sequence similarity network of gustatory receptors (Grs), odorant receptors (Ors), and Grls.</title><p>The first column is the source node, the second column is the target node, and the third column is the E-value derived from MMSeqs2 and gs2.</p></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata3"><label>Figure 3—source data 3.</label><caption><title>Tab delimited text file containing the annotation for the sequence similarity network of gustatory receptors (Grs), odorant receptors (Ors), and Grls.</title><p>The first column is the node identifier (ID) and the second column is the sequence name (SEQ).</p></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata4"><label>Figure 3—source data 4.</label><caption><title>Newick tree file containing the graph splitting tree of odorant receptors (Ors), gustatory receptors (Grs), and Grls, derived from the sequence similarity network by gs2.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data4-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata5"><label>Figure 3—source data 5.</label><caption><title>FASTA file containing the multiple sequence alignment used for illustrating intron and transmembrane domain 7 (TM7) motif conservation between gustatory receptors (Grs) and Grls.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data5-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata6"><label>Figure 3—source data 6.</label><caption><title>FASTA file containing the amino acid sequences of Gr36, Gr59, Grl36a, and Grl43a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data6-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata7"><label>Figure 3—source data 7.</label><caption><title>FASTA file containing the multiple sequence alignment of Gr36, Gr59, Grl36a, and Grl43a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data7-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata8"><label>Figure 3—source data 8.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of Gr36, Gr59, Grl36a, and Grl43a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data8-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata9"><label>Figure 3—source data 9.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Gr36, Gr59, Grl36a, and Grl43a homologs, based on the maximum likelihood phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data9-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata10"><label>Figure 3—source data 10.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of Gr36, Gr59, Grl36a, and Grl43a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data10-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata11"><label>Figure 3—source data 11.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Gr36, Gr59, Grl36a, and Grl43a homologs, based on the Bayesian phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data11-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3sdata12"><label>Figure 3—source data 12.</label><caption><title>Newick tree file containing the strict consensus of the species-aware phylogenies of Gr36, Gr59, Grl36a, and Grl43a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-data12-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Evolution of GrlHolozoa (GrlHz), a family of Grl seven transmembrane domain ion channel (7TMIC) not restricted to flies.</title><p>(<bold>A</bold>) Major taxa/species for which a GrlHz homolog was recovered. (<bold>B</bold>) Phylogenies of a representative set of GrlHz sequences (clustered by 70% sequence identity). The sequence database was assembled using <italic>D. melanogaster</italic> GrlHz as the query sequence. Top: maximum likelihood phylogeny and Bayesian phylogeny. The scale bars represent the average number of substitutions per site. Bottom: phylogenies where weakly supported branches (&lt;95/0.95) have been rearranged and polytomies resolved in a species tree-aware manner. Right: strict consensus of the species tree-aware phylogenies. The fully annotated trees are visualized in <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplements 2</xref>–<xref ref-type="fig" rid="fig3s4">4</xref>. (<bold>C</bold>) Left: the single holozoan copy hypothesis of GrlHz evolution. Under this scenario, a single GrlHz is widely conserved across Holozoa, but has been independently duplicated/lost several times in various taxa. Right: the two-paralog hypothesis of GrlHz evolution. As both the maximum likelihood and Bayesian phylogenies provide evidence for two GrlHz clades, and because some species have two substantially divergent GrlHz sequences, it is possible that there was a gene duplication event early in the evolution of Holozoa. (<bold>D</bold>) Examples of GrlHz structures. Of 196 representative sequences, 31 sequences (mostly from Hymenoptera and Lepidoptera) bear N-terminal WD40 repeats.</p><p><supplementary-material id="fig3s1sdata1"><label>Figure 3—figure supplement 1—source data 1.</label><caption><title>FASTA file containing the amino acid sequences of validated holozoan GrlHolozoa (GrlHz).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata2"><label>Figure 3—figure supplement 1—source data 2.</label><caption><title>FASTA file containing the representative amino acid sequences of holozoan GrlHolozoa (GrlHz) used in phylogenetic analyses.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata3"><label>Figure 3—figure supplement 1—source data 3.</label><caption><title>FASTA file containing the multiple sequence alignment of holozoan GrlHolozoa (GrlHz).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata4"><label>Figure 3—figure supplement 1—source data 4.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of holozoan GrlHolozoa (GrlHz).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data4-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata5"><label>Figure 3—figure supplement 1—source data 5.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of holozoan GrlHolozoa (GrlHz), based on the maximum likelihood phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data5-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata6"><label>Figure 3—figure supplement 1—source data 6.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of holozoan GrlHolozoa (GrlHz).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data6-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata7"><label>Figure 3—figure supplement 1—source data 7.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of holozoan GrlHolozoa (GrlHz), based on the Bayesian phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data7-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s1sdata8"><label>Figure 3—figure supplement 1—source data 8.</label><caption><title>Newick tree file containing the strict consensus of the species-aware phylogenies of holozoan GrlHolozoa (GrlHz).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp1-data8-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Fully annotated phylogenetic trees for GrlHolozoa (GrlHz) homologs.</title><p>For maximum likelihood, the tree was generated using a JTT + F + R9 substitution model. Branch support values for maximum likelihood (UFboot) and Bayesian analyses (posterior probability) are shown at the branches. The scale bars represent the average number of substitutions per site.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp2-v2.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Fully annotated species-aware trees for GrlHolozoa (GrlHz) homologs.</title><p>Trees are based on the maximum likelihood (left) and Bayesian (right) trees. Branches without support values were eligible for rearrangement.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp3-v2.tif"/></fig><fig id="fig3s4" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 4.</label><caption><title>Strict consensus of the species-aware trees for GrlHolozoa (GrlHz) homologs.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp4-v2.tif"/></fig><fig id="fig3s5" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 5.</label><caption><title>Fully annotated graph splitting tree for odorant receptors (Ors), gustatory receptors (Grs), and Grls.</title><p>Key edge perturbation support values are visible on branches. The primary sequence databases were assembled using each of the <italic>D. melanogaster</italic> Grls as query sequences. <italic>D. melanogaster</italic> Or and Gr sequences were manually collected from FlyBase. Sequences from <italic>M. hrabei</italic> (jumping bristletail), <italic>Thermobia domestica</italic> (firebrat), <italic>Ladona filva</italic> (dragonfly), and <italic>Ephemera danica</italic> (green drake mayfly) were added, following the proposal that canonical Ors may have diversified after the emergence of Neoptera (most winged insects) (<xref ref-type="bibr" rid="bib8">Brand et al., 2018</xref>); 2498 additional sequences were collected using the <italic>N. vectensis</italic> GRL1 query sequence (XP_048580785.1); the PSI-BLAST searches were stopped at four iterations, as the search had substantially recovered insect Gr sequences, and further searches returned tens of thousands of sequences. The basal placement of the Grls is unusual given their conservation in flies, as this would suggest they diversified in a common animal ancestor and that the Grls were lost in all animal taxa except flies. This hypothesis seems unlikely given the extreme number of independent gene loss events this would require, and we therefore suspect that this tree topology represents a phylogenetic error, for example, long branch attraction (<xref ref-type="bibr" rid="bib7">Bergsten, 2005</xref>). The inset shows major collapsed clades, where the tip node is sized proportionally to the number of sequences collapsed.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp5-v2.tif"/></fig><fig id="fig3s6" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 6.</label><caption><title>The evolution of Gr36, Gr59, Grl36a, and Grl43a.</title><p>(<bold>A</bold>) Schematic of the gene arrangement of <italic>Grl36a</italic> and <italic>Gr36</italic> homologs in drosophilids, with colors matching trees in (<bold>B</bold>) and (<bold>C</bold>). This panel is reproduced from <xref ref-type="fig" rid="fig3">Figure 3C</xref>. (<bold>B</bold>) Species-aware Bayesian phylogeny of Grl36. (<bold>C</bold>) Species tree-aware Bayesian phylogeny of Gr36. (<bold>D</bold>) Phylogenies of Gr36, Gr59, Grl36a, Grl43a, and other homologous sequences. The sequence database was assembled using <italic>D. melanogaster</italic> Gr36a, Grl36a, and Grl43a as query sequences. Top: maximum likelihood phylogeny and Bayesian phylogeny. The scale bars represent the average number of substitutions per site. Bottom: phylogenies where weakly supported branches (&lt;95/0.95) have been rearranged and polytomies resolved in a species tree-aware manner. (<bold>E</bold>) Strict consensus of the species tree-aware phylogenies. These analyses support that Gr36 and Grl36a/43a are sister clades, which likely split after the Gr59 split. (<bold>F</bold>) Proposed model of Gr36, Gr59, Grl36a, and Grl43a evolution.</p><p><supplementary-material id="fig3s6sdata1"><label>Figure 3—figure supplement 6—source data 1.</label><caption><title>FASTA file containing the amino acid sequences of Gr36 homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata2"><label>Figure 3—figure supplement 6—source data 2.</label><caption><title>FASTA file containing the multiple sequence alignment of Gr36 homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata3"><label>Figure 3—figure supplement 6—source data 3.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of Gr36 homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata4"><label>Figure 3—figure supplement 6—source data 4.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Gr36 homologs, based on the maximum likelihood phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data4-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata5"><label>Figure 3—figure supplement 6—source data 5.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of Gr36 homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data5-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata6"><label>Figure 3—figure supplement 6—source data 6.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Gr36 homologs, based on the Bayesian phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data6-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata7"><label>Figure 3—figure supplement 6—source data 7.</label><caption><title>FASTA file containing the amino acid sequences of Grl36a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data7-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata8"><label>Figure 3—figure supplement 6—source data 8.</label><caption><title>FASTA file containing the multiple sequence alignment of Grl36a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data8-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata9"><label>Figure 3—figure supplement 6—source data 9.</label><caption><title>Newick tree file containing the maximum likelihood phylogeny of Grl36a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data9-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata10"><label>Figure 3—figure supplement 6—source data 10.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Grl36a homologs, based on the maximum likelihood phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data10-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata11"><label>Figure 3—figure supplement 6—source data 11.</label><caption><title>NEXUS tree file containing the Bayesian phylogeny of Grl36a homologs.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data11-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig3s6sdata12"><label>Figure 3—figure supplement 6—source data 12.</label><caption><title>NOTUNG tree file containing the species-aware phylogeny of Grl36a homologs, based on the Bayesian phylogeny.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig3-figsupp6-data12-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp6-v2.tif"/></fig><fig id="fig3s7" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 7.</label><caption><title>Fully annotated phylogenetic trees for Gr36, Gr59, Grl36a, and Grl43a homologs.</title><p>For maximum likelihood, the tree was generated using a JTT + F + R7 substitution model and is rooted. Branch support values for maximum likelihood (UFboot) and Bayesian analyses (posterior probability) are shown at the branches. Non-drosophilid sequences are assumed to be the outgroup. The scale bars represent the average number of substitutions per site.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp7-v2.tif"/></fig><fig id="fig3s8" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 8.</label><caption><title>Fully annotated species-aware trees for Gr36, Gr59, Grl36a, and Grl43a homologs.</title><p>Trees are based on the maximum likelihood (left) and Bayesian (right) trees. Branches without support values were eligible for rearrangement.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp8-v2.tif"/></fig><fig id="fig3s9" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 9.</label><caption><title>Strict consensus of the species-aware trees for Gr36, Gr59, Grl36a, and Grl43a homologs.</title><p>Although the consensus tree has a polytomy near the emergence of Gr59, this is strictly due to disagreement as to whether the lone <italic>Scaptodrosophila</italic> sequence is a Gr59 homolog or an outgroup to all other <italic>Drosophila</italic>/<italic>Sophophora</italic> sequences shown here.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig3-figsupp9-v2.tif"/></fig></fig-group><p>For seven <italic>D. melanogaster</italic> Grls, BLAST searches identified homologs only in drosophilids; for two others (Grl40a and Grl65a) we recovered drosophilid and other fly homologs. By contrast, the Grl originally designated CG3831 has homologs across a wide range of Holozoa (i.e., animals and their closest single-celled, non-fungal relatives), including chordates (e.g., the lancelet <italic>Branchiostoma floridae</italic>) and single-cell eukaryotes (e.g., <italic>Capsaspora owczarzaki</italic>) (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplements 1</xref>–<xref ref-type="fig" rid="fig3s4">4</xref>), leading us to name it GrlHolozoa (GrlHz). A subset of GrlHz homologs bear a long N-terminal domain containing WD40 repeats, which form a structurally predicted beta-propeller domain that is typically involved in protein-protein interactions (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1D</xref>; <xref ref-type="bibr" rid="bib37">Kim and Kim, 2020</xref>).</p><p>Given that nine of these Grls are restricted to flies, a reasonable hypothesis is that they evolved from fly Grs. To infer their evolutionary origins, we therefore examined sequence similarity of Grls with a representative set of Grs, as well as Ors and other animal (i.e., non-insect) GRLs. We found that fly Grls share little or no obvious sequence similarity with any of these other 7TMICs, precluding confident standard phylogenetic analysis and leading us to use an all-to-all graph-based methodology, which does not require a multiple sequence alignment. This approach infers relationships between sequences based on pairwise sequence similarity. This analysis first generates an all-to-all sequence similarity network via MMseqs2, in which sequence families can be identified as clusters in a 2D projection (<xref ref-type="fig" rid="fig3">Figure 3B</xref>), and then a tree by recursive spectral clustering (<xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>; see Methods; <xref ref-type="bibr" rid="bib48">Matsui and Iwasaki, 2020</xref>; <xref ref-type="bibr" rid="bib75">Steinegger and Söding, 2017</xref>). In the network, we observed that several of the Grls were intermingled in clusters (e.g., Grl62b/Grl62c and Grl36a/Grl43a), suggesting relatively recent common ancestry (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). These two clusters were recapitulated as clades in the graph splitting phylogeny (<xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>). For Grl62a/b/c, the possibility of recent ancestry is consistent with the tandem genomic organization of the corresponding genes, which implies their evolution by gene duplication through non-allelic homologous recombination, similar to other families of chemosensory genes (<xref ref-type="bibr" rid="bib54">Nei et al., 2008</xref>). None of the Grl clusters grouped with those of Ors, Grs, or other animal GRLs, rather connecting broadly, but weakly, with all other clusters (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Consistent with this clustering pattern, all Grls were placed near the presumed root of the graph splitting tree (<xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>). This basal placement of Grls was inconsistent with their conservation only in flies, and is likely a phylogenetic artifact (see legend to <xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>).</p><p>Although analysis of amino acid sequences did not provide evidence of ancestry between Grs and Grls, we noted that <italic>Grl36a</italic> was adjacent (separated by 306 bp) to the <italic>Gr36a/b/c</italic> cluster in the <italic>D. melanogaster</italic> genome. This proximity suggested that <italic>Grl36a</italic> might have arisen by gene duplication of a <italic>Gr36</italic>-like ancestor. Indeed, <italic>Grl36a</italic> homologs across drosophilid species were always found in tandem with <italic>Gr36</italic>-related genes in various arrangements (<xref ref-type="fig" rid="fig3">Figure 3C</xref>, <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplement 6</xref>). To further investigate the hypothetical ancestry of <italic>Grl36a</italic> and <italic>Gr36</italic>, we first examined their gene structure. We incorporated into this analysis <italic>Gr59c</italic> and <italic>Gr59d</italic> homologs, which are closely related to <italic>Gr36a/b/c</italic> even though they are distantly located in the genome (<xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>), as well as <italic>Grl43a</italic>, the most closely related paralog to <italic>Grl36a</italic> (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). The <italic>Gr</italic> family is characterized by the general, but not universal, conservation of three phase 0 introns near the 3’ end of these genes (<xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>). <italic>Gr36</italic>, <italic>Gr59c/d,</italic> and homologous non-drosophilid genes possess only one of these introns, which corresponds to the second ancestral <italic>Gr</italic> intron located just before the exon encoding TM7. Both <italic>D. melanogaster Grl36a</italic> and <italic>Grl43a</italic> also have a phase 0 intron immediately before the TM7-encoding exon, which aligns with the <italic>Gr</italic> intron position on a multiple protein sequence alignment (<xref ref-type="fig" rid="fig3">Figure 3D</xref>), suggesting that these <italic>Gr</italic> and <italic>Grl</italic> introns are homologous. We next examined the TM7 motifs in these Grs and Grls. The canonical TM7 motif of Grs is TYhhhhhQF, where h is a hydrophobic residue (<xref ref-type="fig" rid="fig3">Figure 3D</xref>; <xref ref-type="bibr" rid="bib66">Robertson, 2019</xref>; <xref ref-type="bibr" rid="bib73">Scott et al., 2001</xref>). However, Gr36 and Gr59c/d share a variant motif, T(H/N)(S/A)hhhhQ(Y/F/W), and we observed a very similar motif in Grl36a and Grl43a (<xref ref-type="fig" rid="fig3">Figure 3D</xref>).</p><p>The genomic proximity of <italic>Gr36</italic> and <italic>Grl36a</italic>, and similarity in introns and TM7 motifs of these genes (as well as <italic>Gr59c/d</italic> and <italic>Grl43a</italic>) provide evidence that these genes have a relatively recent common ancestry within drosophilids. Phylogenetic analyses of this proposed clade support that a Grl36a/Grl43a clade is the sister clade to Gr36, and that this split occurred after the emergence of the Gr59c/d clade (<xref ref-type="fig" rid="fig3">Figure 3E</xref>, <xref ref-type="fig" rid="fig3s6">Figure 3—figure supplements 6</xref>–<xref ref-type="fig" rid="fig3s9">9</xref>). None of the other <italic>Grl</italic> genes are located adjacent to <italic>Gr</italic> genes, nor do the proteins possess a recognizable TM7 motif. Some other Grls might possess conserved introns of <italic>Gr</italic>s (e.g., <italic>Grl40a</italic> with the first ancestral intron, and <italic>Grl36b</italic> and <italic>Grl65a</italic> with the second ancestral intron [data not shown]), but we cannot conclude with confidence that these are homologous. Thus, the ancestry of most Grls remains unresolved. Nevertheless, the highly restricted taxonomic representation of nine of these Grls and their structural similarity to Grs support a model in which Grls have rapidly evolved and diverged from ancestral Grs.</p><p>To gain insight into the potential role(s) of Grls, we first examined their expression in tissue-specific bulk RNA-seq datasets from the FlyAtlas 2.0 (<xref ref-type="bibr" rid="bib38">Krause et al., 2022</xref>). Most <italic>Grl</italic>s were expressed at very low (&lt;1 fragment per kilobase of exon per million mapped fragments [FPKM]) or undetectable levels in essentially all tissues in these datasets, although <italic>Grl36</italic>b was detected in neuronal tissues (eye, brain, thoracicoabdominal ganglion) (<xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>). The one exception was <italic>GrlHz</italic>, which was expressed (&gt;8 FPKM) in various tissues (e.g., heart, ovary, testis, and larval fat body and garland cells [nephrocytes]). The unique expression and conservation properties of <italic>GrlHz</italic> suggest it has a different function from other <italic>Grl</italic>s.</p><p>The lack of detection of transcripts for most <italic>Grl</italic>s in the FlyAtlas 2.0 suggested that these genes might have highly restricted cellular expression patterns. Given the structural similarity of Grls to Grs, we examined their expression in an RNA-seq dataset of the major taste organ (labellum; a tissue not specifically represented in the FlyAtlas 2.0) (<xref ref-type="bibr" rid="bib19">Dweck et al., 2021</xref>). <italic>D. melanogaster Gr</italic> genes display a wide range of expression levels in the labellar transcriptome, in part reflecting the breadth of expression in different classes of taste neurons. For example, <italic>Gr66a</italic> and <italic>Gr64f</italic> – broadly expressed markers for ‘bitter/aversive’ and ‘sweet/appetitive’ neuronal populations, respectively (<xref ref-type="bibr" rid="bib22">Freeman and Dahanukar, 2015</xref>) – are detected at comparatively high levels (&gt;5 FPKM) (<xref ref-type="fig" rid="fig3">Figure 3F</xref>). By contrast, many receptors expressed in subsets of these major neuron types (e.g., <italic>Gr22e</italic> for bitter and <italic>Gr61a</italic> for sweet; <xref ref-type="bibr" rid="bib22">Freeman and Dahanukar, 2015</xref>) are expressed at much lower levels (~1 FPKM). Similar to this latter type of <italic>Gr</italic>, transcripts for four <italic>Grl</italic>s were detected at &gt;0.5 FPKM: <italic>GrlHz</italic>, <italic>Grl62c</italic>, <italic>Grl62a,</italic> and <italic>Grl36a</italic> (<xref ref-type="fig" rid="fig3">Figure 3F</xref>). Importantly, within the Fly Cell Atlas dataset of the proboscis and maxillary palp (<xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>), three of these were specifically expressed in the cluster of cells corresponding to <italic>Gr66a</italic>-expressing bitter/aversive neurons (<xref ref-type="fig" rid="fig3">Figure 3G</xref>). The fourth, <italic>GrlHz</italic>, was very sparsely expressed in non-neuronal cell types in this tissue, including hemocytes (data not shown; <xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>). None of the other six <italic>Grl</italic>s were detectable in this dataset, consistent with their lower expression in the labellar bulk RNA-seq transcriptome (<xref ref-type="fig" rid="fig3">Figure 3F</xref>). Moreover, no <italic>Grl</italic> (except the broadly expressed <italic>GrlHz</italic>) was detectably expressed in other chemosensory tissue transcriptomes (leg, wing, or antenna) (data not shown; <xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>; <xref ref-type="bibr" rid="bib49">Menuz et al., 2014</xref>). These observations raise the possibility that at least three Grls (Grl36a, Grl62a, and Grl62c) are chemosensory receptors for aversive stimuli.</p></sec><sec id="s2-4"><title>A hypothesis for the evolution of the 7TMIC superfamily</title><p>Two hypotheses could explain the similarities between well-established 7TMICs and the candidate homologs described in this work: homology (i.e., shared ancestry), and thus the existence of a unified 7TMIC superfamily, or convergent evolution of the 7TMIC structure. We discuss the latter possibility in the following section. Here, we consider a detailed hypothesis of a 7TMIC superfamily of single evolutionary origin. Because confident multiprotein alignment of all members was impossible, we used the same all-to-all graph-based approach as for insect Grls to generate a sequence similarity network, and families were identified as clusters in a 2D projection (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). We used the gross connectivity of clusters, and the presence or (putative) absence of these proteins across taxa (<xref ref-type="fig" rid="fig4">Figure 4B</xref>), to make inferences about the ancestry of these proteins.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>A hypothesis for the evolution of the seven transmembrane domain ion channel (7TMIC) superfamily.</title><p>(<bold>A</bold>) Sequence similarity network of the 7TMIC superfamily, generated using the same odorant receptors (Ors) and gustatory receptors (Grs) from <xref ref-type="fig" rid="fig3">Figure 3B</xref>, unicellular eukaryotic Grls from <xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>, and sequence databases assembled using the following query sequences: <italic>N. vectensis</italic> GRL1, <italic>D. melanogaster</italic> Grls and Phtf, <italic>H. sapiens</italic> PHTF1 and PHTF2, <italic>Arabidopsis thaliana</italic> Domain of Unknown Function (DUF) 3537, <italic>C. elegans</italic> SRRs and trypanosome GRLs. The network was generated and visualized as in <xref ref-type="fig" rid="fig3">Figure 3B</xref>. The graph splitting tree is visualized in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>. (<bold>B</bold>) Presence and absence of 7TMICs across taxa: ‘other animal GRL’ refers to GRLs in non-insect animal species previously identified by primary sequence similarity (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib65">Robertson, 2015</xref>; <xref ref-type="bibr" rid="bib70">Saina et al., 2015</xref>) and nematode SRRs. The dashed branch represents several collapsed paraphyletic clades. (<bold>C</bold>) Model of 7TMIC superfamily evolution. The dashed branches represent several collapsed paraphyletic clades and speciation events. The trypanosome 7TMICs are unplaced due to the currently unresolved taxonomy of trypanosomes (<xref ref-type="bibr" rid="bib9">Burki et al., 2020</xref>).</p><p><supplementary-material id="fig4sdata1"><label>Figure 4—source data 1.</label><caption><title>FASTA file containing the amino acid sequences used in the network and graph splitting analysis of eukaryotic seven transmembrane domain ion channels (7TMICs).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig4-data1-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig4sdata2"><label>Figure 4—source data 2.</label><caption><title>Tab delimited text file containing the sequence similarity network of eukaryotic seven transmembrane domain ion channels (7TMICs).</title><p>The first column is the source node, the second column is the target node, and the third column is the E-value derived from MMSeqs2 and gs2.</p></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig4-data2-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig4sdata3"><label>Figure 4—source data 3.</label><caption><title>Tab delimited text file containing the annotation for the sequence similarity network of eukaryotic seven transmembrane domain ion channels (7TMICs).</title><p>The first column is the node identifier (ID) and the second column is the sequence name (SEQ).</p></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig4-data3-v2.zip"/></supplementary-material></p><p><supplementary-material id="fig4sdata4"><label>Figure 4—source data 4.</label><caption><title>Newick tree file containing the graph splitting tree of eukaryotic seven transmembrane domain ion channels (7TMICs), derived from the sequence similarity network by gs2.</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-85537-fig4-data4-v2.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Graph splitting tree for the proposed seven transmembrane domain ion channel (7TMIC) superfamily.</title><p>Key edge perturbation support values are visible on branches. The inset shows major collapsed clades, where the triangular tip is sized proportionally to the number of sequences collapsed. This tree suggests a different branching pattern than the hypothesis in <xref ref-type="fig" rid="fig4">Figure 4C</xref>, consistent with a more complex duplication/loss history for the 7TMIC superfamily. However, as in <xref ref-type="fig" rid="fig3s5">Figure 3—figure supplement 5</xref>, we suspect long branch attraction is present in this analysis, at least for the fly Grls and nematode proteins.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-85537-fig4-figsupp1-v2.tif"/></fig></fig-group><p>In the sequence similarity network, clusters of Ors, Grs, and non-insect animal GRLs (excluding GrlHz) were closely located or intermingled, while insect Grl clusters were more distantly located from this grouping (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). GrlHz formed a distinct cluster, but this connects only with the Or/Gr/Grl clusters (and not plant DUF3537 or PHTF clusters) (<xref ref-type="fig" rid="fig4">Figure 4A</xref>), suggesting that it descended from a Gr-like ancestor. Given that GrlHz was not detected outside of Holozoa (<xref ref-type="fig" rid="fig4">Figure 4B</xref>), the simplest hypothesis is that an ancestral holozoan had a 7TMIC gene that duplicated to produce an ancestral GrlHz and an ancestral Gr (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). The diversity of Ors, Grs, and Grls would then have resulted from taxon-specific diversification of a single, holozoan branch of a hypothetical 7TMIC superfamily (<xref ref-type="fig" rid="fig4">Figure 4C</xref>).</p><p>The plant DUF3537 protein cluster was relatively well connected to the Or/Gr/Grl clusters (<xref ref-type="fig" rid="fig4">Figure 4A</xref>), consistent with the previously recognized sequence similarity between DUF3537 and Grs, which supported their proposed shared ancestry (<xref ref-type="bibr" rid="bib5">Benton, 2015</xref>; <xref ref-type="bibr" rid="bib6">Benton et al., 2020</xref>). If this is correct, a DUF3537/Or/Gr/Grl ancestor must have been present in a common ancestor of plants (part of Diaphoretickes) and animals (part of Amorphea) (<xref ref-type="fig" rid="fig4">Figure 4B–C</xref>). Unicellular eukaryotic 7TMICs were dispersed between Or/Gr/Grl and DUF3537 proteins (<xref ref-type="fig" rid="fig4">Figure 4A</xref>); the simplest hypothesis is that these are related to other 7TMICs in accordance with their species’ taxonomy (e.g., SAR [stremenopiles, alveolates, and Rhizaria] 7TMICs are more closely related to plant DUF3537 proteins than to animal Grs). Alternatively, the generally sparse conservation of unicellular eukaryotic 7TMICs might indicate horizontal gene transfer(s).</p><p>Finally, PHTF also forms a separate cluster (<xref ref-type="fig" rid="fig4">Figure 4A</xref>), and its broad taxonomic representation argues that the <italic>PHTF</italic> ancestral gene must also have been present in a common Amorphea-Diaphoretickes ancestor (<xref ref-type="fig" rid="fig4">Figure 4B–C</xref>). If there was a single ancestral 7TMIC, we hypothesize that this gene must have duplicated in a common eukaryotic ancestor to produce the distinct PHTF and Or/Gr/Grl/DUF3537 lineages (<xref ref-type="fig" rid="fig4">Figure 4C</xref>).</p></sec><sec id="s2-5"><title>Concluding remarks</title><p>Exploiting recent advances in protein structure predictions, we have used a tertiary structure-based screening approach to identify new candidate members of the 7TMIC superfamily. While the founder members of this superfamily, insect Ors and Grs, were thought for many years to define an invertebrate-specific protein family (<xref ref-type="bibr" rid="bib3">Benton, 2006</xref>; <xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>), there is now substantial evidence that these proteins originated in a eukaryotic common ancestor. We also counter previous assumptions that 7TMICs were completely lost in Chordata, through discovery of two lineages within this superfamily: PHTF and GrlHz. Finally, we have identified many previously overlooked putative chemosensory receptors in <italic>D. melanogaster</italic> (and related flies).</p><p>Two important issues remain open. First, are all of the candidate 7TMICs homologous, or does shared tertiary structure reflect convergent evolution in protein folding in at least some cases? Doubts about homology stem, reasonably, from the extreme sequence divergence between 7TMICs to beyond the twilight zone of sequence similarity (<xref ref-type="bibr" rid="bib68">Rost, 1999</xref>). However, sequence divergence over many millions of years of accumulated amino acid substitutions is well appreciated in this superfamily (e.g., pairs of <italic>D. melanogaster</italic> Grs can display as little as 8% amino acid identity; <xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>). Thus, sequence dissimilarity alone is not compelling evidence for structural convergence. Examples of convergent evolution of tertiary protein structures have been described (<xref ref-type="bibr" rid="bib1">Alva et al., 2010</xref>; <xref ref-type="bibr" rid="bib2">Alva et al., 2015</xref>; <xref ref-type="bibr" rid="bib76">Tomii et al., 2012</xref>) but the vast majority of these are small protein domains or motifs, some of which might represent relics of the evolution of proteins from short peptide ancestors (<xref ref-type="bibr" rid="bib2">Alva et al., 2015</xref>; <xref ref-type="bibr" rid="bib45">Lupas et al., 2001</xref>). The core of the 7TMIC fold is &gt;300 amino acids, and the question of homology or convergence is most akin to the unresolved, long-standing debate regarding the evolution of the 7TM G protein-coupled receptor fold of type I and type II rhodopsins (<xref ref-type="bibr" rid="bib46">Mackin et al., 2014</xref>; <xref ref-type="bibr" rid="bib69">Rozenberg et al., 2021</xref>). In the case of 7TMICs, if PHTFs and other family members are not homologous, their taxonomic representation indicates that structural convergence must have occurred in a eukaryotic common ancestor. While it might be impossible to definitively distinguish homology from convergence, both hypotheses have interesting implications for this protein fold: convergent evolution of at least some 7TMICs from several distinct origins would argue that the fold is an energetically favorable packing of seven TMs; if the superfamily had a single origin, this would further highlight the remarkable potential for sequence diversification while maintaining a common tertiary structure (<xref ref-type="bibr" rid="bib72">Schaeffer and Daggett, 2011</xref>).</p><p>Second, what are the biological roles of different 7TMICs? One aspect of this question pertains to their mechanism of action, that is, whether they assemble in multimeric complexes to form ligand-gated ion channels, similar to insect Ors and Grs. The apparent presence of an anchor domain in all 7TMICs, where most inter-subunit contacts occur in Ors (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>; <xref ref-type="bibr" rid="bib16">Del Mármol et al., 2021</xref>), raises the possibility that complex formation is a common biochemical property. Whether they function as ligand-gated ion channels is not necessarily trivial to answer. Even for insect Grs – for which abundant evidence exists for their in vivo requirement in tastant-evoked neuronal activity (<xref ref-type="bibr" rid="bib12">Chen and Dahanukar, 2020</xref>) – definitive demonstration of their chemical ligand-gated ion conduction properties has (with rare exceptions, e.g., <xref ref-type="bibr" rid="bib52">Morinaga et al., 2022</xref>) been elusive. For PHTF or GrlHz proteins, for example, it is currently difficult to anticipate what might be relevant ligands and we cannot exclude that they have a completely different type of biological activity. Nevertheless, available expression data points to roles of different proteins in specific, but diverse cell types, including chemosensory neurons, (developing) spermatocytes, and muscle. The discoveries in this work should stimulate interest in an even broader community of researchers to understand the evolution and biology of 7TMICs.</p></sec></sec><sec id="s3" sec-type="methods"><title>Methods</title><sec id="s3-1"><title>7TMIC candidate homolog identification</title><p>Structural screens for candidate 7TMIC homologs were performed with the AF-DB search tool on the Dali server (<ext-link ext-link-type="uri" xlink:href="http://ekhidna2.biocenter.helsinki.fi/dali/">http://ekhidna2.biocenter.helsinki.fi/dali/</ext-link>; <xref ref-type="bibr" rid="bib28">Holm, 2022</xref>) using as query the <italic>A. bakeri</italic> Orco structure (PDB 6C70-A) (<xref ref-type="bibr" rid="bib10">Butterwick et al., 2018</xref>). As of December 2022, this server permitted screening of the structural proteome of 47 phylogenetically diverse species. Proteins whose structural models had a Z-score &gt;10 were retained for further analysis. Candidate homologs from these screens were assessed first by using these as queries in Dali AF-DB searches of the <italic>D. melanogaster</italic> proteome to ensure Ors and Grs were the best ‘reverse’ hits, and subsequently for secondary structural features using DeepTMHMM (<ext-link ext-link-type="uri" xlink:href="https://dtu.biolib.com/DeepTMHMM/">https://dtu.biolib.com/DeepTMHMM/</ext-link>) (<xref ref-type="bibr" rid="bib25">Hallgren et al., 2022</xref>) and Phobius (<ext-link ext-link-type="uri" xlink:href="https://phobius.sbc.su.se/">https://phobius.sbc.su.se/</ext-link>) (<xref ref-type="bibr" rid="bib34">Käll et al., 2007</xref>). Of the newly identified <italic>D. melanogaster</italic> Grls, we note that three were initially classified as being members of the <italic>Gr</italic> repertoire (Grl36a (Gr36d), Grl43a (Gr43b), and Grl65a (Gr65a), but later excluded (Flybase [<ext-link ext-link-type="uri" xlink:href="http://flybase.org/">flybase.org/</ext-link>] and [<xref ref-type="bibr" rid="bib64">Robertson et al., 2003</xref>])). We also contrast the term ‘Grl’, referring to the proteins in insects (following nomenclature conventions of <italic>D. melanogaster</italic> [Flybase]) with ‘GRL’, referring to proteins in other animals and more distant eukaryotes; the same acronym does not reflect a monophyletic origin. To identify sequences of candidate homologs from other species that were not screened with Dali AF-DB, PSI-BLAST searches against the NCBI refseq_protein database were performed, using the query sequences indicated in each figure and dataset. PSI-BLAST was run with an expected threshold of 1E-10 until convergence. BLASTP searches for Gr36/59 homologs were performed more permissively, using an expected threshold of 0.05. All sequences analyzed in this work are provided in <xref ref-type="supplementary-material" rid="sdata4">Source data 4</xref>.</p></sec><sec id="s3-2"><title>Structure predictions and analysis</title><p>AlphaFold2 protein models (<xref ref-type="bibr" rid="bib33">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="bib77">Varadi et al., 2022</xref>) were downloaded from the AlphaFold Protein Structure Database (alphafold.ebi.ac.uk; release July 2022). For proteins for which structural predictions were not already available, we generated AlphaFold2 models using ColabFold (<xref ref-type="bibr" rid="bib51">Mirdita et al., 2022</xref>). Positive and negative control protein structures were downloaded from the RCSB Protein Data Bank (PDB codes are indicated in <xref ref-type="table" rid="table1">Table 1</xref>). Pairwise structural similarities of protein models were quantitatively assessed with Dali (<xref ref-type="bibr" rid="bib28">Holm, 2022</xref>) and TM-align (<ext-link ext-link-type="uri" xlink:href="https://zhanggroup.org/TM-align/">https://zhanggroup.org/TM-align/</ext-link>) (<xref ref-type="bibr" rid="bib81">Zhang and Skolnick, 2005</xref>). Proteins were aligned to the same coordinate space with Coot (<ext-link ext-link-type="uri" xlink:href="https://www2.mrc-lmb.cam.ac.uk/personal/pemsley/coot/">https://www2.mrc-lmb.cam.ac.uk/personal/pemsley/coot/</ext-link>) (<xref ref-type="bibr" rid="bib21">Emsley et al., 2010</xref>) and visualized in PyMol v2.5.4. All models analyzed in this work are provided in <xref ref-type="supplementary-material" rid="sdata1">Source data 1</xref>.</p></sec><sec id="s3-3"><title>Phylogenetic and network analyses</title><p>Sequence databases assembled using PSI-BLAST (see above) were first curated in a semi-automated pipeline. First, sequences annotated as ‘partial’ or ‘low quality’, or that contained ambiguous sequence characters (e.g., X), were removed. CD-HIT (<ext-link ext-link-type="uri" xlink:href="http://cd-hit.org">http://cd-hit.org</ext-link>) (<xref ref-type="bibr" rid="bib23">Fu et al., 2012</xref>; <xref ref-type="bibr" rid="bib42">Li and Godzik, 2006</xref>) was used to cluster redundant sequences (100% amino acid identity). Using Phobius TM domain predictions, we removed sequences with fewer than four TMs (this number was chosen to allow for the different sensitivity of Phobius compared to DeepTMHMM). In the final PHTF database, we manually excluded a single sequence as a spurious hit (<italic>B. floridae</italic> XP_035670545.1, zinc transporter ZIP10-like); this sequence sorted independently in first-pass phylogenetic analyses (via FastTree2; <xref ref-type="bibr" rid="bib60">Price et al., 2010</xref>), and a search via InterPro (<ext-link ext-link-type="uri" xlink:href="https://ebi.ac.uk/interpro/">ebi.ac.uk/interpro/</ext-link>) revealed that it had no obvious similarity to the other proposed homologs. The database of Gr39/Gr59 homologs was manually curated due to its relatively small size and accurate automatic annotation by RefSeq; here, we excluded BLAST hits not annotated as Grs, and visually inspected a sequence alignment for good alignment.</p><p>To reduce the large curated sequence databases to a size that could be locally analyzed by both maximum likelihood and Bayesian phylogenetic methods, CD-Hit was used to cluster sequences by 70–90% sequence identity, using the longest sequence as the representative for phylogenetic analyses. The clustering used to generate each phylogeny is indicated in the corresponding figure legend.</p><p>We took two separate approaches to infer ancestry. In initial analyses, when comparing insect Grls to Ors/Grs/non-insect GRLs (<xref ref-type="fig" rid="fig3">Figure 3B</xref>) or for the entire 7TMIC superfamily (<xref ref-type="fig" rid="fig4">Figure 4A</xref>), we observed that extremely low sequence similarity severely constrained our ability to generate meaningful multiple sequence alignments (data not shown). We therefore generated all-to-all sequence similarity networks using MMSeqs2 and inferred phylogenies from these networks by the graph splitting method (both implemented in gs2) (<xref ref-type="bibr" rid="bib48">Matsui and Iwasaki, 2020</xref>). MMSeqs2, as implemented in gs2, employs high sensitivity to sequence similarity and is thus capable of networking non-homologous sequences via spurious sequence identity, should it be present. This method does not distinguish between homology and convergence. Rather, the purpose of this analysis was to make inferences about relatedness under the assumption that all sequences are homologous. In the networks, edge weights are E-values from MMSeqs2. In the graph splitting trees, branch support values were generated by the edge perturbation method (1000 replicates) with a transfer bootstrap expectation (<xref ref-type="bibr" rid="bib40">Lemoine et al., 2018</xref>). For visualization of sequence similarity networks, recursive, same-to-same sequence comparisons (resulting in an E-value of 0) were removed using an R script.</p><p>For all other trees, multiple sequence alignments were generated by MAFFT. We made no a priori assumptions about the alignment, so used default settings. Phylogenetic trees were inferred by maximum likelihood and Bayesian methods. Maximum likelihood trees were generated by IQ-TREE (<xref ref-type="bibr" rid="bib50">Minh et al., 2020</xref>), using the best model selected for each analysis by ModelFinder (<xref ref-type="bibr" rid="bib36">Kalyaanamoorthy et al., 2017</xref>) according to the Bayesian information criterion, and with bootstrapping by UFBoot2 (1000 replicates) (<xref ref-type="bibr" rid="bib26">Hoang et al., 2018</xref>). Bayesian trees were generated by MrBayes (<xref ref-type="bibr" rid="bib67">Ronquist and Huelsenbeck, 2003</xref>) using a mixed amino acid substitution model (Markov chain Monte Carlo analyses run until standard deviation of split frequencies &lt;0.05, with 25% burn in). To generate the most parsimonious hypotheses of protein evolution, we used NOTUNG (<xref ref-type="bibr" rid="bib11">Chen et al., 2000</xref>) to rearrange poorly supported branches and resolve polytomies in a species tree-aware fashion (i.e., favoring speciation to gene duplication/horizontal gene transfer in poorly supported branches and polytomies), using default weights/costs (gene duplication 1.5, transfers 3.0, gene loss 1.0). Branches were eligible for rearrangement at branch support values less than UFboot 95 or posterior probability 0.95. Species trees used for rearrangement were based on the NCBI Taxonomy Common Tree, with polytomies randomly resolved for each analysis using the ape (<xref ref-type="bibr" rid="bib59">Paradis and Schliep, 2019</xref>) and phytools (<xref ref-type="bibr" rid="bib63">Revell, 2012</xref>) packages. Strict consensus trees were generated by comparing the species tree-aware maximum likelihood and Bayesian trees via the consensus function in ape.</p><p>The 7TMIC sequence similarity network was visualized and annotated in Cytoscape (<xref ref-type="bibr" rid="bib74">Shannon et al., 2003</xref>). Trees were visualized and annotated in NOTUNG, iTOL (<ext-link ext-link-type="uri" xlink:href="https://itol.embl.de/">itol.embl.de/</ext-link>) (<xref ref-type="bibr" rid="bib41">Letunic and Bork, 2007</xref>), and Adobe Illustrator. Consensus sequence illustrations were adapted from figures generated by WebLogo (<ext-link ext-link-type="uri" xlink:href="https://weblogo.berkeley.edu/">weblogo.berkeley.edu/</ext-link>) (<xref ref-type="bibr" rid="bib14">Crooks et al., 2004</xref>).</p></sec><sec id="s3-4"><title>Synteny and intron mapping</title><p>The locations of <italic>Grl36a</italic>, <italic>Grl43a</italic>, <italic>Gr36,</italic> and <italic>Gr59c/d</italic> genes in different drosophilids were surveyed using the NCBI Genome Data Viewer (<ext-link ext-link-type="uri" xlink:href="https://ncbi.nlm.nih.gov/genome/gdv/">ncbi.nlm.nih.gov/genome/gdv/</ext-link>) (<xref ref-type="bibr" rid="bib62">Rangwala et al., 2021</xref>). Gene intron-exon structures were manually surveyed using publicly available predictions available on RefSeq (via the Genome Data Viewer) and FlyBase, and visualized in SnapGene. The relative positions of introns were assessed via multiple sequence alignment of the protein sequences; for this analysis, we assumed that that entire sequences could be aligned (global alignment), and thus computed the alignment using the G-INS-i (Needleman-Wunsch) option in MAFFT.</p></sec><sec id="s3-5"><title>Expression analysis</title><p><italic>H. sapiens PHTF1</italic> and <italic>PHTF2</italic> tissue-specific RNA expression data were obtained from the GTEx Portal (GTEx Analysis Release V8 [dbGaP Accession phs000424.v8.p2; <ext-link ext-link-type="uri" xlink:href="https://gtexportal.org/home/datasets">https://gtexportal.org/home/datasets</ext-link>]). Tissue/life stage-specific RNA expression data of <italic>Phtf</italic> and <italic>Grl</italic> genes in <italic>D. melanogaster</italic> were downloaded from the Fly Atlas 2.0 (<ext-link ext-link-type="uri" xlink:href="https://motif.mvls.gla.ac.uk/FlyAtlas2">https://motif.mvls.gla.ac.uk/FlyAtlas2</ext-link>) (<xref ref-type="bibr" rid="bib38">Krause et al., 2022</xref>) or, for the labellum, from <xref ref-type="bibr" rid="bib19">Dweck et al., 2021</xref>. <italic>D. melanogaster</italic> scRNA-seq data was from the Fly Cell Atlas (<xref ref-type="bibr" rid="bib43">Li et al., 2022</xref>): proboscis/maxillary palp (10× stringent dataset) and testis/seminal vesicle (10× relaxed dataset), visualized as HVG tSNE or UMAP plots, respectively, in the SCope interface (<ext-link ext-link-type="uri" xlink:href="https://scope.aertslab.org/#/FlyCellAtlas">https://scope.aertslab.org/#/FlyCellAtlas</ext-link>) (<xref ref-type="bibr" rid="bib15">Davie et al., 2018</xref>).</p></sec></sec></body><back><sec sec-type="additional-information" id="s4"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Funding acquisition, Investigation, Methodology, Project administration, Supervision, Validation, Visualization, Writing – original draft, Writing – review and editing, Conceived the project, and performed structural screens/analyses and expression analyses</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Data curation, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing – original draft, Project administration, Writing – review and editing, Performed sequence-based homolog identification, phylogenetic and network analyses, and gene structure/synteny and protein motif analyses</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s5"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-85537-mdarchecklist1-v2.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material><supplementary-material id="sdata1"><label>Source data 1.</label><caption><title>AlphaFold2 models.</title><p>Models of proteins analyzed in this work, either downloaded from the AlphaFold Protein Structure Database or, where not already available, predicted using the AlphaFold2 algorithm implemented in ColabFold (<xref ref-type="bibr" rid="bib51">Mirdita et al., 2022</xref>). The four-letter code in the filename represents the first letter of the genus and the first three letters of the species (e.g., ‘Dmel’ = <italic>D. melanogaster</italic>); species names are given in full in the figures.</p></caption><media xlink:href="elife-85537-data1-v2.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="sdata2"><label>Source data 2.</label><caption><title>Dali screen search results.</title><p>Individual text files represent the output of the Dali AF-DB search using <italic>A. bakeri</italic> Or co-receptor (Orco) chain A (PDB 6C70-A) as query and the structural proteome dataset of the indicated species (note the datasets are from version 1 of the AlphaFold Protein Structure Database; subsequent, improved models were used for the pairwise comparisons in <xref ref-type="table" rid="table1">Table 1</xref>). The four-letter codes in the file names and job titles are as described for <xref ref-type="supplementary-material" rid="sdata1">Source data 1</xref>.</p></caption><media xlink:href="elife-85537-data2-v2.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="sdata3"><label>Source data 3.</label><caption><title>Reverse Dali search results.</title><p>Individual text files represent the output of the Dali AF-DB search using the indicated query candidate seven transmembrane domain ion channels (7TMICs) from <italic>Trypanosoma</italic> (GRL1), <italic>D. melanogaster</italic> (Grls), or <italic>H. sapiens</italic> (PHTF1/2) and the structural proteomic dataset of <italic>D. melanogaster.</italic></p></caption><media xlink:href="elife-85537-data3-v2.zip" mimetype="application" mime-subtype="zip"/></supplementary-material><supplementary-material id="sdata4"><label>Source data 4.</label><caption><title>All uncurated PSI-BLAST sequence databases.</title><p>Each of the FASTA filenames is formatted as follows (with the exception of the <italic>D. melanogaster</italic> odorant receptor (Or) and gustatory receptor (Gr) sequences, which were collected manually from FlyBase): ProteinFamily-QuerySpecies-QuerySequence.fasta.</p></caption><media xlink:href="elife-85537-data4-v2.zip" mimetype="application" mime-subtype="zip"/></supplementary-material></sec><sec sec-type="data-availability" id="s6"><title>Data availability</title><p>All data generated or analysed during this study are included in the manuscript and supporting files.</p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset1"><person-group person-group-type="author"><name><surname>Krause</surname><given-names>SA</given-names></name><name><surname>Overend</surname><given-names>G</given-names></name><name><surname>Dow</surname><given-names>JAT</given-names></name><name><surname>Leader</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>FlyAtlas 2 in 2022: enhancements to the <italic>Drosophila melanogaster</italic> expression atlas</data-title><source>motif</source><pub-id pub-id-type="accession" xlink:href="https://motif.mvls.gla.ac.uk/FlyAtlas2">FlyAtlas2</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><collab>GTEx Portal</collab></person-group><year iso-8601-date="2021">2021</year><data-title>The Genotype-Tissue Expression (GTEx) Project</data-title><source>dbGaP</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000424.v8.p2">phs000424.v8.p2</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We are very grateful to Julia Santiago for advice on protein structure comparisons and instruction on Coot and Pymol. We acknowledge use of data from the Genotype-Tissue Expression (GTEx) Project, which is supported by the Common Fund of the Office of the Director of the National Institutes of Health, and by NCI, NHGRI, NHLBI, NIDA, NIMH, and NINDS. We thank Roman Arguello, Jamin Letcher, Julia Santiago, and members of the Benton laboratory for comments on the manuscript. Research in RB’s laboratory is supported by the University of Lausanne, an ERC Advanced Grant (833548) and the Swiss National Science Foundation. NJH is supported by a Human Frontier Science Program Long-Term Postdoctoral Fellowship (LT-0003/2022L).</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alva</surname><given-names>V</given-names></name><name><surname>Remmert</surname><given-names>M</given-names></name><name><surname>Biegert</surname><given-names>A</given-names></name><name><surname>Lupas</surname><given-names>AN</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A galaxy of folds</article-title><source>Protein Science</source><volume>19</volume><fpage>124</fpage><lpage>130</lpage><pub-id pub-id-type="doi">10.1002/pro.297</pub-id><pub-id pub-id-type="pmid">19937658</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alva</surname><given-names>V</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name><name><surname>Lupas</surname><given-names>AN</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A vocabulary of ancient peptides at the origin of folded proteins</article-title><source>eLife</source><volume>4</volume><elocation-id>e09410</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.09410</pub-id><pub-id pub-id-type="pmid">26653858</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benton</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>On the origin of smell: odorant receptors in insects</article-title><source>Cellular and Molecular Life Sciences</source><volume>63</volume><fpage>1579</fpage><lpage>1585</lpage><pub-id pub-id-type="doi">10.1007/s00018-006-6130-7</pub-id><pub-id pub-id-type="pmid">16786219</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benton</surname><given-names>R</given-names></name><name><surname>Sachse</surname><given-names>S</given-names></name><name><surname>Michnick</surname><given-names>SW</given-names></name><name><surname>Vosshall</surname><given-names>LB</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Atypical membrane topology and heteromeric function of <italic>Drosophila</italic> odorant receptors in vivo</article-title><source>PLOS Biology</source><volume>4</volume><elocation-id>e20</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0040020</pub-id><pub-id pub-id-type="pmid">16402857</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benton</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Multigene family evolution: perspectives from insect chemoreceptors</article-title><source>Trends in Ecology &amp; Evolution</source><volume>30</volume><fpage>590</fpage><lpage>600</lpage><pub-id pub-id-type="doi">10.1016/j.tree.2015.07.009</pub-id><pub-id pub-id-type="pmid">26411616</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benton</surname><given-names>R</given-names></name><name><surname>Dessimoz</surname><given-names>C</given-names></name><name><surname>Moi</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A putative origin of the insect chemosensory receptor superfamily in the last common eukaryotic ancestor</article-title><source>eLife</source><volume>9</volume><elocation-id>e62507</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.62507</pub-id><pub-id pub-id-type="pmid">33274716</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bergsten</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>A review of long-branch attraction</article-title><source>Cladistics</source><volume>21</volume><fpage>163</fpage><lpage>193</lpage><pub-id pub-id-type="doi">10.1111/j.1096-0031.2005.00059.x</pub-id><pub-id pub-id-type="pmid">34892859</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brand</surname><given-names>P</given-names></name><name><surname>Robertson</surname><given-names>HM</given-names></name><name><surname>Lin</surname><given-names>W</given-names></name><name><surname>Pothula</surname><given-names>R</given-names></name><name><surname>Klingeman</surname><given-names>WE</given-names></name><name><surname>Jurat-Fuentes</surname><given-names>JL</given-names></name><name><surname>Johnson</surname><given-names>BR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The origin of the odorant receptor gene family in insects</article-title><source>eLife</source><volume>7</volume><elocation-id>e38340</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.38340</pub-id><pub-id pub-id-type="pmid">30063003</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burki</surname><given-names>F</given-names></name><name><surname>Roger</surname><given-names>AJ</given-names></name><name><surname>Brown</surname><given-names>MW</given-names></name><name><surname>Simpson</surname><given-names>AGB</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The new tree of eukaryotes</article-title><source>Trends in Ecology &amp; Evolution</source><volume>35</volume><fpage>43</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1016/j.tree.2019.08.008</pub-id><pub-id pub-id-type="pmid">31606140</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Butterwick</surname><given-names>JA</given-names></name><name><surname>Del Mármol</surname><given-names>J</given-names></name><name><surname>Kim</surname><given-names>KH</given-names></name><name><surname>Kahlson</surname><given-names>MA</given-names></name><name><surname>Rogow</surname><given-names>JA</given-names></name><name><surname>Walz</surname><given-names>T</given-names></name><name><surname>Ruta</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Cryo-Em structure of the insect olfactory receptor Orco</article-title><source>Nature</source><volume>560</volume><fpage>447</fpage><lpage>452</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0420-8</pub-id><pub-id pub-id-type="pmid">30111839</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>K</given-names></name><name><surname>Durand</surname><given-names>D</given-names></name><name><surname>Farach-Colton</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>NOTUNG: a program for dating gene duplications and optimizing gene family trees</article-title><source>Journal of Computational Biology</source><volume>7</volume><fpage>429</fpage><lpage>447</lpage><pub-id pub-id-type="doi">10.1089/106652700750050871</pub-id><pub-id pub-id-type="pmid">11108472</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y-CD</given-names></name><name><surname>Dahanukar</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Recent advances in the genetic basis of taste detection in <italic>Drosophila</italic></article-title><source>Cellular and Molecular Life Sciences</source><volume>77</volume><fpage>1087</fpage><lpage>1101</lpage><pub-id pub-id-type="doi">10.1007/s00018-019-03320-0</pub-id><pub-id pub-id-type="pmid">31598735</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chi</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>F</given-names></name><name><surname>Ding</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>PHTF2 regulates lipids metabolism in gastric cancer</article-title><source>Aging</source><volume>12</volume><fpage>6600</fpage><lpage>6610</lpage><pub-id pub-id-type="doi">10.18632/aging.102995</pub-id><pub-id pub-id-type="pmid">32335542</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Crooks</surname><given-names>GE</given-names></name><name><surname>Hon</surname><given-names>G</given-names></name><name><surname>Chandonia</surname><given-names>JM</given-names></name><name><surname>Brenner</surname><given-names>SE</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Weblogo: a sequence logo generator: Figure 1</article-title><source>Genome Research</source><volume>14</volume><fpage>1188</fpage><lpage>1190</lpage><pub-id pub-id-type="doi">10.1101/gr.849004</pub-id><pub-id pub-id-type="pmid">15173120</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davie</surname><given-names>K</given-names></name><name><surname>Janssens</surname><given-names>J</given-names></name><name><surname>Koldere</surname><given-names>D</given-names></name><name><surname>De Waegeneer</surname><given-names>M</given-names></name><name><surname>Pech</surname><given-names>U</given-names></name><name><surname>Kreft</surname><given-names>Ł</given-names></name><name><surname>Aibar</surname><given-names>S</given-names></name><name><surname>Makhzami</surname><given-names>S</given-names></name><name><surname>Christiaens</surname><given-names>V</given-names></name><name><surname>Bravo González-Blas</surname><given-names>C</given-names></name><name><surname>Poovathingal</surname><given-names>S</given-names></name><name><surname>Hulselmans</surname><given-names>G</given-names></name><name><surname>Spanier</surname><given-names>KI</given-names></name><name><surname>Moerman</surname><given-names>T</given-names></name><name><surname>Vanspauwen</surname><given-names>B</given-names></name><name><surname>Geurs</surname><given-names>S</given-names></name><name><surname>Voet</surname><given-names>T</given-names></name><name><surname>Lammertyn</surname><given-names>J</given-names></name><name><surname>Thienpont</surname><given-names>B</given-names></name><name><surname>Liu</surname><given-names>S</given-names></name><name><surname>Konstantinides</surname><given-names>N</given-names></name><name><surname>Fiers</surname><given-names>M</given-names></name><name><surname>Verstreken</surname><given-names>P</given-names></name><name><surname>Aerts</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A single-cell transcriptome atlas of the aging <italic>Drosophila</italic> brain</article-title><source>Cell</source><volume>174</volume><fpage>982</fpage><lpage>998</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.05.057</pub-id><pub-id pub-id-type="pmid">29909982</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Del Mármol</surname><given-names>J</given-names></name><name><surname>Yedlin</surname><given-names>MA</given-names></name><name><surname>Ruta</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The structural basis of odorant recognition in insect olfactory receptors</article-title><source>Nature</source><volume>597</volume><fpage>126</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03794-8</pub-id><pub-id pub-id-type="pmid">34349260</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><collab><italic>Drosophila</italic> Odorant Receptor Nomenclature Committee</collab></person-group><year iso-8601-date="2000">2000</year><article-title>A unified nomenclature system for the <italic>Drosophila</italic> odorant receptors</article-title><source>Cell</source><volume>102</volume><fpage>145</fpage><lpage>146</lpage><pub-id pub-id-type="doi">10.1016/S0092-8674(00)00020-9</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dunipace</surname><given-names>L</given-names></name><name><surname>Meister</surname><given-names>S</given-names></name><name><surname>McNealy</surname><given-names>C</given-names></name><name><surname>Amrein</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Spatially restricted expression of candidate taste receptors in the <italic>Drosophila</italic> gustatory system</article-title><source>Current Biology</source><volume>11</volume><fpage>822</fpage><lpage>835</lpage><pub-id pub-id-type="doi">10.1016/s0960-9822(01)00258-5</pub-id><pub-id pub-id-type="pmid">11516643</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dweck</surname><given-names>HK</given-names></name><name><surname>Talross</surname><given-names>GJ</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Carlson</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Evolutionary shifts in taste coding in the fruit pest <italic>Drosophila</italic> suzukii</article-title><source>eLife</source><volume>10</volume><elocation-id>e64317</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.64317</pub-id><pub-id pub-id-type="pmid">33616529</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Edwards</surname><given-names>SL</given-names></name><name><surname>Charlie</surname><given-names>NK</given-names></name><name><surname>Milfort</surname><given-names>MC</given-names></name><name><surname>Brown</surname><given-names>BS</given-names></name><name><surname>Gravlin</surname><given-names>CN</given-names></name><name><surname>Knecht</surname><given-names>JE</given-names></name><name><surname>Miller</surname><given-names>KG</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>A novel molecular solution for ultraviolet light detection in <italic>Caenorhabditis elegans</italic></article-title><source>PLOS Biology</source><volume>6</volume><elocation-id>e198</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0060198</pub-id><pub-id pub-id-type="pmid">18687026</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Emsley</surname><given-names>P</given-names></name><name><surname>Lohkamp</surname><given-names>B</given-names></name><name><surname>Scott</surname><given-names>WG</given-names></name><name><surname>Cowtan</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Features and development of coot</article-title><source>Acta Crystallographica. Section D, Biological Crystallography</source><volume>66</volume><fpage>486</fpage><lpage>501</lpage><pub-id pub-id-type="doi">10.1107/S0907444910007493</pub-id><pub-id pub-id-type="pmid">20383002</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Freeman</surname><given-names>EG</given-names></name><name><surname>Dahanukar</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Molecular neurobiology of <italic>Drosophila</italic> taste</article-title><source>Current Opinion in Neurobiology</source><volume>34</volume><fpage>140</fpage><lpage>148</lpage><pub-id pub-id-type="doi">10.1016/j.conb.2015.06.001</pub-id><pub-id pub-id-type="pmid">26102453</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>L</given-names></name><name><surname>Niu</surname><given-names>B</given-names></name><name><surname>Zhu</surname><given-names>Z</given-names></name><name><surname>Wu</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>CD-HIT: accelerated for clustering the next-generation sequencing data</article-title><source>Bioinformatics</source><volume>28</volume><fpage>3150</fpage><lpage>3152</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id><pub-id pub-id-type="pmid">23060610</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gong</surname><given-names>J</given-names></name><name><surname>Yuan</surname><given-names>Y</given-names></name><name><surname>Ward</surname><given-names>A</given-names></name><name><surname>Kang</surname><given-names>L</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Wu</surname><given-names>Z</given-names></name><name><surname>Peng</surname><given-names>J</given-names></name><name><surname>Feng</surname><given-names>Z</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Xu</surname><given-names>XZS</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The <italic>C. elegans</italic> taste receptor homolog LITE-1 is a photoreceptor</article-title><source>Cell</source><volume>167</volume><fpage>1252</fpage><lpage>1263</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.10.053</pub-id><pub-id pub-id-type="pmid">27863243</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Hallgren</surname><given-names>J</given-names></name><name><surname>Tsirigos</surname><given-names>KD</given-names></name><name><surname>Pedersen</surname><given-names>MD</given-names></name><name><surname>Almagro Armenteros</surname><given-names>JJ</given-names></name><name><surname>Marcatili</surname><given-names>P</given-names></name><name><surname>Nielsen</surname><given-names>H</given-names></name><name><surname>Krogh</surname><given-names>A</given-names></name><name><surname>Winther</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>DeepTMHMM predicts alpha and beta transmembrane proteins using deep neural networks</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.04.08.487609</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hoang</surname><given-names>DT</given-names></name><name><surname>Chernomor</surname><given-names>O</given-names></name><name><surname>von Haeseler</surname><given-names>A</given-names></name><name><surname>Minh</surname><given-names>BQ</given-names></name><name><surname>Vinh</surname><given-names>LS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>UFBoot2: improving the ultrafast bootstrap approximation</article-title><source>Molecular Biology and Evolution</source><volume>35</volume><fpage>518</fpage><lpage>522</lpage><pub-id pub-id-type="doi">10.1093/molbev/msx281</pub-id><pub-id pub-id-type="pmid">29077904</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Holm</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Using dali for protein structure comparison</article-title><source>Methods in Molecular Biology</source><volume>2112</volume><fpage>29</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1007/978-1-0716-0270-6_3</pub-id><pub-id pub-id-type="pmid">32006276</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Holm</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Dali server: structural unification of protein families</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>W210</fpage><lpage>W215</lpage><pub-id pub-id-type="doi">10.1093/nar/gkac387</pub-id><pub-id pub-id-type="pmid">35610055</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Geng</surname><given-names>S</given-names></name><name><surname>Weng</surname><given-names>J</given-names></name><name><surname>Lu</surname><given-names>Z</given-names></name><name><surname>Zeng</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>M</given-names></name><name><surname>Deng</surname><given-names>C</given-names></name><name><surname>Wu</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Du</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Analysis of the expression of PHTF1 and related genes in acute lymphoblastic leukemia</article-title><source>Cancer Cell International</source><volume>15</volume><elocation-id>93</elocation-id><pub-id pub-id-type="doi">10.1186/s12935-015-0242-9</pub-id><pub-id pub-id-type="pmid">26448723</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Illergård</surname><given-names>K</given-names></name><name><surname>Ardell</surname><given-names>DH</given-names></name><name><surname>Elofsson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Structure is three to ten times more conserved than sequence -- a study of structural response in protein cores</article-title><source>Proteins</source><volume>77</volume><fpage>499</fpage><lpage>508</lpage><pub-id pub-id-type="doi">10.1002/prot.22458</pub-id><pub-id pub-id-type="pmid">19507241</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jones</surname><given-names>WD</given-names></name><name><surname>Nguyen</surname><given-names>TAT</given-names></name><name><surname>Kloss</surname><given-names>B</given-names></name><name><surname>Lee</surname><given-names>KJ</given-names></name><name><surname>Vosshall</surname><given-names>LB</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Functional conservation of an insect odorant receptor gene across 250 million years of evolution</article-title><source>Current Biology</source><volume>15</volume><fpage>R119</fpage><lpage>R121</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2005.02.007</pub-id><pub-id pub-id-type="pmid">15723778</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Joseph</surname><given-names>RM</given-names></name><name><surname>Carlson</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title><italic>Drosophila</italic> chemoreceptors: a molecular interface between the chemical world and the brain</article-title><source>Trends in Genetics</source><volume>31</volume><fpage>683</fpage><lpage>695</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2015.09.005</pub-id><pub-id pub-id-type="pmid">26477743</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Evans</surname><given-names>R</given-names></name><name><surname>Pritzel</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Bates</surname><given-names>R</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Potapenko</surname><given-names>A</given-names></name><name><surname>Bridgland</surname><given-names>A</given-names></name><name><surname>Meyer</surname><given-names>C</given-names></name><name><surname>Kohl</surname><given-names>SAA</given-names></name><name><surname>Ballard</surname><given-names>AJ</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Romera-Paredes</surname><given-names>B</given-names></name><name><surname>Nikolov</surname><given-names>S</given-names></name><name><surname>Jain</surname><given-names>R</given-names></name><name><surname>Adler</surname><given-names>J</given-names></name><name><surname>Back</surname><given-names>T</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Reiman</surname><given-names>D</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Zielinski</surname><given-names>M</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Pacholska</surname><given-names>M</given-names></name><name><surname>Berghammer</surname><given-names>T</given-names></name><name><surname>Bodenstein</surname><given-names>S</given-names></name><name><surname>Silver</surname><given-names>D</given-names></name><name><surname>Vinyals</surname><given-names>O</given-names></name><name><surname>Senior</surname><given-names>AW</given-names></name><name><surname>Kavukcuoglu</surname><given-names>K</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Highly accurate protein structure prediction with alphafold</article-title><source>Nature</source><volume>596</volume><fpage>583</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id><pub-id pub-id-type="pmid">34265844</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Käll</surname><given-names>L</given-names></name><name><surname>Krogh</surname><given-names>A</given-names></name><name><surname>Sonnhammer</surname><given-names>ELL</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Advantages of combined transmembrane topology and signal peptide prediction -- the phobius web server</article-title><source>Nucleic Acids Research</source><volume>35</volume><fpage>W429</fpage><lpage>W432</lpage><pub-id pub-id-type="doi">10.1093/nar/gkm256</pub-id><pub-id pub-id-type="pmid">17483518</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Källberg</surname><given-names>M</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>S</given-names></name><name><surname>Peng</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Lu</surname><given-names>H</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Template-Based protein structure modeling using the raptorx web server</article-title><source>Nature Protocols</source><volume>7</volume><fpage>1511</fpage><lpage>1522</lpage><pub-id pub-id-type="doi">10.1038/nprot.2012.085</pub-id><pub-id pub-id-type="pmid">22814390</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kalyaanamoorthy</surname><given-names>S</given-names></name><name><surname>Minh</surname><given-names>BQ</given-names></name><name><surname>Wong</surname><given-names>TKF</given-names></name><name><surname>von Haeseler</surname><given-names>A</given-names></name><name><surname>Jermiin</surname><given-names>LS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>ModelFinder: fast model selection for accurate phylogenetic estimates</article-title><source>Nature Methods</source><volume>14</volume><fpage>587</fpage><lpage>589</lpage><pub-id pub-id-type="doi">10.1038/nmeth.4285</pub-id><pub-id pub-id-type="pmid">28481363</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>Y</given-names></name><name><surname>Kim</surname><given-names>SH</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Wd40-repeat proteins in ciliopathies and congenital disorders of endocrine system</article-title><source>Endocrinology and Metabolism</source><volume>35</volume><fpage>494</fpage><lpage>506</lpage><pub-id pub-id-type="doi">10.3803/EnM.2020.302</pub-id><pub-id pub-id-type="pmid">32894826</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krause</surname><given-names>SA</given-names></name><name><surname>Overend</surname><given-names>G</given-names></name><name><surname>Dow</surname><given-names>JAT</given-names></name><name><surname>Leader</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>FlyAtlas 2 in 2022: enhancements to the <italic>Drosophila melanogaster</italic> expression atlas</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D1010</fpage><lpage>D1015</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab971</pub-id><pub-id pub-id-type="pmid">34718735</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Larsson</surname><given-names>MC</given-names></name><name><surname>Domingos</surname><given-names>AI</given-names></name><name><surname>Jones</surname><given-names>WD</given-names></name><name><surname>Chiappe</surname><given-names>ME</given-names></name><name><surname>Amrein</surname><given-names>H</given-names></name><name><surname>Vosshall</surname><given-names>LB</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Or83b encodes a broadly expressed odorant receptor essential for <italic>Drosophila</italic> olfaction</article-title><source>Neuron</source><volume>43</volume><fpage>703</fpage><lpage>714</lpage><pub-id pub-id-type="doi">10.1016/j.neuron.2004.08.019</pub-id><pub-id pub-id-type="pmid">15339651</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lemoine</surname><given-names>F</given-names></name><name><surname>Domelevo Entfellner</surname><given-names>J-B</given-names></name><name><surname>Wilkinson</surname><given-names>E</given-names></name><name><surname>Correia</surname><given-names>D</given-names></name><name><surname>Dávila Felipe</surname><given-names>M</given-names></name><name><surname>De Oliveira</surname><given-names>T</given-names></name><name><surname>Gascuel</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Renewing felsenstein’s phylogenetic bootstrap in the era of big data</article-title><source>Nature</source><volume>556</volume><fpage>452</fpage><lpage>456</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0043-0</pub-id><pub-id pub-id-type="pmid">29670290</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Letunic</surname><given-names>I</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Interactive tree of life (itol): an online tool for phylogenetic tree display and annotation</article-title><source>Bioinformatics</source><volume>23</volume><fpage>127</fpage><lpage>128</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btl529</pub-id><pub-id pub-id-type="pmid">17050570</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>W</given-names></name><name><surname>Godzik</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Cd-hit: a fast program for clustering and comparing large sets of protein or nucleotide sequences</article-title><source>Bioinformatics</source><volume>22</volume><fpage>1658</fpage><lpage>1659</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btl158</pub-id><pub-id pub-id-type="pmid">16731699</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Janssens</surname><given-names>J</given-names></name><name><surname>De Waegeneer</surname><given-names>M</given-names></name><name><surname>Kolluru</surname><given-names>SS</given-names></name><name><surname>Davie</surname><given-names>K</given-names></name><name><surname>Gardeux</surname><given-names>V</given-names></name><name><surname>Saelens</surname><given-names>W</given-names></name><name><surname>David</surname><given-names>FPA</given-names></name><name><surname>Brbić</surname><given-names>M</given-names></name><name><surname>Spanier</surname><given-names>K</given-names></name><name><surname>Leskovec</surname><given-names>J</given-names></name><name><surname>McLaughlin</surname><given-names>CN</given-names></name><name><surname>Xie</surname><given-names>Q</given-names></name><name><surname>Jones</surname><given-names>RC</given-names></name><name><surname>Brueckner</surname><given-names>K</given-names></name><name><surname>Shim</surname><given-names>J</given-names></name><name><surname>Tattikota</surname><given-names>SG</given-names></name><name><surname>Schnorrer</surname><given-names>F</given-names></name><name><surname>Rust</surname><given-names>K</given-names></name><name><surname>Nystul</surname><given-names>TG</given-names></name><name><surname>Carvalho-Santos</surname><given-names>Z</given-names></name><name><surname>Ribeiro</surname><given-names>C</given-names></name><name><surname>Pal</surname><given-names>S</given-names></name><name><surname>Mahadevaraju</surname><given-names>S</given-names></name><name><surname>Przytycka</surname><given-names>TM</given-names></name><name><surname>Allen</surname><given-names>AM</given-names></name><name><surname>Goodwin</surname><given-names>SF</given-names></name><name><surname>Berry</surname><given-names>CW</given-names></name><name><surname>Fuller</surname><given-names>MT</given-names></name><name><surname>White-Cooper</surname><given-names>H</given-names></name><name><surname>Matunis</surname><given-names>EL</given-names></name><name><surname>DiNardo</surname><given-names>S</given-names></name><name><surname>Galenza</surname><given-names>A</given-names></name><name><surname>O’Brien</surname><given-names>LE</given-names></name><name><surname>Dow</surname><given-names>JAT</given-names></name><collab>FCA Consortium§</collab><name><surname>Jasper</surname><given-names>H</given-names></name><name><surname>Oliver</surname><given-names>B</given-names></name><name><surname>Perrimon</surname><given-names>N</given-names></name><name><surname>Deplancke</surname><given-names>B</given-names></name><name><surname>Quake</surname><given-names>SR</given-names></name><name><surname>Luo</surname><given-names>L</given-names></name><name><surname>Aerts</surname><given-names>S</given-names></name><name><surname>Agarwal</surname><given-names>D</given-names></name><name><surname>Ahmed-Braimah</surname><given-names>Y</given-names></name><name><surname>Arbeitman</surname><given-names>M</given-names></name><name><surname>Ariss</surname><given-names>MM</given-names></name><name><surname>Augsburger</surname><given-names>J</given-names></name><name><surname>Ayush</surname><given-names>K</given-names></name><name><surname>Baker</surname><given-names>CC</given-names></name><name><surname>Banisch</surname><given-names>T</given-names></name><name><surname>Birker</surname><given-names>K</given-names></name><name><surname>Bodmer</surname><given-names>R</given-names></name><name><surname>Bolival</surname><given-names>B</given-names></name><name><surname>Brantley</surname><given-names>SE</given-names></name><name><surname>Brill</surname><given-names>JA</given-names></name><name><surname>Brown</surname><given-names>NC</given-names></name><name><surname>Buehner</surname><given-names>NA</given-names></name><name><surname>Cai</surname><given-names>XT</given-names></name><name><surname>Cardoso-Figueiredo</surname><given-names>R</given-names></name><name><surname>Casares</surname><given-names>F</given-names></name><name><surname>Chang</surname><given-names>A</given-names></name><name><surname>Clandinin</surname><given-names>TR</given-names></name><name><surname>Crasta</surname><given-names>S</given-names></name><name><surname>Desplan</surname><given-names>C</given-names></name><name><surname>Detweiler</surname><given-names>AM</given-names></name><name><surname>Dhakan</surname><given-names>DB</given-names></name><name><surname>Donà</surname><given-names>E</given-names></name><name><surname>Engert</surname><given-names>S</given-names></name><name><surname>Floc’hlay</surname><given-names>S</given-names></name><name><surname>George</surname><given-names>N</given-names></name><name><surname>González-Segarra</surname><given-names>AJ</given-names></name><name><surname>Groves</surname><given-names>AK</given-names></name><name><surname>Gumbin</surname><given-names>S</given-names></name><name><surname>Guo</surname><given-names>Y</given-names></name><name><surname>Harris</surname><given-names>DE</given-names></name><name><surname>Heifetz</surname><given-names>Y</given-names></name><name><surname>Holtz</surname><given-names>SL</given-names></name><name><surname>Horns</surname><given-names>F</given-names></name><name><surname>Hudry</surname><given-names>B</given-names></name><name><surname>Hung</surname><given-names>R-J</given-names></name><name><surname>Jan</surname><given-names>YN</given-names></name><name><surname>Jaszczak</surname><given-names>JS</given-names></name><name><surname>Jefferis</surname><given-names>GSXE</given-names></name><name><surname>Karkanias</surname><given-names>J</given-names></name><name><surname>Karr</surname><given-names>TL</given-names></name><name><surname>Katheder</surname><given-names>NS</given-names></name><name><surname>Kezos</surname><given-names>J</given-names></name><name><surname>Kim</surname><given-names>AA</given-names></name><name><surname>Kim</surname><given-names>SK</given-names></name><name><surname>Kockel</surname><given-names>L</given-names></name><name><surname>Konstantinides</surname><given-names>N</given-names></name><name><surname>Kornberg</surname><given-names>TB</given-names></name><name><surname>Krause</surname><given-names>HM</given-names></name><name><surname>Labott</surname><given-names>AT</given-names></name><name><surname>Laturney</surname><given-names>M</given-names></name><name><surname>Lehmann</surname><given-names>R</given-names></name><name><surname>Leinwand</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>JSS</given-names></name><name><surname>Li</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>T</given-names></name><name><surname>Litovchenko</surname><given-names>M</given-names></name><name><surname>Liu</surname><given-names>H-H</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Lu</surname><given-names>T-C</given-names></name><name><surname>Manning</surname><given-names>J</given-names></name><name><surname>Mase</surname><given-names>A</given-names></name><name><surname>Matera-Vatnick</surname><given-names>M</given-names></name><name><surname>Matias</surname><given-names>NR</given-names></name><name><surname>McDonough-Goldstein</surname><given-names>CE</given-names></name><name><surname>McGeever</surname><given-names>A</given-names></name><name><surname>McLachlan</surname><given-names>AD</given-names></name><name><surname>Moreno-Roman</surname><given-names>P</given-names></name><name><surname>Neff</surname><given-names>N</given-names></name><name><surname>Neville</surname><given-names>M</given-names></name><name><surname>Ngo</surname><given-names>S</given-names></name><name><surname>Nielsen</surname><given-names>T</given-names></name><name><surname>O’Brien</surname><given-names>CE</given-names></name><name><surname>Osumi-Sutherland</surname><given-names>D</given-names></name><name><surname>Özel</surname><given-names>MN</given-names></name><name><surname>Papatheodorou</surname><given-names>I</given-names></name><name><surname>Petkovic</surname><given-names>M</given-names></name><name><surname>Pilgrim</surname><given-names>C</given-names></name><name><surname>Pisco</surname><given-names>AO</given-names></name><name><surname>Reisenman</surname><given-names>C</given-names></name><name><surname>Sanders</surname><given-names>EN</given-names></name><name><surname>dos Santos</surname><given-names>G</given-names></name><name><surname>Scott</surname><given-names>K</given-names></name><name><surname>Sherlekar</surname><given-names>A</given-names></name><name><surname>Shiu</surname><given-names>P</given-names></name><name><surname>Sims</surname><given-names>D</given-names></name><name><surname>Sit</surname><given-names>RV</given-names></name><name><surname>Slaidina</surname><given-names>M</given-names></name><name><surname>Smith</surname><given-names>HE</given-names></name><name><surname>Sterne</surname><given-names>G</given-names></name><name><surname>Su</surname><given-names>Y-H</given-names></name><name><surname>Sutton</surname><given-names>D</given-names></name><name><surname>Tamayo</surname><given-names>M</given-names></name><name><surname>Tan</surname><given-names>M</given-names></name><name><surname>Tastekin</surname><given-names>I</given-names></name><name><surname>Treiber</surname><given-names>C</given-names></name><name><surname>Vacek</surname><given-names>D</given-names></name><name><surname>Vogler</surname><given-names>G</given-names></name><name><surname>Waddell</surname><given-names>S</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Wilson</surname><given-names>RI</given-names></name><name><surname>Wolfner</surname><given-names>MF</given-names></name><name><surname>Wong</surname><given-names>Y-CE</given-names></name><name><surname>Xie</surname><given-names>A</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name><name><surname>Yamamoto</surname><given-names>S</given-names></name><name><surname>Yan</surname><given-names>J</given-names></name><name><surname>Yao</surname><given-names>Z</given-names></name><name><surname>Yoda</surname><given-names>K</given-names></name><name><surname>Zhu</surname><given-names>R</given-names></name><name><surname>Zinzen</surname><given-names>RP</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Fly cell atlas: a single-nucleus transcriptomic atlas of the adult fruit fly</article-title><source>Science</source><volume>375</volume><elocation-id>eabk2432</elocation-id><pub-id pub-id-type="doi">10.1126/science.abk2432</pub-id><pub-id pub-id-type="pmid">35239393</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Ward</surname><given-names>A</given-names></name><name><surname>Gao</surname><given-names>J</given-names></name><name><surname>Dong</surname><given-names>Y</given-names></name><name><surname>Nishio</surname><given-names>N</given-names></name><name><surname>Inada</surname><given-names>H</given-names></name><name><surname>Kang</surname><given-names>L</given-names></name><name><surname>Yu</surname><given-names>Y</given-names></name><name><surname>Ma</surname><given-names>D</given-names></name><name><surname>Xu</surname><given-names>T</given-names></name><name><surname>Mori</surname><given-names>I</given-names></name><name><surname>Xie</surname><given-names>Z</given-names></name><name><surname>Xu</surname><given-names>XZS</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title><italic>C. elegans</italic> phototransduction requires a G protein–dependent cGMP pathway and a taste receptor homolog</article-title><source>Nature Neuroscience</source><volume>13</volume><fpage>715</fpage><lpage>722</lpage><pub-id pub-id-type="doi">10.1038/nn.2540</pub-id><pub-id pub-id-type="pmid">20436480</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lupas</surname><given-names>AN</given-names></name><name><surname>Ponting</surname><given-names>CP</given-names></name><name><surname>Russell</surname><given-names>RB</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>On the evolution of protein folds: are similar motifs in different protein folds the result of convergence, insertion, or relics of an ancient peptide world?</article-title><source>Journal of Structural Biology</source><volume>134</volume><fpage>191</fpage><lpage>203</lpage><pub-id pub-id-type="doi">10.1006/jsbi.2001.4393</pub-id><pub-id pub-id-type="pmid">11551179</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mackin</surname><given-names>KA</given-names></name><name><surname>Roy</surname><given-names>RA</given-names></name><name><surname>Theobald</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>An empirical test of convergent evolution in rhodopsins</article-title><source>Molecular Biology and Evolution</source><volume>31</volume><fpage>85</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1093/molbev/mst171</pub-id><pub-id pub-id-type="pmid">24077848</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Manuel</surname><given-names>A</given-names></name><name><surname>Beaupain</surname><given-names>D</given-names></name><name><surname>Romeo</surname><given-names>PH</given-names></name><name><surname>Raich</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Molecular characterization of a novel gene family (PHTF) conserved from <italic>Drosophila</italic> to mammals</article-title><source>Genomics</source><volume>64</volume><fpage>216</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.1006/geno.1999.6079</pub-id><pub-id pub-id-type="pmid">10729229</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matsui</surname><given-names>M</given-names></name><name><surname>Iwasaki</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Graph splitting: a graph-based approach for superfamily-scale phylogenetic tree reconstruction</article-title><source>Systematic Biology</source><volume>69</volume><fpage>265</fpage><lpage>279</lpage><pub-id pub-id-type="doi">10.1093/sysbio/syz049</pub-id><pub-id pub-id-type="pmid">31364707</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Menuz</surname><given-names>K</given-names></name><name><surname>Larter</surname><given-names>NK</given-names></name><name><surname>Park</surname><given-names>J</given-names></name><name><surname>Carlson</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>An RNA-seq screen of the <italic>Drosophila</italic> antenna identifies a transporter necessary for ammonia detection</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004810</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004810</pub-id><pub-id pub-id-type="pmid">25412082</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minh</surname><given-names>BQ</given-names></name><name><surname>Schmidt</surname><given-names>HA</given-names></name><name><surname>Chernomor</surname><given-names>O</given-names></name><name><surname>Schrempf</surname><given-names>D</given-names></name><name><surname>Woodhams</surname><given-names>MD</given-names></name><name><surname>von Haeseler</surname><given-names>A</given-names></name><name><surname>Lanfear</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era</article-title><source>Molecular Biology and Evolution</source><volume>37</volume><fpage>1530</fpage><lpage>1534</lpage><pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id><pub-id pub-id-type="pmid">32011700</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mirdita</surname><given-names>M</given-names></name><name><surname>Schütze</surname><given-names>K</given-names></name><name><surname>Moriwaki</surname><given-names>Y</given-names></name><name><surname>Heo</surname><given-names>L</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Steinegger</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ColabFold: making protein folding accessible to all</article-title><source>Nature Methods</source><volume>19</volume><fpage>679</fpage><lpage>682</lpage><pub-id pub-id-type="doi">10.1038/s41592-022-01488-1</pub-id><pub-id pub-id-type="pmid">35637307</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morinaga</surname><given-names>S</given-names></name><name><surname>Nagata</surname><given-names>K</given-names></name><name><surname>Ihara</surname><given-names>S</given-names></name><name><surname>Yumita</surname><given-names>T</given-names></name><name><surname>Niimura</surname><given-names>Y</given-names></name><name><surname>Sato</surname><given-names>K</given-names></name><name><surname>Touhara</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Structural model for ligand binding and channel opening of an insect gustatory receptor</article-title><source>The Journal of Biological Chemistry</source><volume>298</volume><elocation-id>102573</elocation-id><pub-id pub-id-type="doi">10.1016/j.jbc.2022.102573</pub-id><pub-id pub-id-type="pmid">36209821</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Murzin</surname><given-names>AG</given-names></name><name><surname>Brenner</surname><given-names>SE</given-names></name><name><surname>Hubbard</surname><given-names>T</given-names></name><name><surname>Chothia</surname><given-names>C</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Scop: a structural classification of proteins database for the investigation of sequences and structures</article-title><source>Journal of Molecular Biology</source><volume>247</volume><fpage>536</fpage><lpage>540</lpage><pub-id pub-id-type="doi">10.1006/jmbi.1995.0159</pub-id><pub-id pub-id-type="pmid">7723011</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nei</surname><given-names>M</given-names></name><name><surname>Niimura</surname><given-names>Y</given-names></name><name><surname>Nozawa</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>The evolution of animal chemosensory receptor gene repertoires: roles of chance and necessity</article-title><source>Nature Reviews. Genetics</source><volume>9</volume><fpage>951</fpage><lpage>963</lpage><pub-id pub-id-type="doi">10.1038/nrg2480</pub-id><pub-id pub-id-type="pmid">19002141</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Otaki</surname><given-names>JM</given-names></name><name><surname>Yamamoto</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Length analyses of <italic>Drosophila</italic> odorant receptors</article-title><source>Journal of Theoretical Biology</source><volume>223</volume><fpage>27</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1016/s0022-5193(03)00068-7</pub-id><pub-id pub-id-type="pmid">12782114</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oyhenart</surname><given-names>J</given-names></name><name><surname>Le Goffic</surname><given-names>R</given-names></name><name><surname>Samson</surname><given-names>M</given-names></name><name><surname>Jégou</surname><given-names>B</given-names></name><name><surname>Raich</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Phtf1 is an integral membrane protein localized in an endoplasmic reticulum domain in maturing male germ cells</article-title><source>Biology of Reproduction</source><volume>68</volume><fpage>1044</fpage><lpage>1053</lpage><pub-id pub-id-type="doi">10.1095/biolreprod.102.009787</pub-id><pub-id pub-id-type="pmid">12604659</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oyhenart</surname><given-names>J</given-names></name><name><surname>Benichou</surname><given-names>S</given-names></name><name><surname>Raich</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2005">2005a</year><article-title>Putative homeodomain transcription factor 1 interacts with the feminization factor homolog FEM1B in male germ cells</article-title><source>Biology of Reproduction</source><volume>72</volume><fpage>780</fpage><lpage>787</lpage><pub-id pub-id-type="doi">10.1095/biolreprod.104.035964</pub-id><pub-id pub-id-type="pmid">15601915</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oyhenart</surname><given-names>J</given-names></name><name><surname>Dacheux</surname><given-names>JL</given-names></name><name><surname>Dacheux</surname><given-names>F</given-names></name><name><surname>Jégou</surname><given-names>B</given-names></name><name><surname>Raich</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2005">2005b</year><article-title>Expression, regulation, and immunolocalization of putative homeodomain transcription factor 1 (PHTF1) in rodent epididymis: evidence for a novel form resulting from proteolytic cleavage</article-title><source>Biology of Reproduction</source><volume>72</volume><fpage>50</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1095/biolreprod.104.029850</pub-id><pub-id pub-id-type="pmid">15342352</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paradis</surname><given-names>E</given-names></name><name><surname>Schliep</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Ape 5.0: an environment for modern phylogenetics and evolutionary analyses in R</article-title><source>Bioinformatics</source><volume>35</volume><fpage>526</fpage><lpage>528</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty633</pub-id><pub-id pub-id-type="pmid">30016406</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Price</surname><given-names>MN</given-names></name><name><surname>Dehal</surname><given-names>PS</given-names></name><name><surname>Arkin</surname><given-names>AP</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>FastTree 2 -- approximately maximum-likelihood trees for large alignments</article-title><source>PLOS ONE</source><volume>5</volume><elocation-id>e9490</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0009490</pub-id><pub-id pub-id-type="pmid">20224823</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raich</surname><given-names>N</given-names></name><name><surname>Mattei</surname><given-names>MG</given-names></name><name><surname>Romeo</surname><given-names>PH</given-names></name><name><surname>Beaupain</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>PHTF, a novel atypical homeobox gene on chromosome 1p13, is evolutionarily conserved</article-title><source>Genomics</source><volume>59</volume><fpage>108</fpage><lpage>109</lpage><pub-id pub-id-type="doi">10.1006/geno.1999.5836</pub-id><pub-id pub-id-type="pmid">10395808</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rangwala</surname><given-names>SH</given-names></name><name><surname>Kuznetsov</surname><given-names>A</given-names></name><name><surname>Ananiev</surname><given-names>V</given-names></name><name><surname>Asztalos</surname><given-names>A</given-names></name><name><surname>Borodin</surname><given-names>E</given-names></name><name><surname>Evgeniev</surname><given-names>V</given-names></name><name><surname>Joukov</surname><given-names>V</given-names></name><name><surname>Lotov</surname><given-names>V</given-names></name><name><surname>Pannu</surname><given-names>R</given-names></name><name><surname>Rudnev</surname><given-names>D</given-names></name><name><surname>Shkeda</surname><given-names>A</given-names></name><name><surname>Weitz</surname><given-names>EM</given-names></name><name><surname>Schneider</surname><given-names>VA</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Accessing NCBI data using the NCBI sequence viewer and genome data viewer (GDV)</article-title><source>Genome Research</source><volume>31</volume><fpage>159</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1101/gr.266932.120</pub-id><pub-id pub-id-type="pmid">33239395</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Revell</surname><given-names>LJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Phytools: an R package for phylogenetic comparative biology (and other things)</article-title><source>Methods in Ecology and Evolution</source><volume>3</volume><fpage>217</fpage><lpage>223</lpage><pub-id pub-id-type="doi">10.1111/j.2041-210X.2011.00169.x</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robertson</surname><given-names>HM</given-names></name><name><surname>Warr</surname><given-names>CG</given-names></name><name><surname>Carlson</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Molecular evolution of the insect chemoreceptor gene superfamily in <italic>Drosophila melanogaster</italic></article-title><source>PNAS</source><volume>100 Suppl 2</volume><fpage>14537</fpage><lpage>14542</lpage><pub-id pub-id-type="doi">10.1073/pnas.2335847100</pub-id><pub-id pub-id-type="pmid">14608037</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robertson</surname><given-names>HM</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The insect chemoreceptor superfamily is ancient in animals</article-title><source>Chemical Senses</source><volume>40</volume><fpage>609</fpage><lpage>614</lpage><pub-id pub-id-type="doi">10.1093/chemse/bjv046</pub-id><pub-id pub-id-type="pmid">26354932</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robertson</surname><given-names>HM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Molecular evolution of the major arthropod chemoreceptor gene families</article-title><source>Annual Review of Entomology</source><volume>64</volume><fpage>227</fpage><lpage>242</lpage><pub-id pub-id-type="doi">10.1146/annurev-ento-020117-043322</pub-id><pub-id pub-id-type="pmid">30312552</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ronquist</surname><given-names>F</given-names></name><name><surname>Huelsenbeck</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>MrBayes 3: bayesian phylogenetic inference under mixed models</article-title><source>Bioinformatics</source><volume>19</volume><fpage>1572</fpage><lpage>1574</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btg180</pub-id><pub-id pub-id-type="pmid">12912839</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rost</surname><given-names>B</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Twilight zone of protein sequence alignments</article-title><source>Protein Engineering</source><volume>12</volume><fpage>85</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.1093/protein/12.2.85</pub-id><pub-id pub-id-type="pmid">10195279</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rozenberg</surname><given-names>A</given-names></name><name><surname>Inoue</surname><given-names>K</given-names></name><name><surname>Kandori</surname><given-names>H</given-names></name><name><surname>Béjà</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Microbial rhodopsins: the last two decades</article-title><source>Annual Review of Microbiology</source><volume>75</volume><fpage>427</fpage><lpage>447</lpage><pub-id pub-id-type="doi">10.1146/annurev-micro-031721-020452</pub-id><pub-id pub-id-type="pmid">34343014</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Saina</surname><given-names>M</given-names></name><name><surname>Busengdal</surname><given-names>H</given-names></name><name><surname>Sinigaglia</surname><given-names>C</given-names></name><name><surname>Petrone</surname><given-names>L</given-names></name><name><surname>Oliveri</surname><given-names>P</given-names></name><name><surname>Rentzsch</surname><given-names>F</given-names></name><name><surname>Benton</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A cnidarian homologue of an insect gustatory receptor functions in developmental body patterning</article-title><source>Nature Communications</source><volume>6</volume><elocation-id>6243</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms7243</pub-id><pub-id pub-id-type="pmid">25692633</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sajko</surname><given-names>S</given-names></name><name><surname>Grishkovskaya</surname><given-names>I</given-names></name><name><surname>Kostan</surname><given-names>J</given-names></name><name><surname>Graewert</surname><given-names>M</given-names></name><name><surname>Setiawan</surname><given-names>K</given-names></name><name><surname>Trübestein</surname><given-names>L</given-names></name><name><surname>Niedermüller</surname><given-names>K</given-names></name><name><surname>Gehin</surname><given-names>C</given-names></name><name><surname>Sponga</surname><given-names>A</given-names></name><name><surname>Puchinger</surname><given-names>M</given-names></name><name><surname>Gavin</surname><given-names>A-C</given-names></name><name><surname>Leonard</surname><given-names>TA</given-names></name><name><surname>Svergun</surname><given-names>DI</given-names></name><name><surname>Smith</surname><given-names>TK</given-names></name><name><surname>Morriswood</surname><given-names>B</given-names></name><name><surname>Djinovic-Carugo</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Structures of three MORN repeat proteins and a re-evaluation of the proposed lipid-binding properties of MORN repeats</article-title><source>PLOS ONE</source><volume>15</volume><elocation-id>e0242677</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0242677</pub-id><pub-id pub-id-type="pmid">33296386</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schaeffer</surname><given-names>RD</given-names></name><name><surname>Daggett</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Protein folds and protein folding</article-title><source>Protein Engineering Design and Selection</source><volume>24</volume><fpage>11</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1093/protein/gzq096</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Scott</surname><given-names>K</given-names></name><name><surname>Brady</surname><given-names>R</given-names></name><name><surname>Cravchik</surname><given-names>A</given-names></name><name><surname>Morozov</surname><given-names>P</given-names></name><name><surname>Rzhetsky</surname><given-names>A</given-names></name><name><surname>Zuker</surname><given-names>C</given-names></name><name><surname>Axel</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>A chemosensory gene family encoding candidate gustatory and olfactory receptors in <italic>Drosophila</italic></article-title><source>Cell</source><volume>104</volume><fpage>661</fpage><lpage>673</lpage><pub-id pub-id-type="doi">10.1016/s0092-8674(01)00263-x</pub-id><pub-id pub-id-type="pmid">11257221</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shannon</surname><given-names>P</given-names></name><name><surname>Markiel</surname><given-names>A</given-names></name><name><surname>Ozier</surname><given-names>O</given-names></name><name><surname>Baliga</surname><given-names>NS</given-names></name><name><surname>Wang</surname><given-names>JT</given-names></name><name><surname>Ramage</surname><given-names>D</given-names></name><name><surname>Amin</surname><given-names>N</given-names></name><name><surname>Schwikowski</surname><given-names>B</given-names></name><name><surname>Ideker</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Cytoscape: a software environment for integrated models of biomolecular interaction networks</article-title><source>Genome Research</source><volume>13</volume><fpage>2498</fpage><lpage>2504</lpage><pub-id pub-id-type="doi">10.1101/gr.1239303</pub-id><pub-id pub-id-type="pmid">14597658</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinegger</surname><given-names>M</given-names></name><name><surname>Söding</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets</article-title><source>Nature Biotechnology</source><volume>35</volume><fpage>1026</fpage><lpage>1028</lpage><pub-id pub-id-type="doi">10.1038/nbt.3988</pub-id><pub-id pub-id-type="pmid">29035372</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tomii</surname><given-names>K</given-names></name><name><surname>Sawada</surname><given-names>Y</given-names></name><name><surname>Honda</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Convergent evolution in structural elements of proteins investigated using cross profile analysis</article-title><source>BMC Bioinformatics</source><volume>13</volume><elocation-id>11</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-13-11</pub-id><pub-id pub-id-type="pmid">22244085</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Varadi</surname><given-names>M</given-names></name><name><surname>Anyango</surname><given-names>S</given-names></name><name><surname>Deshpande</surname><given-names>M</given-names></name><name><surname>Nair</surname><given-names>S</given-names></name><name><surname>Natassia</surname><given-names>C</given-names></name><name><surname>Yordanova</surname><given-names>G</given-names></name><name><surname>Yuan</surname><given-names>D</given-names></name><name><surname>Stroe</surname><given-names>O</given-names></name><name><surname>Wood</surname><given-names>G</given-names></name><name><surname>Laydon</surname><given-names>A</given-names></name><name><surname>Žídek</surname><given-names>A</given-names></name><name><surname>Green</surname><given-names>T</given-names></name><name><surname>Tunyasuvunakool</surname><given-names>K</given-names></name><name><surname>Petersen</surname><given-names>S</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Clancy</surname><given-names>E</given-names></name><name><surname>Green</surname><given-names>R</given-names></name><name><surname>Vora</surname><given-names>A</given-names></name><name><surname>Lutfi</surname><given-names>M</given-names></name><name><surname>Figurnov</surname><given-names>M</given-names></name><name><surname>Cowie</surname><given-names>A</given-names></name><name><surname>Hobbs</surname><given-names>N</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Kleywegt</surname><given-names>G</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Hassabis</surname><given-names>D</given-names></name><name><surname>Velankar</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>AlphaFold protein structure database: massively expanding the structural coverage of protein-sequence space with high-accuracy models</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D439</fpage><lpage>D444</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1061</pub-id><pub-id pub-id-type="pmid">34791371</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vidal</surname><given-names>B</given-names></name><name><surname>Aghayeva</surname><given-names>U</given-names></name><name><surname>Sun</surname><given-names>H</given-names></name><name><surname>Wang</surname><given-names>C</given-names></name><name><surname>Glenwinkel</surname><given-names>L</given-names></name><name><surname>Bayer</surname><given-names>EA</given-names></name><name><surname>Hobert</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>An atlas of <italic>Caenorhabditis elegans</italic> chemoreceptor expression</article-title><source>PLOS Biology</source><volume>16</volume><elocation-id>e2004218</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.2004218</pub-id><pub-id pub-id-type="pmid">29293491</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>J</given-names></name><name><surname>Anishchenko</surname><given-names>I</given-names></name><name><surname>Park</surname><given-names>H</given-names></name><name><surname>Peng</surname><given-names>Z</given-names></name><name><surname>Ovchinnikov</surname><given-names>S</given-names></name><name><surname>Baker</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Improved protein structure prediction using predicted interresidue orientations</article-title><source>PNAS</source><volume>117</volume><fpage>1496</fpage><lpage>1503</lpage><pub-id pub-id-type="doi">10.1073/pnas.1914677117</pub-id><pub-id pub-id-type="pmid">31896580</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Skolnick</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Scoring function for automated assessment of protein structure template quality</article-title><source>Proteins</source><volume>57</volume><fpage>702</fpage><lpage>710</lpage><pub-id pub-id-type="doi">10.1002/prot.20264</pub-id><pub-id pub-id-type="pmid">15476259</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Skolnick</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>TM-align: a protein structure alignment algorithm based on the TM-score</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>2302</fpage><lpage>2309</lpage><pub-id pub-id-type="doi">10.1093/nar/gki524</pub-id><pub-id pub-id-type="pmid">15849316</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.85537.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Desplan</surname><given-names>Claude</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0190ak572</institution-id><institution>New York University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/2022.12.13.519744" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/2022.12.13.519744"/></front-stub><body><p>This article provides fundamental advances to our understanding of the ancestry of insect gustatory and olfactory receptors. It identifies new members of these two related ion channel families in distant species, and the strength of evidence is exceptional. This work will serve as a reference for scientists working on insect olfaction and for those working on molecular evolution</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.85537.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Desplan</surname><given-names>Claude</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0190ak572</institution-id><institution>New York University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Yan</surname><given-names>Hua</given-names></name><role>Reviewer</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02y3ad647</institution-id><institution>University of Florida</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2022.12.13.519744">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2022.12.13.519744v1">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Structural screens identify candidate human homologs of insect chemoreceptors and cryptic <italic>Drosophila</italic> gustatory receptor-like proteins&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, and the evaluation has been overseen by Claude Desplan as the Reviewing and Senior Editor. The following individual involved in the review of your submission has agreed to reveal their identity: Hua Yan (Reviewer #1).</p><p>The reviewers were very positive about the paper and have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>The three reviewers were extremely positive about the content and presentation of the paper. Please send us as soon as possible an edited version of the paper following the suggestions of the reviewers in order for the paper to be finally accepted.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>I strongly recommend publication, after the authors properly address the following concerns:</p><p>Figure 1C: &quot;Note the model contains the extracellular loop 2 (EL2) and intracellular loop 2 (IL2) regions that were not visualized in the cryo-EM structure.&quot; Could you explain why? Is this possibly a technical limitation of EM or an inaccurate prediction from AlphaFold2? If the latter, what would be the solution?</p><p>In Table 1, what if using AlphaFold2 to analyze negative controls? The same method is necessary for comparison. X-ray and EM data should also be listed to see whether the structures predicted by Alpha2 and X-ray/EM are consistent when analyzing other 7TM proteins.</p><p>Better to include one or two vertebrate species beyond mammals/fish in Table 1.</p><p>In the Discussion, please give a brief explanation of why AlphaFold2 identified PHTFs and drosophilid GRLs, while trRosetta and RaptorX did not.</p><p>The paragraph from lines 268-294 described the expression and potential biological function of PHTF, and finally stated &quot;we suggest they also act as ion channels&quot;. However, the evidence (association with E3, cell proliferation, and survival) does not suggest acting as ion channels. If no evidence, can we still say that they belong to 7TMIC? What if later on, we find they are membrane-bound proteins with other functions, but not acting as ion channels?</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>1. Figure 1a Caption: Please change to make it clear this is a reprint of data. There is a reference to the original paper, but I would suggest &quot;reprinted from&quot; or &quot;modified from&quot; so it is clear that this is not original data.</p><p>2. This is a small point but something worth considering at this stage. &quot;7TMICs&quot; is a tough acronym. I think 7TM GPCRs are usually separated by a space or a hyphen to indicate the separate meaning of the two components. Since work on this channel family is likely to take off after this paper and you will determine the nomenclature now, it is worth considering the name that will be carried with it. Maybe add a space or a hyphen so that the meaning is easy to discern from the abbreviation (ex: 7TM-IC or 7TM IC).</p><p>3. Throughout the paper it is helpful to easily be able to find and compare the trypanosome data in various panels. It would be easier to follow if there were a star or something (different color) next to Trypanosoma in 1D so that readers don't need to hunt to relate 1D and 1G, and later to the model in figure 4.</p><p>4. Figure 2F: Why are the top two violins and the bottom bar graphs? I'm sure there is a reason, but this seemed like an inconsistency.</p><p>5. The two shades of blue used in figure 4 are very difficult to separate when printed. This made it hard to interpret. I would recommend finding higher-contrast colors.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.85537.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>I strongly recommend publication, after the authors properly address the following concerns:</p><p>Figure 1C: &quot;Note the model contains the extracellular loop 2 (EL2) and intracellular loop 2 (IL2) regions that were not visualized in the cryo-EM structure.&quot; Could you explain why? Is this possibly a technical limitation of EM or an inaccurate prediction from AlphaFold2? If the latter, what would be the solution?</p></disp-quote><p>This is simply a technical limitation of the cryo-EM structure: the Butterwick Nature 2018 paper states that:</p><p>“Side-chain density was clearly resolved for most of the Orco channel and 82% of the protein could be accurately modelled, with the exception of the second extracellular loop (Val156–Ile170) and second intracellular loop (Leu244–Asn312)”.</p><p>We have expanded the text to clarify this point in the legend.</p><disp-quote content-type="editor-comment"><p>In Table 1, what if using AlphaFold2 to analyze negative controls? The same method is necessary for comparison. X-ray and EM data should also be listed to see whether the structures predicted by Alpha2 and X-ray/EM are consistent when analyzing other 7TM proteins.</p></disp-quote><p>This is a good point; we now present the structural similarity analyses (pairwise Dali and TM-align) for the AlphaFold2 models of the “negative control” proteins in Table 1. The results confirm the analyses with the experimentally-determined structures that these proteins display very low or no similarity to the Orco cryo-EM structure.</p><disp-quote content-type="editor-comment"><p>Better to include one or two vertebrate species beyond mammals/fish in Table 1.</p></disp-quote><p>We assume that the reviewer is referring to the PHTF proteins in Table 1, for which we provide structural similarity scores for the human and <italic>D. melanogaster</italic> homologs. The scores for these different species are very similar, consistent with the high primary sequence similarity of PHTFs across eukaryotes. It is for this reason that we felt that including additional vertebrate PHTFs – which are more closely related to the human proteins than <italic>D. melanogaster</italic> Phtf – in this table would be redundant (and are cognisant that this table is already rather large). We note that the Dali similarity scores of rat, mouse and zebrafish PHTFs to the Orco cryo-EM structure are available in the Supplementary Data, as part of the results of the initial screen.</p><disp-quote content-type="editor-comment"><p>In the Discussion, please give a brief explanation of why AlphaFold2 identified PHTFs and drosophilid GRLs, while trRosetta and RaptorX did not.</p></disp-quote><p>We think that there has been a minor misunderstanding: the identification of PHTFs and drosophilid Grls in this work was made possible because of the availability of AlphaFold2 protein structure database, enabling searching for proteins based upon tertiary structural similarity. To our knowledge, trRosetta and RaptorX have not been used to generate an equivalent database of searchable protein structures.</p><disp-quote content-type="editor-comment"><p>The paragraph from lines 268-294 described the expression and potential biological function of PHTF, and finally stated &quot;we suggest they also act as ion channels&quot;. However, the evidence (association with E3, cell proliferation, and survival) does not suggest acting as ion channels. If no evidence, can we still say that they belong to 7TMIC? What if later on, we find they are membrane-bound proteins with other functions, but not acting as ion channels?</p></disp-quote><p>We fully agree that some (or even many) of the 7TMICs we describe might not function as ion channels, and stress that the “7TMIC” name was principally to avoid the cumbersome “Or/Gr/Grl/DUF3537/PHTF” terminology when collectively referring to these proteins. Having said this, the properties of mammalians PHTFs mentioned by the reviewer are not incompatible with an ionotropic function: the association with the E3 ubiquitin ligase is via the cytoplasmic N-terminus that is distinct from the presumed ion channel domain (and several other known ion channels associate with E3 proteins). Moreover, it remains to be determined how directly PHTFs regulate cell proliferation/survival.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>1. Figure 1a Caption: Please change to make it clear this is a reprint of data. There is a reference to the original paper, but I would suggest &quot;reprinted from&quot; or &quot;modified from&quot; so it is clear that this is not original data.</p></disp-quote><p>We have indicated that the Orco structure we present is “derived from” PDB 6C70 (Buttewick et al., 2018), which we feel is the most appropriate phrasing, in that we used the structural coordinates of the PDB but visually present it in a new way.</p><disp-quote content-type="editor-comment"><p>2. This is a small point but something worth considering at this stage. &quot;7TMICs&quot; is a tough acronym. I think 7TM GPCRs are usually separated by a space or a hyphen to indicate the separate meaning of the two components. Since work on this channel family is likely to take off after this paper and you will determine the nomenclature now, it is worth considering the name that will be carried with it. Maybe add a space or a hyphen so that the meaning is easy to discern from the abbreviation (ex: 7TM-IC or 7TM IC).</p></disp-quote><p>We have thought quite a lot about the nomenclature in preparing this manuscript, and stress that our proposal of “7TMIC” (which we find can be pronounced “seven-tea-mick” fairly fluently!) should be considered a placeholder name that we use here to avoid the cumbersome “Or/Gr/Grl/DUF3537/PHTF” collective terminology in the text. We would welcome any future revision of this family name, particularly when functional data is obtained. Should the proposed name become the standard in the field, we felt that the reviewer’s alternative proposals (with a hyphen or space) would be liable to inconsistent application of the punctuation, compared to the compact “7TMIC”. Note that we avoided using “7TMC” to prevent confusion with various other TMC protein families, which are unrelated to the proteins studied in this work.</p><disp-quote content-type="editor-comment"><p>3. Throughout the paper it is helpful to easily be able to find and compare the trypanosome data in various panels. It would be easier to follow if there were a star or something (different color) next to Trypanosoma in 1D so that readers don't need to hunt to relate 1D and 1G, and later to the model in figure 4.</p></disp-quote><p>We were not sure why the reviewer would like to highlight the Trypanosoma proteins in particular in Figure 1D (the species names are already indicated, and the corresponding proteins are presented in bold in Figure 1G). This plot (reporting the results of the screen) is the basis for the further analysis of Trypanosome GRLs, PHTFs and insect Grls in the subsequent figures, so it seemed to us to be unnecessarily explicit to specifically point to the Trypanosoma hits. In the model in Figure 4C, we state in the figure legend “The trypanosome 7TMICs are unplaced, due to the currently unresolved taxonomy of trypanosomes (Burki et al., 2020).”; nevertheless, these proteins are labelled in Figure 4A-B to permit their easy location.</p><disp-quote content-type="editor-comment"><p>4. Figure 2F: Why are the top two violins and the bottom bar graphs? I'm sure there is a reason, but this seemed like an inconsistency.</p></disp-quote><p>The different plotting simply reflects the nature of the data available: the human <italic>PHTF1/2</italic> expression data is a compilation of many independent RNA-seq studies (available from the GTEx Portal), while the <italic>D. melanogaster Phtf</italic> expression data is the mean FPKM (± SD) that is available from the Fly Atlas 2.0; the individual values of the three biological replicates of this atlas are not available. (There are of course many other differences between these datasets, not least the range of tissue analyzed).</p><disp-quote content-type="editor-comment"><p>5. The two shades of blue used in figure 4 are very difficult to separate when printed. This made it hard to interpret. I would recommend finding higher-contrast colors.</p></disp-quote><p>We have increased the contrast between these two blues and generated a higher quality image for Figure 4A, so that the contrast is more evident.</p></body></sub-article></article>