<?xml version="1.0" ?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.3 20210610//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">elife</journal-id>
<journal-id journal-id-type="publisher-id">eLife</journal-id>
<journal-title-group>
<journal-title>eLife</journal-title>
</journal-title-group>
<issn publication-format="electronic" pub-type="epub">2050-084X</issn>
<publisher>
<publisher-name>eLife Sciences Publications, Ltd</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">93429</article-id>
<article-id pub-id-type="doi">10.7554/eLife.93429</article-id>
<article-id pub-id-type="doi" specific-use="version">10.7554/eLife.93429.1</article-id>
<article-version-alternatives>
<article-version article-version-type="publication-state">reviewed preprint</article-version>
<article-version article-version-type="preprint-version">1.2</article-version>
</article-version-alternatives>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics and Genomics</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Meta-Research: understudied genes are lost in a leaky pipeline between genome-wide assays and reporting of results</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-6058-5886</contrib-id>
<name>
<surname>Richardson</surname>
<given-names>Reese AK</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0001-5441-8101</contrib-id>
<name>
<surname>Navarro</surname>
<given-names>Heliodoro Tejedor</given-names>
</name>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-3762-789X</contrib-id>
<name>
<surname>Amaral</surname>
<given-names>Luis A Nunes</given-names>
</name>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="aff" rid="a3">3</xref>
<xref ref-type="aff" rid="a4">4</xref>
<xref ref-type="aff" rid="a5">5</xref>
<xref ref-type="corresp" rid="cor1">*</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-5540-4278</contrib-id>
<name>
<surname>Stoeger</surname>
<given-names>Thomas</given-names>
</name>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="aff" rid="a6">6</xref>
<xref ref-type="aff" rid="a7">7</xref>
<xref ref-type="corresp" rid="cor1">*</xref>
</contrib>
<aff id="a1"><label>1</label><institution>Interdisciplinary Biological Sciences, Northwestern University</institution></aff>
<aff id="a2"><label>2</label><institution>Department of Chemical and Biological Engineering, Northwestern University</institution></aff>
<aff id="a3"><label>3</label><institution>Northwestern Institute on Complex Systems, Northwestern University</institution></aff>
<aff id="a4"><label>4</label><institution>Department of Physics and Astronomy, Northwestern University</institution></aff>
<aff id="a5"><label>5</label><institution>Department of Molecular Biosciences, Northwestern University</institution></aff>
<aff id="a6"><label>6</label><institution>The Potocsnak Longevity Institute, Northwestern University</institution></aff>
<aff id="a7"><label>7</label><institution>Simpson Querrey Lung Institute for Translational Science, Northwestern University</institution></aff>
</contrib-group>
<contrib-group content-type="section">
<contrib contrib-type="editor">
<name>
<surname>Rodgers</surname>
<given-names>Peter</given-names>
</name>
<role>Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>eLife</institution>
</institution-wrap>
<city>Cambridge</city>
<country>United Kingdom</country>
</aff>
</contrib>
<contrib contrib-type="senior_editor">
<name>
<surname>Rodgers</surname>
<given-names>Peter</given-names>
</name>
<role>Senior Editor</role>
<aff>
<institution-wrap>
<institution>eLife</institution>
</institution-wrap>
<city>Cambridge</city>
<country>United Kingdom</country>
</aff>
</contrib>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>*</label>Correspondence to: L.A.N.A. (<email>amaral@northwestern.edu</email>), T.S. (<email>thomas.stoeger@northwestern.edu</email>)</corresp>
</author-notes>
<pub-date date-type="original-publication" iso-8601-date="2023-12-15">
<day>15</day>
<month>12</month>
<year>2023</year>
</pub-date>
<volume>12</volume>
<elocation-id>RP93429</elocation-id>
<history>
<date date-type="sent-for-review" iso-8601-date="2023-10-18">
<day>18</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<pub-history>
<event>
<event-desc>Preprint posted</event-desc>
<date date-type="preprint" iso-8601-date="2023-10-31">
<day>31</day>
<month>10</month>
<year>2023</year>
</date>
<self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.02.28.530483"/>
</event>
</pub-history>
<permissions>
<copyright-statement>© 2023, Richardson et al</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Richardson et al</copyright-holder>
<ali:free_to_read/>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<ali:license_ref>https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="elife-preprint-93429-v1.pdf"/>
<abstract>
<title>Abstract</title>
<p>Present-day publications on human genes primarily feature genes that already appeared in many publications prior to completion of the Human Genome Project in 2003. These patterns persist despite the subsequent adoption of high-throughput technologies, which routinely identify novel genes associated with biological processes and disease. Although several hypotheses for bias in the selection of genes as research targets have been proposed, their explanatory powers have not yet been compared. Our analysis suggests that understudied genes are systematically abandoned in favor of better-studied genes between the completion of -omics experiments and the reporting of results. Understudied genes are similarly abandoned by studies that cite these -omics experiments. Conversely, we find that publications on understudied genes may even accrue a greater number of citations. Among 45 biological and experimental factors previously proposed to affect which genes are being studied, we find that 35 are significantly associated with the choice of hit genes presented in titles and abstracts of -omics studies. To promote the investigation of understudied genes we condense our insights into a tool, <italic>find my understudied genes</italic> (FMUG), that allows scientists to engage with potential bias during the selection of hits. We demonstrate the utility of FMUG through the identification of genes that remain understudied in vertebrate aging. FMUG is developed in Flutter and is available for download at fmug.amaral.northwestern.edu as a MacOS/Windows app.</p>
</abstract>

</article-meta>
<notes>
<notes notes-type="competing-interest-statement">
<title>Competing Interest Statement</title><p>The authors have declared no competing interest.</p></notes>
<fn-group content-type="summary-of-updates">
<title>Summary of Updates:</title>
<fn fn-type="update"><p>This version has been revised for readability by allocating individual figures to analyses previously only mentioned peripherally. It further has been edited to emphasize the meta-science aspect of the contained results, and now no longer masks names of the genes associated with animal aging.</p></fn>
</fn-group>
<fn-group content-type="external-links">
<fn fn-type="dataset"><p>
<ext-link ext-link-type="uri" xlink:href="https://fmug.amaral.northwestern.edu/">https://fmug.amaral.northwestern.edu/</ext-link>
</p></fn>
</fn-group>
</notes>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Research into human genes concentrates on a subset of genes that were already frequently investigated prior to the completion of the Human Genome Project in 2003<sup><xref ref-type="bibr" rid="c1">1</xref>-<xref ref-type="bibr" rid="c5">5</xref></sup>. This concentration stems from historically acquired research patterns rather than present-day experimental possibilities<sup><xref ref-type="bibr" rid="c6">6</xref>,<xref ref-type="bibr" rid="c7">7</xref></sup>. For most human diseases, these patterns lead to little correlation between the volume of literature published on individual genes and the strength of supporting evidence from genome-wide approaches<sup><xref ref-type="bibr" rid="c8">8</xref>-<xref ref-type="bibr" rid="c13">13</xref></sup>. For instance, we found that 44% of the genes identified as promising Alzheimer’s disease targets by the U.S. National Institutes of Health (NIH) Accelerating Medicine Partnership for Alzheimer’s Disease (AMP-AD) initiative have never appeared in the title or abstract of any publication on Alzheimer’s disease<sup><xref ref-type="bibr" rid="c13">13</xref></sup>. Furthermore, when comparing gene-disease pairs, there is no correlation between the ranks of support by transcriptomics and occurrence in annotation databases<sup><xref ref-type="bibr" rid="c9">9</xref></sup>.</p>
<p>Although -omics technologies can provide insights on numerous genes across the genome at a time and thus offer the promise to counter historically acquired research patterns<sup><xref ref-type="bibr" rid="c14">14</xref>-<xref ref-type="bibr" rid="c17">17</xref></sup>, this discrepancy has persisted<sup>9,18-22</sup> even as the popularity of -omics technologies has risen<sup><xref ref-type="bibr" rid="c5">5</xref>,<xref ref-type="bibr" rid="c23">23</xref>,<xref ref-type="bibr" rid="c24">24</xref></sup>. We therefore sought to use bibliometric data to delineate where and why understudied human protein-coding genes are abandoned as research targets following -omics experiments. In the absence of any prior quantitative testing of existing hypotheses, it remains unclear whether policies to promote the exploration of a greater set of disease-related genes should focus on how experiments are conducted, how results are reported, or how these results are subsequently received by other scientists.</p>
<sec id="s1a">
<title>Data</title>
<p>We considered 450 genome-wide association studies (GWAS, from studies indexed by the NHGRI-EBI GWAS catalog<sup><xref ref-type="bibr" rid="c25">25</xref></sup>), 296 studies using affinity capture mass spectrometry (Aff-MS, indexed by BioGRID<sup><xref ref-type="bibr" rid="c26">26</xref></sup>), 148 transcriptomic studies (indexed by the EBI Gene Expression Atlas, EBI-GXA<sup><xref ref-type="bibr" rid="c27">27</xref></sup>), and 15 genome-wide screens using CRISPR (indexed by BioGRID Open Repository of CRISPR Screens, BioGRID ORCS<sup><xref ref-type="bibr" rid="c26">26</xref></sup>) (see PRISMA diagrams in <bold>Figures S1-S4</bold>). We denote genes that are found to have statistically significant changes in expression or associations with a phenotype as ‘hit’ genes.</p>
<p>As a surrogate for a given gene having been investigated closer, we consider whether it was reported in the title or abstract of a research article. We determined which genes were mentioned in the title or abstract of articles using annotations from gene2pubmed<sup><xref ref-type="bibr" rid="c28">28</xref></sup> and PubTator<sup><xref ref-type="bibr" rid="c29">29</xref></sup>. We used NIH iCite v32 for citations<sup><xref ref-type="bibr" rid="c30">30</xref></sup>. For determining which gene properties were associated with selection as research targets, we synthesized quantitative measures from a variety of authoritative sources (see <bold>Methods</bold>).</p>
</sec>
</sec>
<sec id="s2">
<title>Results</title>
<sec id="s2a">
<title>Understudied genes are abandoned at synthesis/writing stage</title>
<p>We sought to identify at which point in the scientific process understudied genes are ignored as research targets in investigations using -omics experiments (<bold><xref rid="fig1" ref-type="fig">Figure 1A</xref></bold>). To receive scholarly attention, a gene must travel through a pipeline from biological reality to experimental results to write-up of those results. These results must be extended by subsequent research by other scholars. Understudied genes do not progress all the way through the pipeline, but it is unclear where this leak primarily occurs. The first possibility is that seemingly understudied genes are, in fact, not understudied as they would rarely be identified through experiments. Prior studies have, however, shown that understudied genes are frequent hits in high-throughput experiments<sup><xref ref-type="bibr" rid="c8">8</xref>,<xref ref-type="bibr" rid="c9">9</xref>,<xref ref-type="bibr" rid="c31">31</xref></sup>, suggesting that this is not the case. The second possibility is that understudied genes are frequently found as hits in high-throughput experiments but are not investigated further by the authors. The final possibility is that subsequent studies do not continue work on understudied genes revealed by the initial study.</p>
<fig id="fig1" position="float" fig-type="figure">
<label>Figure 1:</label>
<caption><title>While understudied genes appear often as hits high-throughput -omics experiments, they are seldom highlighted by authors.</title>
<p><bold>a</bold>, Conceptual diagram depicting possible points of abandonment for understudied genes in studies using high-throughput -omics experiments. <bold>b</bold>, Bibliometric data reveals that understudied genes are frequently hits in -omics experiments but are not typically highlighted in the title/abstract of reporting articles, nor in the title/abstract or articles citing reporting articles. We considered articles reporting on genome-wide CRISPR screens (CRISPR, n=15 articles), transcriptomics (T-omics, n=148 articles), affinity capture – mass spectrometry (Aff-MS, n=296 articles), and GWAS (n=450 articles). Numbers to the right of each box plot indicate the percentile (in terms of number of articles about that gene) of all genes exceeded by the median gene in each box plot. ** denotes <italic>p</italic> &lt; 0.01 and *** denotes <italic>p</italic> &lt; 0.001 by two-sided Mann-Whitney U test, comparing genes highlighted in title/abstract to genes present in hit lists.</p></caption>
<graphic xlink:href="530483v2_fig1.tif" mimetype="image" mime-subtype="tiff"/>
</fig>
<p>Evaluating the first possibility, we found that understudied genes were frequently found as hits in high-throughput experiments. (<bold><xref rid="fig1" ref-type="fig">Figure 1B</xref>, Figure S5</bold>, and <bold>Figure S6</bold>). This demonstrates, in line with earlier studies<sup><xref ref-type="bibr" rid="c8">8</xref>-<xref ref-type="bibr" rid="c13">13</xref></sup>, that the lack of publications on some genes is not explained by underlying biological experimental evidence.</p>
<p>Evaluating the second possibility, we found that hit genes that are highlighted in the title or abstract are strongly over-represented among the 20% highest-studied genes in all biomedical literature (<bold><xref rid="fig1" ref-type="fig">Figure 1B</xref></bold>). These trends are independent of significance threshold (<bold>Figure S5</bold>) and (except for CRISPR screens) whether we considered the current scientific literature or literature published before 2003, before any of these articles had been published (<bold>Figure S6</bold>).</p>
<p>Understudied genes are least frequently elevated to the title/abstract in transcriptomics experiments and most frequently elevated to the title/abstract in CRISPR screens. GWAS studies tend to return better-studied genes as hits; the median hit gene in GWAS studies was more popular than 75% of genes. Hit genes promoted to the title/abstract in GWAS studies had a median popularity greater 85% of all protein coding genes. This may explain the prior observation that the total number of articles on individual genes partially correlates with the total number of occurrences as a hit in GWAS studies<sup><xref ref-type="bibr" rid="c32">32</xref></sup>.</p>
<p>Evaluating the final possibility, we found that the reception of -omics studies in later scientific literature either reproduced authors’ initial selection of highly studied genes or slightly mitigated it. Jointly, the above findings reinforce that understudied genes become abandoned between the completion of -omics experiments and the reporting of results, rather than being abandoned by later research.</p>
</sec>
<sec id="s2b">
<title>Subsequent reception by other scientists does not penalize studies on understudied genes</title>
<p>The abandonment of understudied genes could be driven by the valid concern of biomedical researchers that focusing on less-investigated genes will yield articles with lower impact<sup><xref ref-type="bibr" rid="c17">17</xref></sup>, as observed around the turn of the millenium<sup><xref ref-type="bibr" rid="c33">33</xref></sup>. If this were the case, preemptively avoiding understudied hits would be the rational decision for authors of -omics studies.</p>
<p>We thus decided to complement our preceding analysis by an analysis explicitly focused on citation impact. Notably, we found that the concern of publications on understudied genes receiving fewer citations does not hold for present-day research on human genes; in biomedical literature at-large, articles focusing on less-investigated genes accumulate more citations, an effect that has held consistently since 2001 (<bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold>). Important to human health, this also holds when only considering disease-related fields (<bold>Figure S7</bold>).</p>
<fig id="fig2" position="float" fig-type="figure">
<label>Figure 2:</label>
<caption><title>Articles focusing on less popular genes tend to accrue more citations.</title>
<p><bold>a</bold>, Density plot shows correlation between articles per gene before 2015 and median citations to articles published in 2015. Contours correspond to deciles in density. Solid red line shows locally weighted scatterplot smoothing (LOWESS) regression. ρ is Spearman rank correlation and p the significance values of the Spearman rank correlation as described by Kendall and Stuart<sup><xref ref-type="bibr" rid="c73">73</xref></sup>. <bold>b</bold>, Spearman correlation of previous gene popularity (i.e. number of articles) to median citations per year since 1990. Solid blue line indicates nominal Spearman correlation, shaded region indicates bootstrapped 95% confidence interval (n=1,000). Only articles with a single gene in the title/abstract are considered, excluding the 30.4% of gene-focused studies which feature more than one gene in the title/abstract.</p></caption>
<graphic xlink:href="530483v2_fig2.tif" mimetype="image" mime-subtype="tiff"/>
</fig>
<p>To rule out that these macroscopic observations stem from us having aggregated over different diseases, we separately analyzed 602 disease-related MeSH terms. We found 29 MeSH terms with a statistically significant Spearman correlation using Benjamini-Hochberg FDR &lt; 0.01 (<bold>Table S4</bold>), of which 27 showed a negative association and only 2 a positive association. This result suggests that is may actually be rational for most scientists to pursue studies focusing on understudied genes, although most scientists specializing in a disease may also not receive more citations when focusing on understudied genes.</p>
<p>Returning to our observation that understudied hits from high-throughput assays are not promoted to the title and abstract of the resulting publication, we next tested if different experimental approaches demonstrated distinct associations between gene popularity and citations (<bold>Figure S8</bold>). Among 264 technique-related MeSH terms tested, there were 20 MeSH terms with a statistically significant Spearman correlation using Benjamini-Hochberg FDR &lt; 0.01 (<bold>Table S5</bold>), of which 16 showed a negative association and only 4 a positive association. Notably, MeSH terms representing high-throughput techniques (e.g. D055106:Genome-Wide Association Study and D020869:Gene Expression Profiling) showed no significant association. This finding suggests that authors of high-throughput studies have little to gain or lose citationwise by highlighting understudied genes.</p>
<p>To summarize, our investigations detail the previously described separation between “largescale” and “small-scale” biological research<sup><xref ref-type="bibr" rid="c34">34</xref>-<xref ref-type="bibr" rid="c36">36</xref></sup>. Authors of high-throughput studies do not highlight understudied genes in the title or abstract of their publications, the sections of the publication most accessible to other scientists. While, overall, understudied genes (and high-throughput assays themselves<sup><xref ref-type="bibr" rid="c5">5</xref></sup>) correlate with increased citation impact, for high-throughput studies any potential gain in citations is either absent or too small to be significant. Thus, there may not be any incentive for authors of high-throughput studies to highlight understudied genes.</p>
</sec>
<sec id="s2c">
<title>Identification of biological and experimental factors associated with selection of highlighted genes</title>
<p>To illuminate why understudied genes are abandoned between experimental results and the write-up of results, we performed a literature review to identify factors that have been proposed to limit studies of understudied genes (<bold>Table S1</bold>). These factors range from evolutionary factors (e.g., whether a gene only has homologs in primates), to chemical factors (e.g., gene length or hydrophobicity of protein product), to historical factors (e.g., whether a gene’s sequence has previously been patented) to materialistic factors affecting experimental design (e.g., whether designed antibodies are robust for immunohistochemistry).</p>
<p>As any of these factors could plausibly affect gene selection within individual domains of biomedical research, we returned to the -omics data described above (<bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold>) and measured how much these factors align with the selective highlighting of hit genes in the title or abstract of GWAS, Aff-MS, transcriptomics, and CRISPR studies.</p>
<p>We identified 45 factors that relate to genes and found 35 (14 out of 23 binary factors and 21 out of 22 continuous factors) associated with selection in at least one assay type at p &lt; 0.001 (<bold><xref rid="fig3" ref-type="fig">Figure 3</xref>, Table S2</bold>, and <bold>Table S3</bold>). Across the four assay types, the most informative binary factor describes whether there is a plasmid available for a gene in the AddGene plasmid catalog. This might reflect that many different research groups produce reagents surrounding the genes that they actively study. The most informative continuous factor is the number of research articles about a gene, supporting the conclusion that gene popularity drives whether it is highlighted or not (<bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold>).</p>
<fig id="fig3" position="float" fig-type="figure">
<label>Figure 3:</label>
<caption><title>We evaluated which gene-related factors are associated with elevation to the title/abstract of an article featuring a high-throughput experiment.</title>
<p><bold>a)</bold> Association between factors with binary (True/False) identities and highlighting hits in title/abstract of reporting articles. Values represent the odds ratio between hits in the collected articles and hits mentioned in the title or abstract of collected articles (e.g. hits with a compound known to affect gene activity are 5.114 times as likely to be mentioned in the title/abstract in an article using transcriptomics). Collected articles are described in <bold>Figure 1B</bold> and <bold>Figures S5</bold> and <bold>S6</bold>. 95% confidence interval of odds ratio is shown in parentheses. * = p &lt; 0.05, ** = p &lt; 0.01, and *** = p &lt; 0.001 by two-sided Fisher exact test. Results are shown numerically in <bold>Table S2</bold>. For consistency between studies, hits were restricted to protein-coding genes. Thus, status as a protein-coding gene could not be tested. †No genes without a defined HUGO symbol were found as hits in GWAS or transcriptomics studies. <bold>b)</bold> Association with factors with continuous identities and highlighting hits in title/abstract of reporting articles. Values represent F, the common-language effect size (equivalent to AUROC, where ∼0.5 indicates little effect, &gt;0.5 indicates positive effect and &lt;0.5 indicates negative effect) of being mentioned in the titles/abstracts of the collected articles described in <bold>Figure 1B</bold> and <bold>Figures S5</bold> and <bold>S6</bold>. * = p &lt; 0.05, ** = p &lt; 0.01, and *** = p &lt; 0.001 by two-sided Mann-Whitney U test. Results are shown numerically in <bold>Table S3</bold>.</p></caption>
<graphic xlink:href="530483v2_fig3.tif" mimetype="image" mime-subtype="tiff"/>
</fig>
<p>To better understand how all 45 factors are related, we performed a cluster analysis of the collected factors (<bold><xref rid="fig4" ref-type="fig">Figure 4</xref>, Figure S9</bold>, and <bold>Figure S10</bold>). This clustering reflects the fact that many suggested factors influencing the abandonment of understudied genes are not independent. For instance, we find that the number of articles about a gene is heavily correlated with the number of annotations for that gene in all surveyed databases. In another case, gene length is heavily correlated with the number of GWAS annotations for a gene, as described before in terms of transcript length and single-nucleotide polymoprhisms<sup><xref ref-type="bibr" rid="c37">37</xref></sup>.</p>
<fig id="fig4" position="float" fig-type="figure">
<label>Figure 4:</label>
<caption><title>Clustermap showing collected factors across all human protein-coding genes.</title>
<p>Factors are shown along the x axis, with genes along the y axis. Eight factors, representing the default factors we selected for FMUG, are shown (all factors are shown in <bold>Figure S7</bold> and <bold>Figure S8</bold>). Binary factors are coded to 0 (purple) and 1 (white), while continuous factors are ranked from 0 to 1 with ties resolved to minimum rank. Clustering was performed with Ward’s method for hierarchical clustering<sup><xref ref-type="bibr" rid="c74">74</xref></sup>.</p></caption>
<graphic xlink:href="530483v2_fig4.tif" mimetype="image" mime-subtype="tiff"/>
</fig>
</sec>
<sec id="s2d">
<title>Study limitations</title>
<p>Our study has several limitations. First, all analysis is subject to annotation errors in the various databases we employ. While these should be rare and not affect our overall findings, they may affect users who are interested in genes with discordant annotations. Second, we focus only on human genes. Different patterns of selection may exist for research on genes in other organisms. Third, our literature review also identified further factors that we could not test more directly because of absent access to fitting data. These are: experts’ tendency to deepen their expertise<sup><xref ref-type="bibr" rid="c3">3</xref></sup>, a perceived lack of accuracy of -omics studies<sup><xref ref-type="bibr" rid="c17">17</xref>,<xref ref-type="bibr" rid="c38">38</xref></sup>, -omics serving research purposes beyond target gene identification<sup><xref ref-type="bibr" rid="c22">22</xref></sup>, the absence of good protocols for mass spectrometry<sup><xref ref-type="bibr" rid="c39">39</xref></sup>, the electronic distribution and reading of research articles<sup><xref ref-type="bibr" rid="c40">40</xref></sup>, rates of reproducibility<sup><xref ref-type="bibr" rid="c17">17</xref></sup>, career prospects of investigators<sup><xref ref-type="bibr" rid="c7">7</xref>,<xref ref-type="bibr" rid="c41">41</xref></sup>, and the human tendency to fall back to simplifying heuristics when making decisions under conditions with uncertainty<sup><xref ref-type="bibr" rid="c42">42</xref></sup>. Fourth, we cannot resolve further which specific step between the conduct of an experiment and the writing of a research article leads to the abandonment of hit genes. Finally, we interpret the results of high-throughput experiments based on their representation in the NHGRI-EBI GWAS, BioGRID, EBI-GXA and BioGRID ORCS databases. The authors of the original studies may have processed their data differently, obtaining different results.</p>
<p>As our present analysis is correlative, it also is tempting to propose controlled trials where published manuscripts on high-throughput studies randomly report hit genes in the abstract even if not investigated further by the authors.</p>
</sec>
</sec>
<sec id="s3">
<title>Discussion</title>
<p>Efforts to address the gaps in detailed knowledge about most genes have crystallized as initiatives promoting the investigation of understudied sets of genes<sup>16,43-47</sup>, an approach to gene scholarship recently termed ‘unknomics’<sup><xref ref-type="bibr" rid="c48">48</xref></sup>. We believe that enabling scientists to consciously engage with bias in research target selection will enable more biomedical researchers to participate in unknomics, to the potential benefit of their own research impact and towards the advancement of our collective understanding of the entire human genome.</p>
<p>To achieve this goal, we combined all the above insights to create a tool we denoted <italic>find my understudied genes</italic> (FMUG). Our literature review revealed several tools and resources aiming to promote research of understudied genes by publicizing understudied genes<sup><xref ref-type="bibr" rid="c49">49</xref>-<xref ref-type="bibr" rid="c56">56</xref></sup> or by providing information about hit genes<sup>7,57-61</sup>. However, we noted the absence of tools enabling scientists to actively engage with factors that align with gene selection. Although such factors are largely correlated when considering all genes (<xref rid="fig4" ref-type="fig">Figure 4</xref>, Figure S9, and Figure S10), some cluster together and the influence of specific factors could vary across laboratories. For instance, scientists could vary in their ability to perform proteomics, or ability to explore orthologous genes in <italic>C. elegans</italic>, or ability to leverage human population data, or perform standardized mouse assays.</p>
<p>Our tool makes selection bias explicit, while acknowledging that different laboratories vary in their techniques and capabilities for follow-up research. Rather than telling scientists about the existence of biases, FMUG aims to prompt scientists to make bias-aware informed decisions to identify and potentially tackle important gaps in knowledge that they are well-suited to address. For this reason, we believe that FMUG will not be of value only to scientists engaging in high-throughput studies, but also by scientists wishing to mine existing datasets for hit genes that they would be well-positioned to investigate further.</p>
<p>FMUG takes a list of genes from the user (ostensibly a hit list from a high-throughput -omics experiment) and provides the kind of information that will allow a user to select genes for further study.</p>
<p>Users can employ filters that reflect the factors identified in our literature review and supported by our analysis. The default information provided to users consists of factors that are representative of the identified clusters (<bold><xref rid="fig4" ref-type="fig">Figure 4</xref></bold>) and strongly associated with gene selection in high-throughput experiments (<bold><xref rid="fig3" ref-type="fig">Figure 3</xref>, Table S2</bold>, and <bold>Table S3</bold>). In extended options, users can select any factor that demonstrated a significant association with the selection of genes. For instance, a user may need to decide whether loss-of-function intolerant genes should be considered for further research or not, or whether there should be robust evidence that a gene is protein-coding. Some of these filters are context aware. For instance, a user may select genes that have already been studied in the general biomedical literature but not yet within the literature of their disease of interest.</p>
<p>To provide real-time feedback, users are, in parallel, presented the number of articles about genes in their initial input list and the number of articles about genes that passed their filters. Users can then export their filtered list of genes. In the interest of researcher privacy, FMUG keeps all information local to the user’s machine. Usage of FMUG is illustrated in <bold><xref rid="fig5" ref-type="fig">Figure 5A</xref></bold> and demonstrated in <bold>Movie S1</bold>. FMUG is developed in Flutter and is available for download at fmug.amaral.northwestern.edu as a MacOS/Windows app. For the development of custom software and analytical code, we provide the data underlying FMUG at github.com/amarallab/fmug_analysis.</p>
<fig id="fig5" position="float" fig-type="figure">
<label>Figure 5:</label>
<caption><title>We created FMUG to help researchers identify understudied genes among their genes of interest and characterize their tractability for future research.</title>
<p><bold>a</bold>, Diagram describing use of FMUG. <bold>b</bold>, An early prototype of FMUG led us to the hypothesis that transcript length negatively correlates with up-regulation during aging. First, we identified genes that strongly associate with age-dependent transcriptional change across multiple cohorts. We then performed a literature review for each of these genes to identify the most direct way the genes (or evolutionally closely related genes or functionally closely related partner proteins) had been studied in aging. 64% had been functionally investigated in aging, 15% shown to change a measure of gene expression, 3% functionally investigated in a biological domain close to aging (such as senescence), and 5% shown to change a measure of gene expression in a biological domain close to aging. For genes reported by others to change expression with age, we identified tissues in which transcripts of the genes change during aging. We computed ‘feasibility scores’ scientific strategies (GEM: G: strong genetic support, E: and experimental potential, M: homolog in invertebrate model organism) as described by Stoeger et al.<sup><xref ref-type="bibr" rid="c7">7</xref></sup> and total number of publications in MEDLINE. Splicing factor, proline- and glutamine-rich (<italic>Sfpq</italic>) had previously been demonstrated by Takeuchi et al. to be required for the transcriptional elongation of long genes<sup><xref ref-type="bibr" rid="c62">62</xref></sup>. When performing a data-driven analysis of factors that could possibly explain age-dependent changes of the entire transcriptome, we thus included gene and transcript lengths, and subsequently found them to be more informative than transcription factors or microRNAs<sup><xref ref-type="bibr" rid="c63">63</xref></sup>.</p></caption>
<graphic xlink:href="530483v2_fig5.tif" mimetype="image" mime-subtype="tiff"/>
</fig>
<p>To determine the practical usefulness of FMUG to scientists we used an early prototype of FMUG to identify understudied genes associated with aging. One of these genes was Splicing factor, proline- and glutamine-rich (<italic>Sfpq</italic>), which had not yet been investigated toward its role in biological aging. We found <italic>Sfpq</italic> to be transcriptionally downregulated during murine aging. Others had shown <italic>Sfpq</italic> to be required for the transcriptional elongation of long genes<sup><xref ref-type="bibr" rid="c62">62</xref></sup>. This led us to hypothesize that during vertebrate aging, the transcripts of long genes become downregulated in most tissues <bold>(<xref rid="fig5" ref-type="fig">Figure 5B</xref>)</bold>. We found this hypothesis to be supported through a multi-species analysis which we published in December 2022 in Nature Aging<sup><xref ref-type="bibr" rid="c63">63</xref></sup>, with another group publishing so in January 2023 in Nature Genetics<sup><xref ref-type="bibr" rid="c64">64</xref></sup>, and a third group in iScience in March 2023<sup><xref ref-type="bibr" rid="c65">65</xref></sup>.</p>
</sec>
<sec id="s4">
<title>Materials and Methods</title>
<sec id="s4a">
<title>Genes information</title>
<p>Homo sapiens gene information was downloaded from NCBI Gene on Aug 16, 2022 [<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/All_Data.gene_info.gz">ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/All_Data.gene_info.gz</ext-link>]. Only genes with an unambiguous mapping of Entrez ID to Ensembl ID were used (n = 36,035). Number of gene synonyms, protein-coding status, and official gene symbol were derived from this dataset. A gene symbol was considered undefined if the gene’s entry for HGNC gene symbol was “-”.</p>
</sec>
<sec id="s4b">
<title>Genes in title/abstract of primary research articles</title>
<p>Homo sapiens gene information was downloaded from NCBI Gene on Aug 16, 2022 [<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/Mammalia/Homo_sapiens.gene_info.gz">ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/Mammalia/Homo_sapiens.gene_info.gz</ext-link>]. gene2pubmed was download from NCBI Gene on Aug 16, 2022 [<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nlm.nih.gov/gene/DATA/gene2pubmed.gz">ftp.ncbi.nlm.nih.gov/gene/DATA/gene2pubmed.gz</ext-link>]<sup><xref ref-type="bibr" rid="c28">28</xref></sup>. PubTator gene annotations were downloaded from NIH-NLM on July 12, 2022 [<ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/pub/lu/PubTatorCentral/">https://ftp.ncbi.nlm.nih.gov/pub/lu/PubTatorCentral/</ext-link>] <sup><xref ref-type="bibr" rid="c28">28</xref>,<xref ref-type="bibr" rid="c66">66</xref></sup>. PubMed was downloaded on Dec 17, 2021 [<ext-link ext-link-type="uri" xlink:href="https://ftp.ncbi.nlm.nih.gov/pubmed/baseline/">https://ftp.ncbi.nlm.nih.gov/pubmed/baseline/</ext-link>].</p>
<p>Only using PMIDs annotated as primary research articles, a human gene was considered as mentioned in the title/abstract of the publication if gene was annotated as being in the title/abstract by PubTator and the article appeared in gene2pubmed.</p>
</sec>
<sec id="s4c">
<title>CRISPR articles</title>
<p>BioGRID ORCS<sup><xref ref-type="bibr" rid="c26">26</xref></sup> v1.1.6 was downloaded on April 25, 2022 [<ext-link ext-link-type="uri" xlink:href="https://downloads.thebiogrid.org/BioGRID-ORCS/Release-Archive/BIOGRID-ORCS-1.1.6/">https://downloads.thebiogrid.org/BioGRID-ORCS/Release-Archive/BIOGRID-ORCS-1.1.6/</ext-link>]. Any genome-wide CRISPR knockout screens in human with an associated PubMed ID in which hit genes were mentioned in the title or abstract was considered (n = 15). 9,268 unique genes were found as hits. Of these, 18 (0.19%) were elevated to titles/abstracts in the reporting articles and 19 (0.21%) were elevated to titles/abstracts in citing articles. A full list of PubMed IDs is available in Supplementary File 2.</p>
</sec>
<sec id="s4d">
<title>Transcriptomics articles</title>
<p>EBI-GXA<sup><xref ref-type="bibr" rid="c27">27</xref></sup> release 36 was downloaded on Sep 15, 2020 [<ext-link ext-link-type="uri" xlink:href="https://web.archive.org/web/20201022184159/">https://web.archive.org/web/20201022184159/</ext-link> <ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/gxa/download">https://www.ebi.ac.uk/gxa/download</ext-link>]. This is the most recent release of EBI-GXA available as a bulk download. Any transcriptomics comparisons with an associated PubMed ID in which hit genes were mentioned in the title or abstract was considered (n= 148). Analysis was restricted to protein-coding genes (some screens featured non-protein-coding genes, but this was not common to all analyses). DE was called at Benjamini-Hochberg FDR <italic>q</italic> &lt; 0.05. 18,295 unique genes were found as hits. Of these, 161 (0.88%) were elevated to titles/abstracts in the reporting articles and 692 (3.78%) were elevated to titles/abstracts in citing articles. A full list of PubMed IDs is available in Supplementary File 2.</p>
</sec>
<sec id="s4e">
<title>Affinity capture – mass spectrometry articles</title>
<p>BioGRID<sup><xref ref-type="bibr" rid="c26">26</xref></sup> v3.5.186 was downloaded on April 25, 2022</p>
<p>[<ext-link ext-link-type="uri" xlink:href="https://downloads.thebiogrid.org/BioGRID/Release-Archive/BIOGRID-3.5.186/">https://downloads.thebiogrid.org/BioGRID/Release-Archive/BIOGRID-3.5.186/</ext-link>]. Any interactions involving a human gene as the prey protein with an experimental evidence code of ‘Affinity Capture-MS’ labeled as ‘High-Throughput’ that had an associated PubMed ID in which hit genes were mentioned in the title or abstract was considered (n= 296). Prey proteins in these interactions were considered hits. 7,919 unique genes were found as hits. Of these, 311 (3.93%) were elevated to titles/abstracts in reporting articles and 407 (5.14%) were elevated to titles/abstracts in citing articles. A full list of PubMed IDs is available in Supplementary File 2.</p>
</sec>
<sec id="s4f">
<title>GWAS articles</title>
<p>The NHGRI-EBI GWAS catalog<sup><xref ref-type="bibr" rid="c25">25</xref></sup> (associations and studies) was download on Aug 17, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/gwas/docs/file-downloads">https://www.ebi.ac.uk/gwas/docs/file-downloads</ext-link>]. Any GWAS screens with an associated PubMed ID in which hit genes were mentioned in the title or abstract was considered (n= 450). Only SNPs occurring within a gene were considered hits. 1,043 unique genes were found as hits. Of these, 413 (39.6%) were elevated to titles/abstracts in reporting articles and 319 (30.6%) were elevated to titles/abstracts in citing articles. A full list of PubMed IDs is available in Supplementary File 2.</p>
</sec>
<sec id="s4g">
<title>Citing articles</title>
<p>NIH iCite v32 was downloaded on Aug 25, 2022<sup><xref ref-type="bibr" rid="c30">30</xref></sup> [<ext-link ext-link-type="uri" xlink:href="https://nih.figshare.com/collections/iCite_Database_Snapshots_NIH_Open_Citation_Collection_/4586573/32">https://nih.figshare.com/collections/iCite_Database_Snapshots_NIH_Open_Citation_Collection_/4586573/32</ext-link>].</p>
</sec>
<sec id="s4h">
<title>Functional annotations</title>
<p>Mapping of genes to Gene Ontology / Protein Interaction Database / WikiPathways / Reactome / Kyoto Encyclopedia of Genes and Genomes / Human Phenotype Ontology / BioCarta categories was derived from MSigDB v7.5 Entrez ID .gmt files, downloaded on Apr 12, 2022 [<ext-link ext-link-type="uri" xlink:href="http://www.gsea-msigdb.org/gsea/downloads_archive.jsp">http://www.gsea-msigdb.org/gsea/downloads_archive.jsp</ext-link>].</p>
</sec>
<sec id="s4i">
<title>Between-species homology</title>
<p>Homologene Build 68 was used to determine interspecies homology [<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nih.gov/pub/HomoloGene/build68/">ftp.ncbi.nih.gov/pub/HomoloGene/build68/</ext-link>]. Human = taxid:9606, mouse = taxid:10090, rat = taxid:10116, c. elegans = taxid:6239, d. melanogaster = taxid:7227, yeast = taxid:559292, zebrafish = taxid:7955.</p>
</sec>
<sec id="s4j">
<title>Primate specificity</title>
<p>Human genes were considered primate-specific if the only other members of their homology group belonged to primate genomes. Primate taxonomy ids were downloaded from NCBI Taxonomy on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/taxonomy/?term=txid9443[Subtree]">https://www.ncbi.nlm.nih.gov/taxonomy/?term=txid9443[Subtree]</ext-link>].</p>
</sec>
<sec id="s4k">
<title>Number of publications in model organisms</title>
<p>Gene information was downloaded from NCBI Gene on Aug 16, 2022 [<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/All_Data.gene_info.gz">ftp.ncbi.nlm.nih.gov/gene/DATA/GENE_INFO/All_Data.gene_info.gz</ext-link>].</p>
<p>Only using PMIDs annotated as primary research articles, genes was considered as mentioned in the title/abstract of the publication if gene was annotated as being in the title/abstract by PubTator and the article appeared in gene2pubmed.</p>
<p>Genes in model organisms were mapped to human genes and the number of articles on those mapping to human genes were counted. If a model organism’s gene had homology to human but no associated publications, the number of publications was resolved to zero. Otherwise, counts were listed as NA.</p>
</sec>
<sec id="s4l">
<title>Mouse phenotype hits</title>
<p>International Mouse Phenotyping Consortium data release 17.0 was downloaded on Aug 18, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.mousephenotype.org/data/release">https://www.mousephenotype.org/data/release</ext-link>]. Mouse genes were matched to human genes with Homologene.</p>
</sec>
<sec id="s4m">
<title>Gene Expression Atlas (EBI-GXA)</title>
<p>EBI-GXA release 36 was downloaded on Sep 15, 2020 [<ext-link ext-link-type="uri" xlink:href="https://web.archive.org/web/20201022184159/">https://web.archive.org/web/20201022184159/</ext-link> <ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/gxa/download">https://www.ebi.ac.uk/gxa/download</ext-link>]. This is the most recent release of EBI-GXA available as a bulk download. For probability of DE, only RNA-seq comparisons were considered and DE was called at Benjamini-Hochberg q &lt; 0.05.</p>
</sec>
<sec id="s4n">
<title>Global RNA expression</title>
<p>RNA consensus tissue gene data from HPA release 21.1 was downloaded on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.proteinatlas.org/about/download">https://www.proteinatlas.org/about/download</ext-link>]. Global RNA expression was estimated by taking the median expression (nTPM) across tissues for each gene and the proportion of tissues with detectable (≥1 nTPM) expression for each gene.</p>
</sec>
<sec id="s4o">
<title>Expression in HeLa cells</title>
<p>RNA cell line gene data from HPA release 21.1 was downloaded on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.proteinatlas.org/about/download">https://www.proteinatlas.org/about/download</ext-link>]. Expression is in nTPM.</p>
</sec>
<sec id="s4p">
<title>Previous patent activity</title>
<p>Genes with patent activity were defined from Table S1 of Rosenfeld and Mason, 2013<sup><xref ref-type="bibr" rid="c67">67</xref></sup>. Genes were mapped with their HGNC symbol. This analysis aligned sequences in patents to the human genome to estimate patent coverage of human coding sequences. Although this does not necessarily reflect whether the mapped genes were claimed directly by the patent holder, as noted by others<sup><xref ref-type="bibr" rid="c68">68</xref></sup>, this analysis remains the most comprehensive available for determining patent coverage of the human genome.</p>
</sec>
<sec id="s4q">
<title>Druggability</title>
<p>Druggable genes were identified from Table S1 of Finan et al., 2017<sup><xref ref-type="bibr" rid="c69">69</xref></sup>. Genes were mapped with their Ensembl identifier.</p>
</sec>
<sec id="s4r">
<title>Gene length</title>
<p>GenBank was downloaded in spring 2017 (genome version GRCh38.p10). Gene length is defined here as the span of the longest transcript on the chromosome. This aligns with the model of gene length used in Stoeger et al., 2018<sup><xref ref-type="bibr" rid="c7">7</xref></sup>.</p>
</sec>
<sec id="s4s">
<title>Solubility</title>
<p>SwissProt protein sequences and mapping tables to Entrez GeneIDs were downloaded from Uniprot in spring 2017. Protein GRAVY score (ignoring Pyrrolysine and Selenocysteine) was estimated with BioPython<sup><xref ref-type="bibr" rid="c70">70</xref></sup>.</p>
</sec>
<sec id="s4t">
<title>Loss of function intolerance</title>
<p>Data was obtained from Karczewski et al.<sup><xref ref-type="bibr" rid="c71">71</xref></sup>. pLI scores &gt; 0.9 on main transcripts, as flagged by authors, were considered as highly loss-of-function intolerant as described by Lek et al.<sup><xref ref-type="bibr" rid="c72">72</xref></sup>.</p>
</sec>
<sec id="s4u">
<title>Number of GWAS hits</title>
<p>EBI GWAS catalog<sup><xref ref-type="bibr" rid="c25">25</xref></sup> (associations and studies) was download on Aug 17, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/gwas/docs/file-downloads">https://www.ebi.ac.uk/gwas/docs/file-downloads</ext-link>]. Loci were mapped to the nearest gene.</p>
</sec>
<sec id="s4v">
<title>Status as understudied protein</title>
<p>The Illuminating the Druggable Genome understudied protein list was downloaded on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://github.com/druggablegenome/IDGTargets/blob/master/IDG_TargetList_CurrentVersion.json">https://github.com/druggablegenome/IDGTargets/blob/master/IDG_TargetList_CurrentVersion.json</ext-link>].</p>
</sec>
<sec id="s4w">
<title>Human Protein Atlas</title>
<p>HPA release 21.1 was downloaded on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.proteinatlas.org/search">https://www.proteinatlas.org/search</ext-link>]. Evidence for a protein’s existence, as determined by NeXtProt, HPA, or UniProt was resolved as True if the respective evidence entry was annotated as “Evidence at protein level”. Status as a membrane protein was determined by whether the ‘Protein class’ column contained the string ‘membrane protein’. Antibodies were considered available for each protein if the protein’s entry in the ‘Antibody’ column was not null.</p>
</sec>
<sec id="s4x">
<title>Availability of plasmids</title>
<p>The AddGene plasmid catalog was downloaded on Aug 12, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.addgene.org/browse/gene/gene-list-data/?_=1666368044314">https://www.addgene.org/browse/gene/gene-list-data/?_=1666368044314</ext-link>].</p>
</sec>
<sec id="s4y">
<title>Availability of compounds</title>
<p>The catalog of gene targets was downloaded from ChEMBL on Sep 20, 2022 [<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/chembl/g/#browse/targets">https://www.ebi.ac.uk/chembl/g/#browse/targets</ext-link>]. UniProt IDs were converted to Entrez IDs to identify which human genes were affected by any compound.</p>
</sec>
<sec id="s4z">
<title>Mendelian inheritance</title>
<p>Autosomal dominant [<ext-link ext-link-type="uri" xlink:href="https://hpo.jax.org/app/browse/term/HP:0000006">https://hpo.jax.org/app/browse/term/HP:0000006</ext-link>] and autosomal recessive [<ext-link ext-link-type="uri" xlink:href="https://hpo.jax.org/app/browse/term/HP:0000007">https://hpo.jax.org/app/browse/term/HP:0000007</ext-link>] inherited disease-gene associations were downloaded from the Human Phenotype Ontology on Sep 20, 2022. Genes were considered to have evidence of Mendelian inheritance if they appeared in these lists of associations.</p>
</sec>
<sec id="s4aa">
<title>Code</title>
<p>Code for analysis is available at github.com/amarallab/fmug_analysis. Code for FMUG is available at github.com/amarallab/fmug.</p>
</sec>
</sec>
<sec id="s5">
<title>Data Availability</title>
<p>All underlying data for figures are available at github.com/amarallab/fmug_analysis.</p>
</sec>
<sec id="d1e1196" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="d1e1279">
<label>Supplemental Figures, Tables, Methods</label>
<media xlink:href="supplements/530483_file02.pdf"/>
</supplementary-material>
<supplementary-material id="d1e1286">
<label>Supplemental Movie.</label>
<media xlink:href="supplements/530483_file03.mp4"/>
</supplementary-material>
<supplementary-material id="d1e1293">
<label>Installer of FMUG for windows</label>
<media xlink:href="supplements/530483_file04.zip"/>
</supplementary-material>
</sec>
</body>
<back>
<ack>
<title>Acknowledgements</title>
<p>We thank Xiaojing Sui for testing FMUG and Northwestern Information Technology for technical assistance. RAKR was supported in part by the National Institutes of Health Training Grant (T32GM008449) through Northwestern University’s Biotechnology Training Program. RAKR also acknowledges support from the Dr. John N. Nicholson fellowship from Northwestern University and Moderna Inc., “Identifying bias and improving reproducibility in RNA-seq computational pipelines”. LANA was supported by NSF 1956338, NIH U19AI135964 and Simons Foundation DMS-1764421. TS was supported by NIH K99AG068544. We thank Alexander Misharin, Richard Morimoto, and Scott Budinger for feedback on an early prototype of FMUG which we used as part of our shared research into the biology of aging.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="c1"><label>1</label><mixed-citation publication-type="journal"><string-name><surname>Hoffmann</surname>, <given-names>R.</given-names></string-name> &amp; <string-name><surname>Valencia</surname>, <given-names>A.</given-names></string-name> <article-title>Life cycles of successful genes</article-title>. <source>Trends Genet</source>. <volume>19</volume>, <fpage>79</fpage>–<lpage>81</lpage>, doi:<pub-id pub-id-type="doi">10.1016/S0168-9525(02)00014-8</pub-id> (<year>2003</year>).</mixed-citation></ref>
<ref id="c2"><label>2</label><mixed-citation publication-type="journal"><string-name><surname>Su</surname>, <given-names>A. I.</given-names></string-name> &amp; <string-name><surname>Hogenesch</surname>, <given-names>J. B.</given-names></string-name> <article-title>Power-law-like distributions in biomedical publications and research funding</article-title>. <source>Genome Biol</source>. <volume>8</volume>, <fpage>404</fpage>, doi:<pub-id pub-id-type="doi">10.1186/gb-2007-8-4-404</pub-id> (<year>2007</year>).</mixed-citation></ref>
<ref id="c3"><label>3</label><mixed-citation publication-type="journal"><string-name><surname>Edwards</surname>, <given-names>A. M.</given-names></string-name> <etal>et al.</etal> <article-title>Too many roads not taken</article-title>. <source>Nature</source> <volume>470</volume>, <fpage>163</fpage>–<lpage>165</lpage>, doi:<pub-id pub-id-type="doi">10.1038/470163a</pub-id> (<year>2011</year>).</mixed-citation></ref>
<ref id="c4"><label>4</label><mixed-citation publication-type="journal"><string-name><surname>Gillis</surname>, <given-names>J.</given-names></string-name> &amp; <string-name><surname>Pavlidis</surname>, <given-names>P.</given-names></string-name> <article-title>Assessing identity, redundancy and confounds in Gene Ontology annotations over time</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>476</fpage>–<lpage>482</lpage>, doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/bts727</pub-id> (<year>2013</year>).</mixed-citation></ref>
<ref id="c5"><label>5</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name> &amp; <string-name><surname>Nunes Amaral</surname>, <given-names>L.</given-names></string-name> A. <article-title>The characteristics of early-stage research into human genes are substantially different from subsequent research</article-title>. <source>PLoS Biol</source> <volume>20</volume>, <fpage>e3001520</fpage>, doi:<pub-id pub-id-type="doi">10.1371/journal.pbio.3001520</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c6"><label>6</label><mixed-citation publication-type="journal"><string-name><surname>Grueneberg</surname>, <given-names>D. A.</given-names></string-name> <etal>et al.</etal> <article-title>Kinase requirements in human cells: I. Comparing kinase requirements across various cell types</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A</source>. <volume>105</volume>, <fpage>16472</fpage>–<lpage>16477</lpage>, doi:<pub-id pub-id-type="doi">10.1073/pnas.0808019105</pub-id> (<year>2008</year>).</mixed-citation></ref>
<ref id="c7"><label>7</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Gerlach</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Morimoto</surname>, <given-names>R. I.</given-names></string-name> &amp; <string-name><surname>Nunes Amaral</surname>, <given-names>L.</given-names></string-name> A. <article-title>Large-scale investigation of the reasons why potentially important genes are ignored</article-title>. <source>PLoS Biol</source> <volume>16</volume>, <fpage>e2006643</fpage>, doi:<pub-id pub-id-type="doi">10.1371/journal.pbio.2006643</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c8"><label>8</label><mixed-citation publication-type="journal"><string-name><surname>Riba</surname>, <given-names>M.</given-names></string-name> <etal>et al.</etal> <article-title>Revealing the acute asthma ignorome: characterization and validation of uninvestigated gene networks</article-title>. <source>Sci Rep</source> <volume>6</volume>, <fpage>24647</fpage>, doi:<pub-id pub-id-type="doi">10.1038/srep24647</pub-id> (<year>2016</year>).</mixed-citation></ref>
<ref id="c9"><label>9</label><mixed-citation publication-type="journal"><string-name><surname>Haynes</surname>, <given-names>W. A.</given-names></string-name>, <string-name><surname>Tomczak</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Khatri</surname>, <given-names>P.</given-names></string-name> <article-title>Gene annotation bias impedes biomedical research</article-title>. <source>Sci Rep</source> <volume>8</volume>, <fpage>1362</fpage>, doi:<pub-id pub-id-type="doi">10.1038/s41598-018-19333-x</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c10"><label>10</label><mixed-citation publication-type="journal"><string-name><surname>Border</surname>, <given-names>R.</given-names></string-name> <etal>et al.</etal> <article-title>No support for historical candidate gene or candidate gene-by-interaction hypotheses for major depression across multiple large samples</article-title>. <source>American Journal of Psychiatry</source> <volume>176</volume>, <fpage>376</fpage>–<lpage>387</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c11"><label>11</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name> &amp; <string-name><surname>Nunes Amaral</surname>, <given-names>L.</given-names></string-name> A. <article-title>COVID-19 research risks ignoring important host genes due to pre-established research patterns</article-title>. <source>Elife</source> <volume>9</volume>, doi:<pub-id pub-id-type="doi">10.7554/eLife.61981</pub-id> (<year>2020</year>).</mixed-citation></ref>
<ref id="c12"><label>12</label><mixed-citation publication-type="journal"><string-name><surname>Zhang</surname>, <given-names>D.</given-names></string-name> <etal>et al.</etal> <article-title>Incomplete annotation has a disproportionate impact on our understanding of Mendelian and complex neurogenetic disorders</article-title>. <source>Science Advances</source> <volume>6</volume>, <fpage>eaay8299</fpage> (<year>2020</year>).</mixed-citation></ref>
<ref id="c13"><label>13</label><mixed-citation publication-type="journal"><string-name><surname>Byrne</surname>, <given-names>J. A.</given-names></string-name> <etal>et al.</etal> <article-title>Protection of the human gene research literature from contract cheating organizations known as research paper mills</article-title>. <source>Nucleic Acids Research</source> <volume>50</volume>, <fpage>12058</fpage>–<lpage>12070</lpage>, doi:<pub-id pub-id-type="doi">10.1093/nar/gkac1139</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c14"><label>14</label><mixed-citation publication-type="journal"><string-name><surname>Collins</surname>, <given-names>F. S.</given-names></string-name>, <string-name><surname>Green</surname>, <given-names>E. D.</given-names></string-name>, <string-name><surname>Guttmacher</surname>, <given-names>A. E.</given-names></string-name>, <string-name><surname>Guyer</surname>, <given-names>M. S.</given-names></string-name> &amp; <string-name><surname>Institute</surname>, <given-names>U. S. N. H. G. R.</given-names></string-name> <article-title>A vision for the future of genomics research</article-title>. <source>Nature</source> <volume>422</volume>, <fpage>835</fpage>–<lpage>847</lpage>, doi:<pub-id pub-id-type="doi">10.1038/nature01626</pub-id> (<year>2003</year>).</mixed-citation></ref>
<ref id="c15"><label>15</label><mixed-citation publication-type="journal"><string-name><surname>Shendure</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Findlay</surname>, <given-names>G. M.</given-names></string-name> &amp; <string-name><surname>Snyder</surname>, <given-names>M. W.</given-names></string-name> <article-title>Genomic Medicine-Progress, Pitfalls, and Promise</article-title>. <source>Cell</source> <volume>177</volume>, <fpage>45</fpage>–<lpage>57</lpage>, doi:<pub-id pub-id-type="doi">10.1016/j.cell.2019.02.003</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c16"><label>16</label><mixed-citation publication-type="journal"><string-name><surname>Lloyd</surname>, <given-names>K. C. K.</given-names></string-name> <etal>et al.</etal> <article-title>The Deep Genome Project</article-title>. <source>Genome Biol</source> <volume>21</volume>, <fpage>18</fpage>, doi:<pub-id pub-id-type="doi">10.1186/s13059-020-1931-9</pub-id> (<year>2020</year>).</mixed-citation></ref>
<ref id="c17"><label>17</label><mixed-citation publication-type="journal"><string-name><surname>Kustatscher</surname>, <given-names>G.</given-names></string-name> <etal>et al.</etal> <article-title>Understudied proteins: opportunities and challenges for functional proteomics</article-title>. <source>Nat Methods</source> <volume>19</volume>, <fpage>774</fpage>–<lpage>779</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41592-022-01454-x</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c18"><label>18</label><mixed-citation publication-type="journal"><string-name><surname>Rodriguez-Esteban</surname>, <given-names>R.</given-names></string-name> &amp; <string-name><surname>Jiang</surname>, <given-names>X.</given-names></string-name> <article-title>Differential gene expression in disease: a comparison between high-throughput studies and the literature</article-title>. <source>BMC Med Genomics</source> <volume>10</volume>, <fpage>59</fpage>, doi:<pub-id pub-id-type="doi">10.1186/s12920-017-0293-y</pub-id> (<year>2017</year>).</mixed-citation></ref>
<ref id="c19"><label>19</label><mixed-citation publication-type="journal"><string-name><surname>Oprea</surname>, <given-names>T. I.</given-names></string-name> <etal>et al.</etal> <article-title>Unexplored therapeutic opportunities in the human genome</article-title>. <source>Nat Rev Drug Discov</source> <volume>17</volume>, <fpage>317</fpage>–<lpage>332</lpage>, doi:<pub-id pub-id-type="doi">10.1038/nrd.2018.14</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c20"><label>20</label><mixed-citation publication-type="journal"><string-name><surname>Sinha</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Eisenhaber</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Jensen</surname>, <given-names>L. J.</given-names></string-name>, <string-name><surname>Kalbuaji</surname>, <given-names>B.</given-names></string-name> &amp; <string-name><surname>Eisenhaber</surname>, <given-names>F.</given-names></string-name> <article-title>Darkness in the Human Gene and Protein Function Space: Widely Modest or Absent Illumination by the Life Science Literature and the Trend for Fewer Protein Function Discoveries Since 2000</article-title>. <source>Proteomics</source> <volume>18</volume>, <fpage>e1800093</fpage>, doi:<pub-id pub-id-type="doi">10.1002/pmic.201800093</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c21"><label>21</label><mixed-citation publication-type="journal"><string-name><surname>Wood</surname>, <given-names>V.</given-names></string-name> <etal>et al.</etal> <article-title>Hidden in plain sight: what remains to be discovered in the eukaryotic proteome?</article-title> <source>Open Biol</source> <volume>9</volume>, <fpage>180241</fpage>, doi:<pub-id pub-id-type="doi">10.1098/rsob.180241</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c22"><label>22</label><mixed-citation publication-type="other"><string-name><surname>Donohue</surname>, <given-names>C.</given-names></string-name> &amp; <string-name><surname>Love</surname>, <given-names>A.</given-names></string-name> <article-title>Perspectives on the Human Genome Project and genomics</article-title>. <source>Minnesota Studies in the Philosophy of Science</source> <volume>23</volume> (in press).</mixed-citation></ref>
<ref id="c23"><label>23</label><mixed-citation publication-type="journal"><string-name><surname>Pena-Castillo</surname>, <given-names>L.</given-names></string-name> &amp; <string-name><surname>Hughes</surname>, <given-names>T. R.</given-names></string-name> <article-title>Why are there still over 1000 uncharacterized yeast genes?</article-title> <source>Genetics</source> <volume>176</volume>, <fpage>7</fpage>–<lpage>14</lpage>, doi:<pub-id pub-id-type="doi">10.1534/genetics.107.074468</pub-id> (<year>2007</year>).</mixed-citation></ref>
<ref id="c24"><label>24</label><mixed-citation publication-type="journal"><string-name><surname>Ellens</surname>, <given-names>K. W.</given-names></string-name> <etal>et al.</etal> <article-title>Confronting the catalytic dark matter encoded by sequenced genomes</article-title>. <source>Nucleic Acids Res</source> <volume>45</volume>, <fpage>11495</fpage>–<lpage>11514</lpage>, doi:<pub-id pub-id-type="doi">10.1093/nar/gkx937</pub-id> (<year>2017</year>).</mixed-citation></ref>
<ref id="c25"><label>25</label><mixed-citation publication-type="journal"><string-name><surname>Buniello</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title>The NHGRI-EBI GWAS Catalog of published genome-wide association studies, targeted arrays and summary statistics 2019</article-title>. <source>Nucleic Acids Res</source> <volume>47</volume>, <fpage>D1005</fpage>–<lpage>D1012</lpage>, doi:<pub-id pub-id-type="doi">10.1093/nar/gky1120</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c26"><label>26</label><mixed-citation publication-type="journal"><string-name><surname>Oughtred</surname>, <given-names>R.</given-names></string-name> <etal>et al.</etal> <article-title>The BioGRID database: A comprehensive biomedical resource of curated protein, genetic, and chemical interactions</article-title>. <source>Protein Science</source> <volume>30</volume>, <fpage>187</fpage>–<lpage>200</lpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c27"><label>27</label><mixed-citation publication-type="journal"><string-name><surname>Papatheodorou</surname>, <given-names>I.</given-names></string-name> <etal>et al.</etal> <article-title>Expression Atlas: gene and protein expression across multiple studies and organisms</article-title>. <source>Nucleic acids research</source> <volume>46</volume>, <fpage>D246</fpage>–<lpage>D251</lpage> (<year>2018</year>).</mixed-citation></ref>
<ref id="c28"><label>28</label><mixed-citation publication-type="journal"><string-name><surname>Maglott</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Ostell</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Pruitt</surname>, <given-names>K. D.</given-names></string-name> &amp; <string-name><surname>Tatusova</surname>, <given-names>T.</given-names></string-name> <article-title>Entrez Gene: gene-centered information at NCBI</article-title>. <source>Nucleic acids research</source> <volume>35</volume>, <fpage>D26</fpage>–<lpage>D31</lpage> (<year>2007</year>).</mixed-citation></ref>
<ref id="c29"><label>29</label><mixed-citation publication-type="journal"><string-name><surname>Wei</surname>, <given-names>C.-H.</given-names></string-name>, <string-name><surname>Allot</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Leaman</surname>, <given-names>R.</given-names></string-name> &amp; <string-name><surname>Lu</surname>, <given-names>Z.</given-names></string-name> <article-title>PubTator central: automated concept annotation for biomedical full text articles</article-title>. <source>Nucleic acids research</source> <volume>47</volume>, <fpage>W587</fpage>–<lpage>W593</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c30"><label>30</label><mixed-citation publication-type="other"><string-name><surname>Hutchins</surname>, <given-names>B. I.</given-names></string-name>, <source>Santangelo, George. iCite</source>, &lt;<pub-id pub-id-type="doi">10.35092/yhjc.c.4586573</pub-id> &gt; (<year>2019</year>).</mixed-citation></ref>
<ref id="c31"><label>31</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name> &amp; <string-name><surname>Nunes Amaral</surname>, <given-names>L.</given-names></string-name> A. <article-title>COVID-19 research risks ignoring important host genes due to pre-established research patterns</article-title>. <source>Elife</source> <volume>9</volume>, <fpage>e61981</fpage> (<year>2020</year>).</mixed-citation></ref>
<ref id="c32"><label>32</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Gerlach</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Morimoto</surname>, <given-names>R. I.</given-names></string-name> &amp; <string-name><surname>Nunes Amaral</surname>, <given-names>L.</given-names></string-name> A. <article-title>Large-scale investigation of the reasons why potentially important genes are ignored</article-title>. <source>PLoS biology</source> <volume>16</volume>, <fpage>e2006643</fpage> (<year>2018</year>).</mixed-citation></ref>
<ref id="c33"><label>33</label><mixed-citation publication-type="journal"><string-name><surname>Pfeiffer</surname>, <given-names>T.</given-names></string-name> &amp; <string-name><surname>Hoffmann</surname>, <given-names>R.</given-names></string-name> <article-title>Temporal patterns of genes in scientific publications</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A</source>. <volume>104</volume>, <fpage>12052</fpage>–<lpage>12056</lpage>, doi:<pub-id pub-id-type="doi">10.1073/pnas.0701315104</pub-id> (<year>2007</year>).</mixed-citation></ref>
<ref id="c34"><label>34</label><mixed-citation publication-type="book"><string-name><surname>Knorr Cetina</surname>, <given-names>K.</given-names></string-name> <source>Epistemic Cultures</source>. (<publisher-name>Harvard University Press</publisher-name>, <year>1999</year>).</mixed-citation></ref>
<ref id="c35"><label>35</label><mixed-citation publication-type="book"><string-name><surname>Alberts</surname>, <given-names>B. M.</given-names></string-name> <chapter-title>Limits to growth: In biology, small science is good science</chapter-title>. <source>Cell</source> (<publisher-loc>Cambridge</publisher-loc>) <volume>41</volume>, <fpage>337</fpage>–<lpage>338</lpage> (<year>1985</year>).</mixed-citation></ref>
<ref id="c36"><label>36</label><mixed-citation publication-type="book"><string-name><surname>Richardson</surname>, <given-names>S. S. a. S.</given-names></string-name>, <source>Hallam</source>. (<publisher-name>Duke University Press</publisher-name>, <year>2015</year>).</mixed-citation></ref>
<ref id="c37"><label>37</label><mixed-citation publication-type="journal"><string-name><surname>Lopes</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Altab</surname>, <given-names>G.</given-names></string-name>, <string-name><surname>Raina</surname>, <given-names>P.</given-names></string-name> &amp; <string-name><surname>De Magalhães</surname>, <given-names>J. P.</given-names></string-name> <article-title>Gene size matters: an analysis of gene length in the human genome</article-title>. <source>Frontiers in Genetics</source> <volume>12</volume>, <fpage>559998</fpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c38"><label>38</label><mixed-citation publication-type="journal"><string-name><surname>Brown</surname>, <given-names>L. A.</given-names></string-name> &amp; <string-name><surname>Peirson</surname>, <given-names>S. N.</given-names></string-name> <article-title>Improving Reproducibility and Candidate Selection in Transcriptomics Using Meta-analysis</article-title>. <source>Journal of Experimental Neuroscience</source> <volume>12</volume>, 1179069518756296, doi:<pub-id pub-id-type="doi">10.1177/1179069518756296</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c39"><label>39</label><mixed-citation publication-type="journal"><string-name><surname>Cesar-Razquin</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title>A Call for Systematic Research on Solute Carriers</article-title>. <source>Cell</source> <volume>162</volume>, <fpage>478</fpage>–<lpage>487</lpage>, doi:<pub-id pub-id-type="doi">10.1016/j.cell.2015.07.022</pub-id> (<year>2015</year>).</mixed-citation></ref>
<ref id="c40"><label>40</label><mixed-citation publication-type="journal"><string-name><surname>Evans</surname>, <given-names>J. A.</given-names></string-name> <article-title>Electronic publication and the narrowing of science and scholarship</article-title>. <source>Science</source> <volume>321</volume>, <fpage>395</fpage>–<lpage>399</lpage>, doi:<pub-id pub-id-type="doi">10.1126/science.1150473</pub-id> (<year>2008</year>).</mixed-citation></ref>
<ref id="c41"><label>41</label><mixed-citation publication-type="journal"><string-name><surname>Alberts</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Kirschner</surname>, <given-names>M. W.</given-names></string-name>, <string-name><surname>Tilghman</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>Varmus</surname>, <given-names>H.</given-names></string-name> <article-title>Rescuing US biomedical research from its systemic flaws</article-title>. <source>Proc Natl Acad Sci U S A</source> <volume>111</volume>, <fpage>5773</fpage>–<lpage>5777</lpage>, doi:<pub-id pub-id-type="doi">10.1073/pnas.1404402111</pub-id> (<year>2014</year>).</mixed-citation></ref>
<ref id="c42"><label>42</label><mixed-citation publication-type="book"><string-name><surname>Gilovich</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Griffin</surname>, <given-names>D. W.</given-names></string-name> &amp; <string-name><surname>Kahneman</surname>, <given-names>D.</given-names></string-name> <source>Heuristics and biases : the psychology of intuitive judgment</source>. (<publisher-name>Cambridge University Press</publisher-name>, <year>2002</year>).</mixed-citation></ref>
<ref id="c43"><label>43</label><mixed-citation publication-type="journal"><string-name><surname>Gerlt</surname>, <given-names>J. A.</given-names></string-name> <etal>et al.</etal> <article-title>The Enzyme Function Initiative</article-title>. <source>Biochemistry</source> <volume>50</volume>, <fpage>9950</fpage>–<lpage>9962</lpage>, doi:<pub-id pub-id-type="doi">10.1021/bi201312u</pub-id> (<year>2011</year>).</mixed-citation></ref>
<ref id="c44"><label>44</label><mixed-citation publication-type="journal"><string-name><surname>Carter</surname>, <given-names>A. J.</given-names></string-name> <etal>et al.</etal> <article-title>Target 2035: probing the human proteome</article-title>. <source>Drug Discov Today</source> <volume>24</volume>, <fpage>2111</fpage>–<lpage>2115</lpage>, doi:<pub-id pub-id-type="doi">10.1016/j.drudis.2019.06.020</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c45"><label>45</label><mixed-citation publication-type="journal"><string-name><surname>Rodgers</surname>, <given-names>G.</given-names></string-name> <etal>et al.</etal> <article-title>Glimmers in illuminating the druggable genome</article-title>. <source>Nat Rev Drug Discov</source> <volume>17</volume>, <fpage>301</fpage>–<lpage>302</lpage>, doi:<pub-id pub-id-type="doi">10.1038/nrd.2017.252</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c46"><label>46</label><mixed-citation publication-type="journal"><string-name><surname>Kustatscher</surname>, <given-names>G.</given-names></string-name> <etal>et al.</etal> <article-title>An open invitation to the Understudied Proteins Initiative</article-title>. <source>Nat Biotechnol</source> <volume>40</volume>, <fpage>815</fpage>–<lpage>817</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41587-022-01316-z</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c47"><label>47</label><mixed-citation publication-type="web"><source>EUbOPEN</source>, &lt;<ext-link ext-link-type="uri" xlink:href="https://www.eubopen.org/">https://www.eubopen.org/</ext-link>&gt; (</mixed-citation></ref>
<ref id="c48"><label>48</label><mixed-citation publication-type="journal"><string-name><surname>Rocha</surname>, <given-names>J. J.</given-names></string-name> <etal>et al.</etal> <article-title>Functional unknomics: Systematic screening of conserved genes of unknown function</article-title>. <source>PLoS biology</source> <volume>21</volume>, <fpage>e3002222</fpage> (<year>2023</year>).</mixed-citation></ref>
<ref id="c49"><label>49</label><mixed-citation publication-type="journal"><string-name><surname>Duek</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Gateau</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Bairoch</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Lane</surname>, <given-names>L.</given-names></string-name> <article-title>Exploring the Uncharacterized Human Proteome Using neXtProt</article-title>. <source>J Proteome Res</source> <volume>17</volume>, <fpage>4211</fpage>–<lpage>4226</lpage>, doi:<pub-id pub-id-type="doi">10.1021/acs.jproteome.8b00537</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c50"><label>50</label><mixed-citation publication-type="journal"><string-name><surname>Crow</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Lim</surname>, <given-names>N.</given-names></string-name>, <string-name><surname>Ballouz</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Pavlidis</surname>, <given-names>P.</given-names></string-name> &amp; <string-name><surname>Gillis</surname>, <given-names>J.</given-names></string-name> <article-title>Predictability of human differential gene expression</article-title>. <source>Proc Natl Acad Sci U S A</source> <volume>116</volume>, <fpage>6491</fpage>–<lpage>6500</lpage>, doi:<pub-id pub-id-type="doi">10.1073/pnas.1802973116</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c51"><label>51</label><mixed-citation publication-type="journal"><string-name><surname>Perdigao</surname>, <given-names>N.</given-names></string-name> &amp; <string-name><surname>Rosa</surname>, <given-names>A.</given-names></string-name> <article-title>Dark Proteome Database: Studies on Dark Proteins</article-title>. <source>High Throughput</source> <volume>8</volume>, doi:<pub-id pub-id-type="doi">10.3390/ht8020008</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c52"><label>52</label><mixed-citation publication-type="journal"><string-name><surname>Essegian</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Khurana</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Stathias</surname>, <given-names>V.</given-names></string-name> &amp; <string-name><surname>Schurer</surname>, <given-names>S. C.</given-names></string-name> <article-title>The Clinical Kinase Index: A Method to Prioritize Understudied Kinases as Drug Targets for the Treatment of Cancer</article-title>. <source>Cell Rep Med</source> <volume>1</volume>, <fpage>100128</fpage>, doi:<pub-id pub-id-type="doi">10.1016/j.xcrm.2020.100128</pub-id> (<year>2020</year>).</mixed-citation></ref>
<ref id="c53"><label>53</label><mixed-citation publication-type="journal"><string-name><surname>Sheils</surname>, <given-names>T. K.</given-names></string-name> <etal>et al.</etal> <article-title>TCRD and Pharos 2021: mining the human proteome for disease biology</article-title>. <source>Nucleic Acids Res</source> <volume>49</volume>, <fpage>D1334</fpage>–<lpage>D1346</lpage>, doi:<pub-id pub-id-type="doi">10.1093/nar/gkaa993</pub-id> (<year>2021</year>).</mixed-citation></ref>
<ref id="c54"><label>54</label><mixed-citation publication-type="other"><string-name><surname>Rocha</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title>Functional unknomics: closing the knowledge gap to accelerate biomedical research</article-title>. <source>bioRxiv</source> (<year>2022</year>).</mixed-citation></ref>
<ref id="c55"><label>55</label><mixed-citation publication-type="journal"><string-name><surname>Higgins</surname>, <given-names>D. P.</given-names></string-name>, <string-name><surname>Weisman</surname>, <given-names>C. M.</given-names></string-name>, <string-name><surname>Lui</surname>, <given-names>D. S.</given-names></string-name>, <string-name><surname>D’Agostino</surname>, <given-names>F. A.</given-names></string-name> &amp; <string-name><surname>Walker</surname>, <given-names>A. K.</given-names></string-name> <article-title>Defining characteristics and conservation of poorly annotated genes in Caenorhabditis elegans using WormCat 2.0</article-title>. <source>Genetics</source> <volume>221</volume>, doi:<pub-id pub-id-type="doi">10.1093/genetics/iyac085</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c56"><label>56</label><mixed-citation publication-type="journal"><string-name><surname>Wainberg</surname>, <given-names>M.</given-names></string-name> <etal>et al.</etal> <article-title>A genome-wide atlas of co-essential modules assigns function to uncharacterized genes</article-title>. <source>Nat Genet</source> <volume>53</volume>, <fpage>638</fpage>–<lpage>649</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41588-021-00840-z</pub-id> (<year>2021</year>).</mixed-citation></ref>
<ref id="c57"><label>57</label><mixed-citation publication-type="journal"><string-name><surname>Rebhan</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Chalifa-Caspi</surname>, <given-names>V.</given-names></string-name>, <string-name><surname>Prilusky</surname>, <given-names>J.</given-names></string-name> &amp; <string-name><surname>Lancet</surname>, <given-names>D.</given-names></string-name> <article-title>GeneCards: a novel functional genomics compendium with automated data mining and query reformulation support</article-title>. <source>Bioinformatics</source> <volume>14</volume>, <fpage>656</fpage>–<lpage>664</lpage>, doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/14.8.656</pub-id> (<year>1998</year>).</mixed-citation></ref>
<ref id="c58"><label>58</label><mixed-citation publication-type="journal"><string-name><surname>Tan</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title>ADAGE signature analysis: differential expression analysis with data-defined gene sets</article-title>. <source>Bmc Bioinformatics</source> <volume>18</volume>, <fpage>512</fpage>, doi:<pub-id pub-id-type="doi">10.1186/s12859-017-1905-4</pub-id> (<year>2017</year>).</mixed-citation></ref>
<ref id="c59"><label>59</label><mixed-citation publication-type="journal"><string-name><surname>Kustatscher</surname>, <given-names>G.</given-names></string-name> <etal>et al.</etal> <article-title>Co-regulation map of the human proteome enables identification of protein functions</article-title>. <source>Nat Biotechnol</source> <volume>37</volume>, <fpage>1361</fpage>–<lpage>1371</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41587-019-0298-5</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c60"><label>60</label><mixed-citation publication-type="journal"><string-name><surname>Wu</surname>, <given-names>T.</given-names></string-name> <etal>et al.</etal> <article-title>clusterProfiler 4.0: A universal enrichment tool for interpreting omics data</article-title>. <source>Innovation (Camb)</source> <volume>2</volume>, <fpage>100141</fpage>, doi:<pub-id pub-id-type="doi">10.1016/j.xinn.2021.100141</pub-id> (<year>2021</year>).</mixed-citation></ref>
<ref id="c61"><label>61</label><mixed-citation publication-type="journal"><string-name><surname>Jiang</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title>Systematic illumination of druggable genes in cancer genomes</article-title>. <source>Cell Rep</source> <volume>38</volume>, <fpage>110400</fpage>, doi:<pub-id pub-id-type="doi">10.1016/j.celrep.2022.110400</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c62"><label>62</label><mixed-citation publication-type="journal"><string-name><surname>Takeuchi</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title>Loss of Sfpq Causes Long-Gene Transcriptopathy in the Brain</article-title>. <source>Cell Rep</source> <volume>23</volume>, <fpage>1326</fpage>–<lpage>1341</lpage>, doi:<pub-id pub-id-type="doi">10.1016/j.celrep.2018.03.141</pub-id> (<year>2018</year>).</mixed-citation></ref>
<ref id="c63"><label>63</label><mixed-citation publication-type="journal"><string-name><surname>Stoeger</surname>, <given-names>T.</given-names></string-name> <etal>et al.</etal> <article-title>Aging is associated with a systemic length-associated transcriptome imbalance</article-title>. <source>Nature Aging</source> <volume>2</volume>, <fpage>1191</fpage>–<lpage>1206</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s43587-022-00317-6</pub-id> (<year>2022</year>).</mixed-citation></ref>
<ref id="c64"><label>64</label><mixed-citation publication-type="journal"><string-name><surname>Gyenis</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title>Genome-wide RNA polymerase stalling shapes the transcriptome during aging</article-title>. <source>Nat Genet</source> <volume>55</volume>, <fpage>268</fpage>–<lpage>279</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41588-022-01279-6</pub-id> (<year>2023</year>).</mixed-citation></ref>
<ref id="c65"><label>65</label><mixed-citation publication-type="other"><string-name><surname>Ibañez-Solé</surname>, <given-names>O.</given-names></string-name>, <string-name><surname>Barrio</surname>, <given-names>I.</given-names></string-name> &amp; <string-name><surname>Izeta</surname>, <given-names>A.</given-names></string-name> <article-title>Age or lifestyle-induced accumulation of genotoxicity is associated with a length-dependent decrease in gene expression</article-title>. <source>iScience</source>, <fpage>106368</fpage>, doi:<pub-id pub-id-type="doi">10.1016/j.isci.2023.106368</pub-id> (<year>2023</year>).</mixed-citation></ref>
<ref id="c66"><label>66</label><mixed-citation publication-type="journal"><string-name><surname>Wei</surname>, <given-names>C. H.</given-names></string-name>, <string-name><surname>Allot</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Leaman</surname>, <given-names>R.</given-names></string-name> &amp; <string-name><surname>Lu</surname>, <given-names>Z.</given-names></string-name> <article-title>PubTator central: automated concept annotation for biomedical full text articles</article-title>. <source>Nucleic Acids Res</source> <volume>47</volume>, <fpage>W587</fpage>–<lpage>W593</lpage>, doi:<pub-id pub-id-type="doi">10.1093/nar/gkz389</pub-id> (<year>2019</year>).</mixed-citation></ref>
<ref id="c67"><label>67</label><mixed-citation publication-type="journal"><string-name><surname>Rosenfeld</surname>, <given-names>J. A.</given-names></string-name> &amp; <string-name><surname>Mason</surname>, <given-names>C. E.</given-names></string-name> <article-title>Pervasive sequence patents cover the entire human genome</article-title>. <source>Genome Med</source>. <volume>5</volume>, <fpage>27</fpage>, doi:<pub-id pub-id-type="doi">10.1186/gm431</pub-id> (<year>2013</year>).</mixed-citation></ref>
<ref id="c68"><label>68</label><mixed-citation publication-type="journal"><string-name><surname>Tu</surname>, <given-names>S.</given-names></string-name> <etal>et al.</etal> <article-title>Response to ‘pervasive sequence patents cover the entire human genome’</article-title>. <source>Genome medicine</source> <volume>6</volume>, <fpage>1</fpage>–<lpage>3</lpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c69"><label>69</label><mixed-citation publication-type="journal"><string-name><surname>Finan</surname>, <given-names>C.</given-names></string-name> <etal>et al.</etal> <article-title>The druggable genome and support for target identification and validation in drug development</article-title>. <source>Science translational medicine</source> <volume>9</volume>, <fpage>eaag1166</fpage> (<year>2017</year>).</mixed-citation></ref>
<ref id="c70"><label>70</label><mixed-citation publication-type="journal"><string-name><surname>Cock</surname>, <given-names>P. J. A.</given-names></string-name> <etal>et al.</etal> <article-title>Biopython: freely available Python tools for computational molecular biology and bioinformatics</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>1422</fpage>–<lpage>1423</lpage>, doi:<pub-id pub-id-type="doi">10.1093/bioinformatics/btp163</pub-id> (<year>2009</year>).</mixed-citation></ref>
<ref id="c71"><label>71</label><mixed-citation publication-type="journal"><string-name><surname>Karczewski</surname>, <given-names>K. J.</given-names></string-name> <etal>et al.</etal> <article-title>The mutational constraint spectrum quantified from variation in 141,456 humans</article-title>. <source>Nature</source> <volume>581</volume>, <fpage>434</fpage>–<lpage>443</lpage>, doi:<pub-id pub-id-type="doi">10.1038/s41586-020-2308-7</pub-id> (<year>2020</year>).</mixed-citation></ref>
<ref id="c72"><label>72</label><mixed-citation publication-type="journal"><string-name><surname>Lek</surname>, <given-names>M.</given-names></string-name> <etal>et al.</etal> <article-title>Analysis of protein-coding genetic variation in 60,706 humans</article-title>. <source>Nature</source> <volume>536</volume>, <fpage>285</fpage>–<lpage>291</lpage>, doi:<pub-id pub-id-type="doi">10.1038/nature19057</pub-id> (<year>2016</year>).</mixed-citation></ref>
<ref id="c73"><label>73</label><mixed-citation publication-type="journal"><string-name><surname>Kendall</surname>, <given-names>M. G.</given-names></string-name> &amp; <string-name><surname>Stuart</surname>, <given-names>A.</given-names></string-name> <source>Inference and Relationship The Advanced Theory of Statistics</source> <volume>2</volume> (<year>1973</year>).</mixed-citation></ref>
<ref id="c74"><label>74</label><mixed-citation publication-type="journal"><string-name><surname>Ward Jr</surname>, <given-names>J.</given-names></string-name> H. <article-title>Hierarchical grouping to optimize an objective function</article-title>. <source>Journal of the American statistical association</source> <volume>58</volume>, <fpage>236</fpage>–<lpage>244</lpage> (<year>1963</year>).</mixed-citation></ref>
</ref-list>
</back>
<sub-article id="sa0" article-type="editor-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.93429.1.sa3</article-id>
<title-group>
<article-title>eLife Assessment</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Rodgers</surname>
<given-names>Peter</given-names>
</name>
<role specific-use="editor">Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>eLife</institution>
</institution-wrap>
<city>Cambridge</city>
<country>United Kingdom</country>
</aff>
</contrib>
</contrib-group>
<kwd-group kwd-group-type="evidence-strength">
<kwd>Solid</kwd>
</kwd-group>
<kwd-group kwd-group-type="claim-importance">
<kwd>Valuable</kwd>
</kwd-group>
</front-stub>
<body>
<p>This study investigated the factors related to understudied genes in biomedical research. It showed that understudied genes are largely abandoned at the writing stage, and it identified a number of biological and experimental factors that influence which genes are selected for investigation. The study is a <bold>valuable</bold> contribution to this branch of meta-research, and while the evidence in support of the findings is <bold>solid</bold>, the interpretation and presentation of the results (especially the figures) needs to be improved.</p>
</body>
</sub-article>
<sub-article id="sa1" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.93429.1.sa2</article-id>
<title-group>
<article-title>Reviewer #1 (Public Review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary and strengths</p>
<p>
The authors tried to address why only a subset of genes are highlighted in many publications. Is it because these highlighted genes are more important than others? Or is it because there are non-genetic reasons? This is a critical question because in the effort to discover new genes for drug targets and clinical benefit, we need to expand a pool of genes for deep analyses. So I appreciate the authors' efforts in this study, as it is timely and important. They also provided a framework called FMUG (short for Find My Understudied Gene) to evaluate genes for a number of features for subsequent analyses.</p>
<p>Weaknesses</p>
<p>
Many of the figures are hard to comprehend, and the figure legends do not sufficiently explain them.</p>
<p>
# For example, what was plotted in Fig 1b? The number of articles increased from results -&gt; write-ups -&gt; follow-ups in all four categories with different degrees. But it does not seem to match what the authors meant to deliver.</p>
<p>
# Fig 4 is also confusing. It appears that the genes were clustered by many features that the authors developed. But does it have any relationship with genes being under- or over-studied?</p>
</body>
</sub-article>
<sub-article id="sa2" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.93429.1.sa1</article-id>
<title-group>
<article-title>Reviewer #2 (Public Review)</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary and strengths</p>
<p>
In this manuscript the authors analyse the trajectory of understudied genes (UGs) from experiment to publication and study the reasons for why UGs remain underrepresented in the scientific literature. They show that UGs are not underrepresented in experimental datasets, but in the titles and abstracts of the manuscripts reporting experimental data as well as subsequent studies referring to those large-scale studies. They also develop an app that allows researchers to find UGs and their annotation state. Overall, this is a timely article that makes an important contribution to the field. It could help to boost the future investigation of understudied genes, a fundamental challenge in the life sciences. It is concise and overall well-written, and I very much enjoyed reading it. However, there are a few points that I think the authors should address.</p>
<p>Weaknesses</p>
<p>
The authors conclude that many UGs &quot;are lost&quot; from genome-wide assay at the manuscript writing stage. If I understand correctly, this is based on gene names not being reported in the title or abstract of these manuscripts. However, for genome-wide experiments, it would be quite difficult for authors to mention large numbers of understudied genes in the abstract. In contrast, one might highlight the expected behaviour of a well-studied protein simply to highlight that the genome-wide study provides credible results. Could this bias the authors' conclusions and, if so, how could this be addressed? For example, would it be worth to normalise studies based on the total number of genes they cover?</p>
<p>Figure 1B is confusing in its present form. I think the plot and/or the legend need revising. For example, what &quot;numbers to the right of each box plot&quot; are the authors referring to? Also, I assume that the filled boxes are understudied genes and the empty/white box is &quot;all genes&quot;, but that's not explained in the legend. In the main text, the figure is referred to with the sentence &quot;we found that hit genes that are highlighted in the title or abstract are strongly over-represented among the 20% highest-studied genes in all biomedical literature &quot;. I cannot follow how the figure shows this. My interpretation is that the y-axis is not showing the number of articles, but represents the percentage of articles mentioning a gene in the title/abstract, displayed on a log scale. If so, perhaps a better axis labels and legend text could be sufficient. But then one would also need to somehow connect this to the statement in the main text about the 20% highest-studied genes (a dashed line?). Alternatively, the authors could consider other ways of plotting these data, e.g. simply plotting the &quot;% of publication in which a gene appears&quot; from 0-100% or so.</p>
</body>
</sub-article>
<sub-article id="sa3" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.93429.1.sa0</article-id>
<title-group>
<article-title>Reviewer #3 (Public Review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary and strengths</p>
<p>
The manuscript investigated the factors related to understudied genes in biomedical research. It showed that understudied are largely abandoned at the writing stage and identified biological and experimental factors associated with selection of highlighted genes.</p>
<p>It is very important for the research community to recognize the systematic bias in research of human genes and take precautions when designing experiments and interpreting results. The authors have tried to profile this issue comprehensively and promoted more awareness and investigation of understudied genes.</p>
<p>Weaknesses</p>
<p>
Regarding result section 1 &quot;Understudied genes are abandoned at synthesis/writing stage&quot;, the figures are not clear and do not convey the messages written in the main text. For example, in Figure 1B, figure S5 and S6,</p>
<p>
- There is no &quot;numbers to the right of each box plot&quot;.</p>
<p>
- Do these box plots only show understudied genes? How many genes are there in each box plot? The definition and numbers of understudied genes are not clear.</p>
<p>
- &quot;We found that hit genes that are highlighted in the title or abstract are strongly over-represented among the 20% highest-studied genes in all biomedical literature (Figure 1B)&quot;. This is not clear from the figure.</p>
<p>Regarding result section 2 &quot;Subsequent reception by other scientists does not penalize studies on understudied genes&quot;, the authors showed in figure 2 that there is a negative correlation between articles per gene before 2015 and median citations to articles published in 2015. Another explanation could be that for popular genes, there are more low-quality articles that didn't get citations, not necessarily that less popular genes attract more citations.</p>
<p>Regarding result section 3 &quot;Identification of biological and experimental factors associated with selection of highlighted genes&quot;, in Figure 3 and table s2, the author stated that &quot;hits with a compound known to affect gene activity are 5.114 times as likely to be mentioned in the title/abstract in an article using transcriptomics&quot;, The number 5.144 comes out of nowhere both in the figure and the table. In addition, figure 4 is not informative enough to be included as a main figure.</p>
</body>
</sub-article>
<sub-article id="sa4" article-type="author-comment">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.93429.1.sa4</article-id>
<title-group>
<article-title>Author Response</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Richardson</surname>
<given-names>Reese AK</given-names>
</name>
<role specific-use="author">Author</role>
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-6058-5886</contrib-id></contrib>
<contrib contrib-type="author">
<name>
<surname>Navarro</surname>
<given-names>Heliodoro Tejedor</given-names>
</name>
<role specific-use="author">Author</role>
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0001-5441-8101</contrib-id></contrib>
<contrib contrib-type="author">
<name>
<surname>Amaral</surname>
<given-names>Luis A Nunes</given-names>
</name>
<role specific-use="author">Author</role>
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-3762-789X</contrib-id></contrib>
<contrib contrib-type="author">
<name>
<surname>Stoeger</surname>
<given-names>Thomas</given-names>
</name>
<role specific-use="author">Author</role>
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-5540-4278</contrib-id></contrib>
</contrib-group>
</front-stub>
<body>
<p>We thank the reviewers for their fair assessment of our work and will submit a revised version edited for clarity of presentation and precision of interpretations.</p>
</body>
</sub-article>
</article>