<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">88039</article-id><article-id pub-id-type="doi">10.7554/eLife.88039</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.88039.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>The impact of stability considerations on genetic fine-mapping</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Aw</surname><given-names>Alan J</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9455-7878</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="pa1">†</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Jin</surname><given-names>Lionel Chentian</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Ioannidis</surname><given-names>Nilah</given-names></name><email>nilah@berkeley.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Song</surname><given-names>Yun S</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0734-9868</contrib-id><email>yss@berkeley.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01an7q238</institution-id><institution>Department of Statistics, University of California</institution></institution-wrap><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01an7q238</institution-id><institution>Center for Computational Biology, University of California</institution></institution-wrap><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01gmv5d77</institution-id><institution>McKinsey &amp; Company</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01an7q238</institution-id><institution>Computer Science Division, University of California</institution></institution-wrap><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Young</surname><given-names>Alexander</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/046rm7j60</institution-id><institution>University of California, Los Angeles</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><author-notes><fn fn-type="present-address" id="pa1"><label>†</label><p>Department of Genetics, University of Pennsylvania, Philadelphia, United States</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>16</day><month>03</month><year>2026</year></pub-date><volume>12</volume><elocation-id>RP88039</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-04-11"><day>11</day><month>04</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-04-13"><day>13</day><month>04</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.04.11.536456"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-07-18"><day>18</day><month>07</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88039.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2026-01-08"><day>08</day><month>01</month><year>2026</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88039.2"/></event></pub-history><permissions><copyright-statement>© 2023, Aw et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Aw et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-88039-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-88039-figures-v1.pdf"/><abstract><p>Fine-mapping methods, which aim to identify genetic variants responsible for complex traits following genetic association studies, typically assume that sufficient adjustments for confounding within the association study cohort have been made, for example, through regressing out the top principal components (i.e., residualization). Despite its widespread use, however, residualization may not completely remove all sources of confounding. Here, we propose a complementary stability-guided approach that does not rely on residualization, which identifies consistently fine-mapped variants across different genetic backgrounds or environments. Simulations show that stability guidance neither outperforms nor underperforms residualization, but each approach picks up different variants considerably often. Critically, prioritizing variants that match between the residualization and stability-guided approaches enhances recovery of causal variants. We further demonstrate the utility of the stability approach by applying it to fine-map eQTLs in the GEUVADIS data. Using 378 different functional annotations of the human genome, including recent deep learning-based annotations (e.g., Enformer), we compare enrichments of these annotations among variants for which the stability and traditional residualization-based fine-mapping approaches agree against those for which they disagree and find that the stability approach enhances the power of traditional fine-mapping methods in identifying variants with functional impact. Finally, in cases where the two approaches report distinct variants, our approach identifies variants comparably enriched for functional annotations. Our findings suggest that the stability principle, as a conceptually simple device, complements existing approaches to fine-mapping, reinforcing recent advocacy of evaluating cross-population and cross-environment portability of biological findings. To support visualization and interpretation of our results, we provide a Shiny app, accessible at <ext-link ext-link-type="uri" xlink:href="https://github.com/songlab-cal/StableFM">https://github.com/songlab-cal/StableFM</ext-link>.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>fine-mapping</kwd><kwd>functional genomics</kwd><kwd>computational predictors</kwd><kwd>variant interpretation</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35-GM134922</award-id><principal-award-recipient><name><surname>Song</surname><given-names>Yun S</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02qenvm24</institution-id><institution>Chan Zuckerberg Initiative</institution></institution-wrap></funding-source><award-id>CZF2019-002449</award-id><principal-award-recipient><name><surname>Song</surname><given-names>Yun S</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection, and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>In statistical fine-mapping, signals stable across stratified subgroups can capture functionally important loci missed by covariate adjustment approaches, and prioritizing agreement between both approaches enhances functional variant discovery.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>An important challenge faced by computational precision health research is the lack of generalizability of biological findings, which are often obtained by studying cohorts that do not include particular communities of individuals. Known as the <italic>cross-population generalizability</italic> or <italic>portability problem</italic>, the challenge persists in multiple settings, including gene expression prediction (<xref ref-type="bibr" rid="bib33">Keys et al., 2020</xref>) and polygenic risk prediction (<xref ref-type="bibr" rid="bib47">Mostafavi et al., 2020</xref>). Generalizable biological signals, such as the functional impact of a variant, are important, as they ensure that general conclusions drawn from cohort-specific analyses are not based on spurious discoveries. Portability problems are potentially attributable to cohort-biased discoveries being treated as generalizable true signals. Efforts to address such problems have included the use of diverse cohorts typically representing multiple population ancestries (<xref ref-type="bibr" rid="bib43">Márquez-Luna et al., 2017</xref>), meta-analyses of earlier studies across diverse cohorts (<xref ref-type="bibr" rid="bib56">Turley et al., 2021</xref>; <xref ref-type="bibr" rid="bib25">Han and Eskin, 2012</xref>; <xref ref-type="bibr" rid="bib46">Morris, 2011</xref>; <xref ref-type="bibr" rid="bib61">Willer et al., 2010</xref>), or focusing on biological markers that are more likely a priori to play a causal role (including proxy variables), for example, rare variants (<xref ref-type="bibr" rid="bib65">Zaidi and Mathieson, 2020</xref>) or the transcriptome (<xref ref-type="bibr" rid="bib38">Liang et al., 2022</xref>).</p><p>In this paper, we consider an approach based on the notion of <italic>stability</italic> to improve generalizability. Being a pillar of veridical data science (<xref ref-type="bibr" rid="bib63">Yu and Kumbier, 2020</xref>), stability refers to the robustness of statistical conclusions to <italic>perturbations</italic> of the data. Perturbations are not arbitrary, but instead, they encode the practitioner’s beliefs about the quality of the data and the nature of the relationship between variables. Well-chosen perturbation schemes will help the practitioner obtain statistical conclusions that are robust, in the sense that they are generalizable rather than spurious findings, and therefore more likely to capture the true signal. For example, in sparse linear models, prioritizing the stability of effect sizes to cross-validation ‘perturbations’ leads to a much smaller set of selected features without reducing predictive performance (<xref ref-type="bibr" rid="bib39">Lim and Yu, 2016</xref>). In another example involving the application of random forests to detect higher-order interactions between gene regulation features (<xref ref-type="bibr" rid="bib8">Basu et al., 2018</xref>), interactions stable to bootstrap perturbations are also largely consistent with known physical interactions between the associated transcription factor binding or enhancer sites.</p><p>To further investigate the utility of the stability approach, we focus on a procedure known as (genetic) fine-mapping (<xref ref-type="bibr" rid="bib53">Schaid et al., 2018</xref>). Fine-mapping is the task of identifying genetic variants that causally affect some trait of interest. From the viewpoint of stability, previous works focusing on cross-population stability of fine-mapped variants typically use Bayesian linear models. The linear model has allowed modeling of heterogeneous effect sizes across different user-defined populations (e.g., ethnic groups, ancestrally distinct populations, or study cohorts), and it is common to assume that causal variants share correlated effects across populations (see, e.g., eq. (24) of <xref ref-type="bibr" rid="bib35">LaPierre et al., 2021</xref> or eq. (9) of <xref ref-type="bibr" rid="bib59">Wen et al., 2015</xref>).</p><p>Unlike the parametric approaches described above, we here apply a <italic>non-parametric</italic> fine-mapping method to GEUVADIS (<xref ref-type="bibr" rid="bib36">Lappalainen et al., 2013</xref>), a database of gene expression traits measured across individuals of diverse genetic ancestries and from different geographical environments, whose accompanying genotypes are available through the 1000 Genomes Project (1 kGP) (<xref ref-type="bibr" rid="bib3">Auton et al., 2015</xref>). We apply fine-mapping through two approaches, one that is commonly used in practice and another that is guided by stability. First, we perform simulations using 1kGP genotypes to evaluate the strength of each approach. We next evaluate the agreement of fine-mapped variants between the two approaches on GEUVADIS data, and then measure the functional significance of variants picked by both approaches using a much wider range of functional annotations than considered in previous studies. By performing various statistical tests on the functional annotations, we evaluate the advantages brought about by the incorporation of stability. Finally, we investigate the broader applicability of the stability approach by implementing and analyzing a stability-guided version of the fine-mapping algorithm SuSiE on simulated data. To support visualization of results obtained from the GEUVADIS analysis at both the single gene and genome-wide levels, we provide an interactive Shiny app which is accessible at: <ext-link ext-link-type="uri" xlink:href="https://github.com/songlab-cal/StableFM">https://github.com/songlab-cal/StableFM</ext-link>. Our app is open-source and designed to support geneticists in interpreting our results, thereby also addressing a growing demand for accessible, <italic>integrative</italic> software to interpret genetic findings in the age of big data and variant annotation databases.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Experimental design</title><p>We apply the fine-mapping method PICS (<xref ref-type="bibr" rid="bib55">Taylor et al., 2021</xref>; <xref ref-type="bibr" rid="bib19">Farh et al., 2015</xref>; see Algorithm 1 in Probabilistic Identification of Causal SNPs) to the GEUVADIS data (<xref ref-type="bibr" rid="bib36">Lappalainen et al., 2013</xref>), which consists of <inline-formula><alternatives><mml:math id="inf1"><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>22</mml:mn><mml:mo>,</mml:mo><mml:mn>720</mml:mn></mml:math><tex-math id="inft1">\begin{document}$T=22,720$\end{document}</tex-math></alternatives></inline-formula> gene expression traits measured across <inline-formula><alternatives><mml:math id="inf2"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>445</mml:mn></mml:math><tex-math id="inft2">\begin{document}$N=445$\end{document}</tex-math></alternatives></inline-formula> individuals, with their accompanying genotypes obtained from the 1000 Genomes Project (<xref ref-type="bibr" rid="bib3">Auton et al., 2015</xref>). These individuals are of either European or African ancestry, with about four-fifths of the cohort made up of individuals of (self-identified) European descent. In particular, these ancestrally different subpopulations have distinct linkage disequilibrium patterns and environmental exposures, which constitute potential confounders that we wish to stabilize the fine-mapping procedure against.</p><p>PICS, like many eQTL analysis methods, requires the lead variant at a locus to compute posterior probabilities, so we perform marginal regressions of each gene expression trait against variants within the fine-mapping locus. Our implementation of PICS generates three sets of variants, which represent putatively causal variants that are (marginally) highly associated, moderately associated, and weakly associated with the gene expression phenotype. (See Materials and methods for details.) As illustrated in <xref ref-type="fig" rid="fig1">Figure 1A</xref>, we compare two versions of the fine-mapping procedure—one that is typically performed in practice, and another that is motivated by the stability approach.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>An overview of our study of the impact of stability considerations on genetic fine-mapping.</title><p>(<bold>A</bold>) The two ways in which we perform fine-mapping, the first of which (colored in green) prioritizes the stability of variant discoveries to subpopulation perturbations. The data illustrates the case where there are two distinct environments, or subpopulations (denoted <inline-formula><alternatives><mml:math id="inf3"><mml:msub><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft3">\begin{document}$E_{1}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf4"><mml:msub><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft4">\begin{document}$E_{2}$\end{document}</tex-math></alternatives></inline-formula>), that split the observations. (<bold>B</bold>) Key steps in our comparison of the stability-guided approach with the popular residualization approach.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig1-v1.tif"/></fig><sec id="s2-1-1"><title>Stable variant</title><p>Specifically, to incorporate stability into fine-mapping and obtain what we call the <italic>stable variant</italic> or stable SNP, we encode into the algorithm our belief that for a gene expression trait, a causal variant would presumably act through the same mechanism regardless of the population from which they originate. Indeed, if a set of genetic variants were causal, the same algorithm should report it, if run on subsets of the data corresponding to heterogeneous populations. This belief is consistent with recent simulation studies showing that the inclusion of GWAS variants discovered in diverse populations both mitigates false discoveries driven by linkage disequilibrium differences (<xref ref-type="bibr" rid="bib37">Li et al., 2022</xref>) and improves generalizability of polygenic score construction (<xref ref-type="bibr" rid="bib13">Cavazos and Witte, 2021</xref>). Hence, the stable variant is the variant with the highest probability of being causal, conditioned on being reported by PICS in the most number of subpopulations. Incorporating stability provides a formal definition of the stable variant.</p></sec><sec id="s2-1-2"><title>Top variant</title><p>The stability consideration we have described is closely related to a popular procedure known as correction for population structure, which residualizes the trait using measured confounders (e.g., top principal components computed from the genotype matrix). We call the variant returned by the residualization approach the <italic>top variant</italic> or top SNP. The residualization approach removes the effects of genetic ancestry and environmental exposures on a trait and is used to avoid the risk of low statistical power borne by stratified analyses. The top variant is formally defined toward the end of PICS. We remark that the top variant is a function of the residualized trait, as opposed to the stable variant, which is a function of the unresidualized trait (see <xref ref-type="fig" rid="fig1">Figure 1A</xref>).</p></sec></sec><sec id="s2-2"><title>Simulation study</title><p>We simulated 100 genes from GEUVADIS gene expression data, sampled proportionally across the 22 autosomes. Mirroring the approach described in Section 4 of <xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref>, synthetic gene expression phenotypes were simulated based on the sampled genes and involved two parameters: <inline-formula><alternatives><mml:math id="inf5"><mml:mi>C</mml:mi></mml:math><tex-math id="inft5">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula>, the number of causal variants, and <inline-formula><alternatives><mml:math id="inf6"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft6">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>, the proportion of variance in gene expression explained by variants in cis. Whenever available, gene expression canonical TSS was used to include only variants lying within 1 Mb upstream and downstream when simulating gene expression phenotypes. We considered all combinations of <inline-formula><alternatives><mml:math id="inf7"><mml:mi>C</mml:mi><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft7">\begin{document}$C\in\{1,2,3\}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf8"><mml:mi>ϕ</mml:mi><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>0.05</mml:mn><mml:mo>,</mml:mo><mml:mn>0.1</mml:mn><mml:mo>,</mml:mo><mml:mn>0.2</mml:mn><mml:mo>,</mml:mo><mml:mn>0.4</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft8">\begin{document}$\phi\in\{0.05,0.1,0.2,0.4\}$\end{document}</tex-math></alternatives></inline-formula>, and simulated two replicates for each gene and for each combination of <inline-formula><alternatives><mml:math id="inf9"><mml:mi>C</mml:mi></mml:math><tex-math id="inft9">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf10"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft10">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>. Altogether, we generated 100 × 3 × 4 × 2 = 2400 datasets, on which we ran PICS and SuSiE. For PICS, we ran four versions: the <italic>stability-guided</italic> version and <italic>residualization</italic> version, which return the stable variant and top variant, respectively; a <italic>combined</italic> version, which runs stability-guided PICS on covariate-residualized phenotypes; as well as a <italic>plain</italic> version, where no correction for population stratification or stability guidance was applied. For SuSiE, we ran two versions: the stability-guided version and the residualization version. We also ran a separate set of simulations comprising six scenarios of environmental heterogeneity by ancestry on 10 genes from Chromosomes 1, 20, 21, and 22 (6 × 10 × 3 × 4 × 2 = 1440 generated datasets), on which we ran the plain and stability-guided versions of PICS.</p><sec id="s2-2-1"><title>Evaluation on simulated genes</title><p>Following <xref ref-type="bibr" rid="bib44">Mazumder, 2020</xref>, we analyze the recovery probability of all fine-mapping algorithms as a function of the signal-to-noise ratio (SNR), defined here as the ratio of true signal to background noise: <inline-formula><alternatives><mml:math id="inf11"><mml:mtext>SNR</mml:mtext><mml:mo>=</mml:mo><mml:mtext>var</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft11">\begin{document}$\text{SNR}=\text{var}(\mathbf{X}\mathbf{b})/\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> (see Simulation study and evaluation details in Materials and methods for details).</p></sec><sec id="s2-2-2"><title>Analysis of GEUVADIS data</title><p>We compare the results obtained by using each approach, across a range of categories of functional annotations including conservation, pathogenicity, chromatin accessibility, transcription factor binding, and histone modification, and across biological assays and computational predictions. See <xref ref-type="table" rid="table1">Table 1</xref> for a full list of annotations covered. (Appendix 5 contains a full description of each quantity and its interpretation.)</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>A list of 378 functional annotations across which the biological significances of stable and top fine-mapped single nucleotide polymorphisms are compared.</title><p>Annotations that report multiple scores have the total number of scores reported shown in parentheses. Scores mined from the FAVOR database (<xref ref-type="bibr" rid="bib66">Zhou et al., 2023</xref>) are indicated by an asterisk (TSS = transcription start site, bp = base pair).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Functional annotation type</th><th align="left" valign="bottom">Functional annotation</th></tr></thead><tbody><tr><td align="left" valign="middle" rowspan="2">Ensembl</td><td align="left" valign="bottom">Distance to Canonical <italic>TSS</italic> (<xref ref-type="bibr" rid="bib15">Cunningham et al., 2022</xref>)</td></tr><tr><td align="left" valign="bottom">Regulatory Features (6; <xref ref-type="bibr" rid="bib15">Cunningham et al., 2022</xref>)</td></tr><tr><td align="left" valign="middle" rowspan="14">Computational predictions</td><td align="left" valign="bottom">CADD∗ (2; <xref ref-type="bibr" rid="bib50">Rentzsch et al., 2019</xref>)</td></tr><tr><td align="left" valign="bottom">SIFTVal∗ (<xref ref-type="bibr" rid="bib48">Ng and Henikoff, 2003</xref>)</td></tr><tr><td align="left" valign="bottom">FATHMM-XF∗ (<xref ref-type="bibr" rid="bib51">Rogers et al., 2018</xref>)</td></tr><tr><td align="left" valign="bottom">LINSIGHT∗ (<xref ref-type="bibr" rid="bib28">Huang et al., 2017</xref>)</td></tr><tr><td align="left" valign="bottom">Polyphen∗ (<xref ref-type="bibr" rid="bib2">Adzhubei et al., 2010</xref>)</td></tr><tr><td align="left" valign="bottom">PhyloP∗ (3; <xref ref-type="bibr" rid="bib49">Pollard et al., 2010</xref>)</td></tr><tr><td align="left" valign="bottom">Gerp∗ (2; <xref ref-type="bibr" rid="bib17">Davydov et al., 2010</xref>)</td></tr><tr><td align="left" valign="bottom">B Statistic∗ (<xref ref-type="bibr" rid="bib45">McVicker et al., 2009</xref>)</td></tr><tr><td align="left" valign="bottom">FunSeq2∗ (<xref ref-type="bibr" rid="bib21">Fu et al., 2014</xref>)</td></tr><tr><td align="left" valign="bottom">ALoFT∗ (<xref ref-type="bibr" rid="bib6">Balasubramanian et al., 2017</xref>)</td></tr><tr><td align="left" valign="bottom">Percent CpG in 75 bp window∗ (<xref ref-type="bibr" rid="bib50">Rentzsch et al., 2019</xref>)</td></tr><tr><td align="left" valign="bottom">Percent GC in 75 bp window∗ (<xref ref-type="bibr" rid="bib50">Rentzsch et al., 2019</xref>)</td></tr><tr><td align="left" valign="bottom">FIRE (<xref ref-type="bibr" rid="bib30">Ioannidis et al., 2017</xref>)</td></tr><tr><td align="left" valign="bottom">Enformer (177 tracks × 2 scores per track; <xref ref-type="bibr" rid="bib4">Avsec et al., 2021</xref>)</td></tr></tbody></table></table-wrap><p><xref ref-type="fig" rid="fig1">Figure 1B</xref> summarizes the key steps of our investigation. To be clear, a first comparison is between the set of genes for which the top and stable variants match and the set of genes for which the top and stable variants disagree; comparisons are made between matching variants and one of the non-matching sets of variants (top or stable). A second comparison is restricted to the set of genes with non-matching top and stable variants, that is, between <italic>paired</italic> sets of variants fine-mapped to a gene, where one set of a pair corresponds to the output of the residualization approach whereas the other corresponds to the output of the stability-guided approach. We additionally compare across three fine-mapping output sets, corresponding to variants that have a high, moderate, or weak marginal association with the expression phenotype. For brevity, we term these three sets Potential Set 1, Potential Set 2, and Potential Set 3, respectively. (See Algorithm 1 for details.) For each Potential Set, we find the top and stable variants as described in Incorporating stability.</p></sec></sec><sec id="s2-3"><title>Simulations justify exploration of stability guidance</title><p>To better understand how stability guidance may support the discovery of causal variants across ancestrally diverse cohorts that possess confounding exogenous factors, we selected 10 genes from Chromosomes 1, 20, 21, and 22 to simulate six scenarios where environmental heterogeneity by ancestry influences gene expression. We consider four scenarios where environmental <italic>variance</italic> differed between ancestries and two scenarios where environmental <italic>mean</italic> differed between ancestries. We also vary the number of causal variants and proportion of variance in gene expression expressed in <italic>cis</italic>. We run Stable PICS, which returns stable variants; alongside a ‘plain’ version of PICS (Plain PICS), which neither performs PC-residualization of phenotypes nor incorporates stability.</p><p>We find that, across these simulations, performance was not significantly different between Stable PICS and Plain PICS: both approaches recover the true causal variant in Potential Set 1 with similar frequencies in simulations involving one causal variant, and in simulations involving two or more causal variants the leading potential sets recover at least one causal variant with similar frequency (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). These also hold for lower potential sets (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). While the two approaches agree more often than not, there is greater disagreement when considering simulations with low SNR (which may potentially reflect real gene expression data—see Stable variants frequently do not match top variants in GEUVADIS). For example, we identified 83% matching variants in Potential Set 1 across all simulations, but this dropped to 74% when restricting to simulations with <inline-formula><alternatives><mml:math id="inf12"><mml:mi>ϕ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft12">\begin{document}$\phi=0.05$\end{document}</tex-math></alternatives></inline-formula>. We next split results by whether the variant returned by Plain PICS had a large posterior probability (PP)—defined as &gt;0.9 to reflect thresholds reported in practice—and found strong agreement between the two approaches when PP was large: for Potential Set 1 the agreement was 98%, Potential Set 2 was 90%, and Potential Set 3 was 93% (<xref ref-type="table" rid="app12table1">Appendix 12—table 1</xref>). These observations together imply that the two approaches disagree particularly when SNR is low and the PP of variants returned by Plain PICS is not large, and Stable PICS recovers causal variants accurately in certain cases where Plain PICS fails to do so (otherwise their performances would be different).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Simulation study results.</title><p>(<bold>A</bold>) The frequency with which at least one causal variant is recovered in Potential Set 1 by Plain PICS and Stable PICS, across 1440 simulated gene expression data that incorporate ancestry-mediated environmental heterogeneity. Recovery frequencies are stratified by simulations differing in the number of causal variants, and the Venn diagram reports the number of matching and non-matching variants in Potential Set 1 across all simulations. (<bold>B</bold>) The frequency with which at least one causal variant is recovered in Potential Set 1 by Combined PICS, Stable PICS, and Top PICS, across 2400 simulated gene expression data. Recovery frequencies are stratified by the SNR parameter <inline-formula><alternatives><mml:math id="inf13"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft13">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> used in simulations, and the Venn diagram reports the number of matching and non-matching variants in Potential Set 1 across all simulations. (<bold>C</bold>) The frequency with which at least one causal variant is recovered in Credible Set 1 by Stable SuSiE and Top SuSiE. Venn diagram reports the number of matching and non-matching variants in Potential Set 1 across all simulations. (<bold>D</bold>) The frequency with which matching and non-matching variants in the first credible or potential set recover a causal variant, obtained from comparing top and stable approaches to an algorithm. We report approximate 95% confidence intervals for each point estimate, by multiplying the associated standard error of the estimate by 1.96.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Plain PICS vs Stable PICS (Potential Sets 2 and 3).</title><p>Frequency with which at least one causal variant is recovered in Potential Sets 2 and 3 by Plain PICS and Stable PICS, across 1440 simulated gene expression phenotypes. Recovery frequencies are stratified by simulations differing in the number of causal variants, but Venn diagrams report the number of matching and non-matching variants across all simulations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Performance of PICS algorithms (Potential Sets 2 and 3).</title><p>Frequency with which at least one causal variant is recovered in Potential Sets 2 and 3 by Combined PICS, Stable PICS, and Top PICS, across 2400 simulated gene expression phenotypes. Recovery frequencies are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf14"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft14">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>, but Venn diagrams report the number of matching and non-matching variants across all simulations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Performance of SuSiE algorithms (Credible Sets 2 and 3).</title><p>Frequency with which at least one causal variant is recovered in Potential Sets 2 and 3 by Stable SuSiE and Top SuSiE, across 2400 simulated gene expression phenotypes. Recovery frequencies are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf15"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft15">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>, but Venn diagrams report the number of matching and non-matching variants across all simulations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp3-v1.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Matching vs non-matching variants (Potential and Credible Sets 1 and 2).</title><p>Frequencies with which matching and non-matching variants in the credible or potential set recover a causal variant, obtained from comparing top and stable approaches to an algorithm. Analysis is performed over 2400 simulated gene expression phenotypes, and recovery frequencies are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf16"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft16">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>. (<bold>A</bold>) Credible or Potential Set 2. (<bold>B</bold>) Credible or Potential Set 3.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp4-v1.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Stable PICS vs Stable SuSiE (one causal variant).</title><p>Empirical discrete probability distributions over the number of causal variants (0 or 1) recovered by stability-guided algorithms in simulations with one causal variant, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp5-v1.tif"/></fig><fig id="fig2s6" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 6.</label><caption><title>Stable PICS vs Stable SuSiE (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by stability-guided algorithms in simulations with two causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp6-v1.tif"/></fig><fig id="fig2s7" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 7.</label><caption><title>Stable PICS vs Stable SuSiE (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by stability-guided algorithms in simulations with three causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp7-v1.tif"/></fig><fig id="fig2s8" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 8.</label><caption><title>Matching Top vs Stable SNP posterior probabilities.</title><p>Posterior probabilities of matching top and stable variants across 2400 simulated gene expression phenotypes. Points are colored by the number of causal variants,,<inline-formula><alternatives><mml:math id="inf17"><mml:mi>S</mml:mi><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft17">\begin{document}$S\in\{1,2,3\}$\end{document}</tex-math></alternatives></inline-formula> set in simulations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp8-v1.tif"/></fig><fig id="fig2s9" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 9.</label><caption><title>Non-matching Top vs Stable SNP posterior probabilities.</title><p>Posterior probabilities of non-matching top and stable variants across 2400 simulated gene expression phenotypes. Points are colored by the number of causal variants,,<inline-formula><alternatives><mml:math id="inf18"><mml:mi>S</mml:mi><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft18">\begin{document}$S\in\{1,2,3\}$\end{document}</tex-math></alternatives></inline-formula> set in simulations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp9-v1.tif"/></fig><fig id="fig2s10" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 10.</label><caption><title>Stable PICS vs SuSiE (one causal variant).</title><p>Empirical discrete probability distributions over the number of causal variants (0 or 1) recovered by Stable PICS or Top SuSiE in simulations with one causal variant, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp10-v1.tif"/></fig><fig id="fig2s11" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 11.</label><caption><title>Stable PICS vs SuSiE (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by Stable PICS or Top SuSiE in simulations with two causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp11-v1.tif"/></fig><fig id="fig2s12" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 12.</label><caption><title>Stable PICS vs SuSiE (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Stable PICS or Top SuSiE in simulations with three causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible or potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp12-v1.tif"/></fig><fig id="fig2s13" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 13.</label><caption><title>Distribution of the number of variants recovered by PICS (one causal variant).</title><p>Empirical discrete probability distributions over the number of causal variants (0 or 1) recovered by PICS algorithms in simulations with one causal variant, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a larger number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp13-v1.tif"/></fig><fig id="fig2s14" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 14.</label><caption><title>Distribution of the number of variants recovered by PICS (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by PICS algorithms in simulations with two causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a larger number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp14-v1.tif"/></fig><fig id="fig2s15" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 15.</label><caption><title>Distribution of the number of variants recovered by PICS (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by PICS algorithms in simulations with three causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp15-v1.tif"/></fig><fig id="fig2s16" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 16.</label><caption><title>Distribution of the number of variants recovered by SuSiE (one causal variant).</title><p>Empirical discrete probability distributions over the number of causal variants (0 or 1) recovered by SuSiE algorithms in simulations with one causal variant, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp16-v1.tif"/></fig><fig id="fig2s17" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 17.</label><caption><title>Distribution of number of variants recovered by SuSiE (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by SuSiE algorithms in simulations with two causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp17-v1.tif"/></fig><fig id="fig2s18" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 18.</label><caption><title>Distribution of number of variants recovered by SuSiE (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by SuSiE algorithms in simulations with three causal variants, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of credible sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp18-v1.tif"/></fig><fig id="fig2s19" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 19.</label><caption><title>Variant recovery frequency of PICS and SuSiE matching and non-matching variants (one causal variant).</title><p>Frequency with which the causal variant is recovered by a matching variant, non-matching top variant, or non-matching stable variant in simulations involving one causal variant. Recovery frequencies are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf19"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft19">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp19-v1.tif"/></fig><fig id="fig2s20" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 20.</label><caption><title>Distribution of number of variants recovered by PICS matching and non-matching variants (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by matching Top and Stable PICS variants, non-matching Top PICS variants, and non-matching Stable PICS variants, in simulations involving two causal variants. Empirical distributions are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf20"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft20">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> (increasing SNR from left to right).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp20-v1.tif"/></fig><fig id="fig2s21" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 21.</label><caption><title>Distribution of the number of variants recovered by SuSiE matching and non-matching variants (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by matching Top and Stable SuSiE variants, non-matching Top SuSiE variants, and non-matching Stable SuSiE variants, in simulations involving two causal variants. Empirical distributions are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf21"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft21">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> (increasing SNR from left to right).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp21-v1.tif"/></fig><fig id="fig2s22" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 22.</label><caption><title>Distribution of number of variants recovered by PICS matching and non-matching variants (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by matching Top and Stable PICS variants, non-matching Top PICS variants, and non-matching Stable PICS variants, in simulations involving three causal variants. Empirical distributions are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf22"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft22">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> (increasing SNR from left to right).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp22-v1.tif"/></fig><fig id="fig2s23" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 23.</label><caption><title>Distribution of number of variants recovered by SuSiE matching and non-matching variants (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by matching Top and Stable SuSiE variants, non-matching Top SuSiE variants, and non-matching Stable SuSiE variants, in simulations involving three causal variants. Empirical distributions are stratified by simulations differing in the signal-to-noise ratio (SNR) parameter <inline-formula><alternatives><mml:math id="inf23"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft23">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> (increasing SNR from left to right).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp23-v1.tif"/></fig><fig id="fig2s24" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 24.</label><caption><title>Plain PICS vs Stable PICS in environmental heterogeneity simulations (one causal variant).</title><p>Empirical discrete probability distributions over the number of causal variants (0 or 1) recovered by Plain or Stable PICS in simulations with one causal variant and environmental heterogeneity, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp24-v1.tif"/></fig><fig id="fig2s25" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 25.</label><caption><title>Plain PICS vs Stable PICS in environmental heterogeneity simulations (two causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, or 2) recovered by Plain or Stable PICS in simulations with two causal variants and environmental heterogeneity, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a greater number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp25-v1.tif"/></fig><fig id="fig2s26" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 26.</label><caption><title>Plain PICS vs Stable PICS in environmental heterogeneity simulations (three causal variants).</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in simulations with three causal variants and environmental heterogeneity, stratified by the SNR parameter used in simulations (increasing SNR from left to right). The impact of including a larger number of potential sets on the distribution is shown (increasing number of included sets from top to bottom).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp26-v1.tif"/></fig><fig id="fig2s27" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 27.</label><caption><title>Plain PICS vs Stable PICS in ‘variance shift (<italic>t</italic> = 8)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf24"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>8</mml:mn></mml:math><tex-math id="inft24">\begin{document}$t=8$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (variance shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp27-v1.tif"/></fig><fig id="fig2s28" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 28.</label><caption><title>Plain PICS vs Stable PICS in ‘variance shift (<italic>t</italic> = 16)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf25"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>16</mml:mn></mml:math><tex-math id="inft25">\begin{document}$t=16$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (variance shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp28-v1.tif"/></fig><fig id="fig2s29" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 29.</label><caption><title>Plain PICS vs Stable PICS in ‘variance shift (<italic>t</italic> = 128)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf26"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>128</mml:mn></mml:math><tex-math id="inft26">\begin{document}$t=128$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (variance shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp29-v1.tif"/></fig><fig id="fig2s30" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 30.</label><caption><title>Plain PICS vs Stable PICS in ‘variance shift (<italic>t</italic> = 256)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf27"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>256</mml:mn></mml:math><tex-math id="inft27">\begin{document}$t=256$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (variance shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp30-v1.tif"/></fig><fig id="fig2s31" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 31.</label><caption><title>Plain PICS vs Stable PICS in ‘mean shift (|<italic>i</italic> − 3|)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf28"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>i</mml:mi><mml:mo>−</mml:mo><mml:mn>3</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:math><tex-math id="inft28">\begin{document}$|i-3|$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (mean shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp31-v1.tif"/></fig><fig id="fig2s32" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 32.</label><caption><title>Plain PICS vs Stable PICS in ‘mean shift (<italic>i</italic> = 3)’ environmental heterogeneity simulations.</title><p>Empirical discrete probability distributions over the number of causal variants (0, 1, 2, or 3) recovered by Plain or Stable PICS in ‘<inline-formula><alternatives><mml:math id="inf29"><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math><tex-math id="inft29">\begin{document}$i=3$\end{document}</tex-math></alternatives></inline-formula>’ simulations involving environmental heterogeneity (mean shift scenario), stratified by the SNR parameter used in simulations (increasing SNR from left to right). Each row reports the distribution for simulations with a specific number of causal variants (1, 2, or 3), and we use all three potential sets to compute the number of causal variants recovered in each case.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig2-figsupp32-v1.tif"/></fig></fig-group><p>Looking closely at each simulation scenario, we see that Stable PICS recovers a causal variant more frequently than Plain PICS in scenarios where there are multiple causal variants and environmental heterogeneity is driven by a ‘spiked mean shift’ (Simulation study and evaluation details), while Plain PICS generally outperforms when environmental heterogeneity is driven by differences in environmental variation. For example, in a set of spiked mean shift simulations with three causal variants, with low SNR (<inline-formula><alternatives><mml:math id="inf30"><mml:mi>ϕ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.1</mml:mn></mml:math><tex-math id="inft30">\begin{document}$\phi=0.1$\end{document}</tex-math></alternatives></inline-formula>) and with exogenous variable drawn from <inline-formula><alternatives><mml:math id="inf31"><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mi>σ</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft31">\begin{document}$N(2\sigma,\sigma^{2})$\end{document}</tex-math></alternatives></inline-formula> for only GBR individuals (rest are drawn from <inline-formula><alternatives><mml:math id="inf32"><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft32">\begin{document}$N(0,\sigma^{2})$\end{document}</tex-math></alternatives></inline-formula>), the frequencies at which at least one causal variant is recovered were 80% and 75% for Stable PICS and Plain PICS, respectively (see <xref ref-type="fig" rid="fig2s32">Figure 2—figure supplement 32</xref>). However, none of these differences in performance is statistically significant. In summary, these findings demonstrate that non-genetic confounding in cohorts can reduce power in methods not adjusting or accounting for ancestral confounding but can be remedied by approaches that do so. (Performances stratified by SNR and number of potential sets included are summarized in <xref ref-type="fig" rid="fig2s24">Figure 2—figure supplements 24</xref>–<xref ref-type="fig" rid="fig2s26">26</xref>; performances stratified by each scenario are summarized in <xref ref-type="fig" rid="fig2s27">Figure 2—figure supplements 27</xref>–<xref ref-type="fig" rid="fig2s32">32</xref>).</p></sec><sec id="s2-4"><title>Simulations reveal advantages (and disadvantages) of stability guidance</title><p>We next conduct a larger set of simulations using 100 genes selected across all 22 autosomes. Similar to the earlier set of simulations, we vary both the number of causal variants and proportion of variance in gene expression explained by variants in <italic>cis</italic>. However, here we compare stability-guided fine-mapping against residualization fine-mapping, a commonly used approach.</p><sec id="s2-4-1"><title>No difference in power between stability guidance and residualization, although considerably many variants do not match</title><p>Comparing stable and top variants returned by Stable PICS and PICS run via the residualization approach (Top PICS), we observed similar causal variant recovery rates. Across all 2400 simulated gene expression phenotypes, the stable variant in Potential Set 1 was causal with frequency 0.63, which is close to and not significantly different from the frequency at which the Potential Set 1 top variant was causal (0.64; McNemar test p-value = 0.38). Similarly close frequencies were observed for lower potential sets (Potential Set 2: <inline-formula><alternatives><mml:math id="inf33"><mml:mtext>stable</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.12</mml:mn><mml:mo>,</mml:mo><mml:mtext>top</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.13</mml:mn></mml:math><tex-math id="inft33">\begin{document}$\text{stable}=0.12,\text{top}=0.13$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 3: <inline-formula><alternatives><mml:math id="inf34"><mml:mtext>stable</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.041</mml:mn><mml:mo>,</mml:mo><mml:mtext>top</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.047</mml:mn></mml:math><tex-math id="inft34">\begin{document}$\text{stable}=0.041,\text{top}=0.047$\end{document}</tex-math></alternatives></inline-formula>), as well as when results were stratified by SNR (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). However, top and stable variants often disagreed, with Potential Set 1 having 72% matching top and stable variants, Potential Set 2 matching at 40% and Potential Set 3 matching at 25% (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). Stratifying results by simulation parameters (number of causal variants, <inline-formula><alternatives><mml:math id="inf35"><mml:mi>S</mml:mi></mml:math><tex-math id="inft35">\begin{document}$S$\end{document}</tex-math></alternatives></inline-formula>, and SNR, <inline-formula><alternatives><mml:math id="inf36"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft36">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>), we generally observed matching fractions increasing with SNR, at least in Potential Set 1: for example, in simulations involving one causal variant, the matching fraction is 69.5% under <inline-formula><alternatives><mml:math id="inf37"><mml:mtext>SNR</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mn>0.05</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>≈</mml:mo><mml:mn>0.053</mml:mn></mml:math><tex-math id="inft37">\begin{document}$\text{SNR}=0.05/(1-0.05)\approx 0.053$\end{document}</tex-math></alternatives></inline-formula>, and it increases monotonically to 89.5% under <inline-formula><alternatives><mml:math id="inf38"><mml:mtext>SNR</mml:mtext><mml:mo>=</mml:mo><mml:mn>0.4</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mn>0.4</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>≈</mml:mo><mml:mn>0.67</mml:mn></mml:math><tex-math id="inft38">\begin{document}$\text{SNR}=0.4/(1-0.4)\approx 0.67$\end{document}</tex-math></alternatives></inline-formula> (<xref ref-type="table" rid="app12table2">Appendix 12—table 2</xref>). There was no clear relationship between matching fractions and the number of causal variants simulated.</p></sec><sec id="s2-4-2"><title>Searching for matching variants between Top PICS and Stable PICS improves causal variant recovery</title><p>Given that stable and top variants frequently do not match despite achieving similar performance, we suspect that, similar to the smaller set of simulations on which Plain PICS and Stable PICS were compared, each approach can recover causal variants in scenarios where the other does not. We thus explore ways to combine the residualization and stability-driven approaches, by considering (1) combining them into a single fine-mapping algorithm (we call the resulting procedure <italic>Combined PICS</italic>); and (2) prioritizing matching variants between the two algorithms. Comparing the performance of Combined PICS against both Top and Stable PICS, however, we find no significant difference in its ability to recover causal variants (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). This conclusion held even when we analyzed performance by stratifying simulations by SNR and number of causal variants simulated (<inline-formula><alternatives><mml:math id="inf39"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft39">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf40"><mml:mi>S</mml:mi></mml:math><tex-math id="inft40">\begin{document}$S$\end{document}</tex-math></alternatives></inline-formula> parameters; see <xref ref-type="fig" rid="fig2s13">Figure 2—figure supplements 13</xref>–<xref ref-type="fig" rid="fig2s15">15</xref>). On the other hand, matching variants between Top and Stable PICS are significantly more likely to be causal. Across all simulations, a matching variant in Potential Set 1 was 2.5× as likely to be causal than either a non-matching top or stable variant (<xref ref-type="fig" rid="fig2">Figure 2D</xref>)—a result that was qualitatively consistent even when we stratified simulations by SNR and number of causal variants simulated (<xref ref-type="fig" rid="fig2s19 fig2s20">Figure 2—figure supplements 19, 20</xref> and <xref ref-type="fig" rid="fig2s22">Figure 2—figure supplement 22</xref>). A similar trend was observed for Potential Set 2, although we do not see much higher causal variant recovery when comparing matching and non-matching variants in Potential Set 3 (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>).</p></sec><sec id="s2-4-3"><title>Stability guidance improves causal variant recovery in SuSiE</title><p>To explore the applicability of stability guidance beyond PICS, we developed a similar procedure that runs the fine-mapping algorithm SuSiE (<xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref>) on multiple slices and returns a stable variant. We call this approach <italic>Stable SuSiE</italic>. We then compared Stable SuSiE against SuSiE applied to residualized phenotypes (<italic>Top SuSiE</italic>), analogous to our comparisons for PICS earlier. First, similar to PICS, we generally observe no significant differences in performance between Stable and Top SuSiE, although empirically more causal variants are recovered by Top SuSiE in larger SNR simulation settings (<xref ref-type="fig" rid="fig2">Figure 2C</xref>), while more causal variants are recovered by Stable SuSiE in lower credible sets (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>; see also <xref ref-type="fig" rid="fig2s16">Figure 2—figure supplements 16</xref>–<xref ref-type="fig" rid="fig2s18">18</xref>). Similar to PICS, we also observed matching fractions increasing with SNR in Potential Set 1 (<xref ref-type="table" rid="app12table3">Appendix 12—table 3</xref>). Next, comparing matching vs non-matching variants between the approaches, we observe a boost in causal variant recovery similar to that observed in PICS (<xref ref-type="fig" rid="fig2">Figure 2D</xref>), which persisted even when we stratified results by simulation parameters (<xref ref-type="fig" rid="fig2s19">Figure 2—figure supplements 19</xref> and <xref ref-type="fig" rid="fig2s21">21</xref>, <xref ref-type="fig" rid="fig2s23">Figure 2—figure supplement 23</xref>). However, one difference is that, unlike in PICS, this boost was unique to Credible Set 1—for the other two credible sets, improvements in causal variant recovery were observed only when SNR is high, with non-matching Stable SuSiE variants recovering at least one causal variant with the highest frequency in low SNR scenarios (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>).</p></sec><sec id="s2-4-4"><title>Other analyses</title><p>Our key findings from the simulation study are as follows. Stability guidance, on its own, does not significantly outperform nor underperform other standard approaches at causal variant recovery. However, prioritizing variants that agree between stability-guided fine-mapping and standard fine-mapping can significantly boost causal variant recovery. In the Supplement, we also describe findings from investigations into the impact of including more potential sets on matching frequency and causal variant recovery (Appendix 2, with <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplements 5</xref>–<xref ref-type="fig" rid="fig2s7">7</xref> discussed), the differences in PP between non-matching variants (Appendix 3, with <xref ref-type="fig" rid="fig2s8">Figure 2—figure supplements 8</xref> and <xref ref-type="fig" rid="fig2s9">9</xref> discussed), the relative performance of SuSiE and Stable PICS (Appendix 4, with <xref ref-type="fig" rid="fig2s10">Figure 2—figure supplements 10</xref>–<xref ref-type="fig" rid="fig2s12">12</xref> discussed), and interpreting matching variants with very low (stable) PP (Appendix 11).</p></sec></sec><sec id="s2-5"><title>Stable variants frequently do not match top variants in GEUVADIS</title><p>Having identified scenarios that favor stability guidance, as well as strategies for using stability guidance to recover causal variants, we now turn to analysis of real gene expression data from GEUVADIS. As a preliminary analysis, when interrogating if the residualization and stability-guided approaches produce the same candidate causal variants, we find that for Potential Set 1, which corresponds to the set typically reported in fine-mapping studies, 56.2% of genes had matching top and stable variants (see <xref ref-type="fig" rid="fig3">Figure 3</xref>). Moving down potential sets, we find less agreement (Potential Set 2: 36.2%, Potential Set 3: 25.6%), providing evidence that as the marginal association of a variant with the expression phenotype decreases, the two approaches prioritize different signals when searching for putatively causal variants. This result is most consistent with our simulations using <inline-formula><alternatives><mml:math id="inf41"><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math><tex-math id="inft41">\begin{document}$S=3$\end{document}</tex-math></alternatives></inline-formula> causal variants and low SNR (<inline-formula><alternatives><mml:math id="inf42"><mml:mi>ϕ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft42">\begin{document}$\phi=0.05$\end{document}</tex-math></alternatives></inline-formula>), where we observed 56.5%, 37%, and 24.5% matching top and stable variants in Potential Sets 1, 2, and 3, respectively.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Venn diagram showing the number of matching and non-matching variants for Potential Set 1 in GEUVADIS fine-mapped variants.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Matching GEUVADIS Top vs Stable SNP posterior probabilities.</title><p>Pair density plot of posterior probabilities of the top variant and the stable variant, in case they match.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Non-matching GEUVADIS Top vs Stable SNP posterior probabilities.</title><p>Pair density plot of posterior probabilities of the top variant and the stable variant, in case they do not match.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig3-figsupp2-v1.tif"/></fig></fig-group><p>Because our simulations demonstrate that prioritizing matching variants boosts causal variant recovery, we proceeded with comparing functional impact of matching stable and top variants against non-matching variants. For the rest of this section, we report only results for Potential Set 1, given that it corresponds to the set typically reported in fine-mapping. Results for the other potential sets are provided in Appendices 7 and 8.</p></sec><sec id="s2-6"><title>Matching vs non-matching variants</title><p>For each gene, our algorithm finds the top eQTL variant and the stable eQTL variant, which may not coincide. We thus run (one-sided) unpaired Wilcoxon tests on matching and non-matching sets of variants to detect significant functional enrichment of one set of variants over the other. We find that the top variants that are also stable for the corresponding gene (<inline-formula><alternatives><mml:math id="inf43"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>12743</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft43">\begin{document}$N_{\text{match}}=12743$\end{document}</tex-math></alternatives></inline-formula> maximum annotatable) score significantly higher in functional annotations than the top variants that are not stable (<inline-formula><alternatives><mml:math id="inf44"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>non-match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>9921</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft44">\begin{document}$N_{\text{non-match}}=9921$\end{document}</tex-math></alternatives></inline-formula> maximum annotatable). Notably, 361 out of 378 functional annotations report one-sided greater p-values &lt;0.05 for the matching (i.e., both top and stable) variants after correcting for multiple testing using the Benjamini–Hochberg (BH) procedure, with many of these annotations measuring magnitudes of functional impact or functional enrichment (e.g., Enformer perturbation scores, FATHMM.XF score). Among the 17 remaining functional annotations, none has significantly lower scores for the matching variants. (Appendix 7 lists these functional annotations in detail.) We also find that empirically, the matching variants tend to have greater agreement in posterior probabilities than non-matching variants (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplements 1</xref> and <xref ref-type="fig" rid="fig2s2">2</xref>).</p><p>As an example of a significant functional annotation, consider raw CADD scores (<xref ref-type="bibr" rid="bib50">Rentzsch et al., 2019</xref>)—a higher value of which indicates a greater likelihood of deleterious effects. Out of the 22,559 genes for which both the top and the stable variant are annotatable, looking at the distribution of top variant scores, the one corresponding to the 12,685 genes with matching top and stable variants stochastically dominates the one corresponding to the remaining 9874 genes (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). This relationship is more pronounced when we inspect PHRED-scaled CADD scores, where we apply a sliding cutoff threshold for calling variant deleteriousness (i.e., potential pathogenicity—see <xref ref-type="bibr" rid="bib50">Rentzsch et al., 2019</xref>). We find that a greater proportion of matching variants than of either non-matching variant is classified as deleterious under the typical range of deleteriousness cutoffs (<xref ref-type="fig" rid="fig4">Figure 4B</xref>).</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Distribution of computational VEP scores across matching and non-matching variants.</title><p><italic>Top row</italic>. CADD scores. (<bold>A</bold>) Empirical cumulative distribution functions of raw CADD scores of matching and non-matching variants across all genes, for Potential Set 1. Non-matching variants are further divided into stable and top variants, with a score lower threshold of 1.0 and upper threshold of 5.0 used to improve visualization. (<bold>B</bold>) For a deleteriousness cutoff, the percent of (1) all matching variants, (2) all non-matching top variants, and (3) all non-matching stable variants, which are classified as deleterious. We use a sliding cutoff threshold ranging from 10 to 20 as recommended by CADD authors. For each value along the x-axis, 95% confidence intervals for point estimates on the y-axis were obtained using the Sison-Glaz method for constructing multinomial distribution standard errors (R command DescTools::MultinomCI(...)). <italic>Bottom row</italic>. Empirical cumulative distribution functions of perturbation scores of Enformer-predicted H3K27me3 ChIP-seq track. Score upper threshold of 0.015 and empirical CDF lower threshold of 0.5 used to improve visualization. (<bold>C</bold>) Perturbation scores computed from predictions based on centering input sequences on the gene TSS as well as its two flanking positions. (<bold>D</bold>) Perturbation scores computed from predictions based on centering input sequences on the gene TSS only.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig4-v1.tif"/></fig><p>Another example demonstrating significant functional enrichment of matching variants over non-matching variants is the perturbation scores on H3K9me3 ChIP-seq peaks, as predicted by the Enformer (<xref ref-type="bibr" rid="bib4">Avsec et al., 2021</xref>). Out of the 6364 genes for which the distance of both the top and the stable variant to the TSS are within the Enformer input sequence length constraint, looking at the distribution of top variant scores, the one corresponding to the 4491 genes with matching top and stable variants stochastically dominates the one corresponding to the remaining 1873 genes (<xref ref-type="fig" rid="fig4">Figure 4C,D</xref>). This relationship is true regardless of whether perturbation scores are calculated from an average of input sequences centered on the gene TSS and its two flanking positions (<xref ref-type="fig" rid="fig4">Figure 4C</xref>), or from input sequences centered on the gene TSS only (<xref ref-type="fig" rid="fig4">Figure 4D</xref>). (See Appendix 6 for details on Enformer annotation calculation.)</p><p>Similarly, amongst all stable variants, those variants that match the top variant are significantly more enriched in functional annotations: 363 functional annotations report one-sided greater BH-adjusted p-values &lt;0.05 for the matching variants, and none of the remaining 15 annotations present significant depletion for the matching variants set.</p></sec><sec id="s2-7"><title>Top vs stable variants when they do not match</title><p>Focusing on the genes for which the top and stable variants are different, we run (one-sided) paired Wilcoxon tests to detect significant functional enrichment of one set of variants over the other. We find in general that for some genes, stable variants can carry more functional impact than top variants, and for other genes, top variants carry more functional impact—although neither of these patterns is statistically significant genome-wide after multiple testing correction using the BH procedure. For example, for raw CADD scores, out of the 9874 genes for which the top and the stable variants do not match, 4906 genes have higher scoring stable variants than top variants, whereas 4968 genes have higher scoring top variants than stable variants (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). Looking at the accompanying PHRED-scaled CADD scores (CADD PHRED) for Potential Set 1, when applying a sliding cutoff threshold for calling variant deleteriousness, we find that even though a higher fraction of genes have top but not stable variant classified as deleterious, the difference is usually not significant (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). <xref ref-type="fig" rid="fig5">Figure 5B</xref> also demonstrates that no matter how the cutoff is chosen, there are genes for which the top variant is not classified as deleterious while the stable variant is. Taken together, these observations suggest that the stability-guided approach can sometimes be more useful at identifying variants of functional significance, and in a broad sense, both fine-mapped variants should be equally prioritized for potential of carrying functional impact.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Comparison of CADD scores across non-matching top and stable variants.</title><p>(<bold>A</bold>) Paired scatterplot of raw CADD scores of both top and stable variant for each gene, for Potential Set 1. (<bold>B</bold>) Percent of genes that are classified as (1) having deleterious top variant only, (2) having deleterious stable variant only, and (3) having both top and stable variant deleterious, using a sliding cutoff threshold ranging from 10 to 20 as recommended by CADD authors.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig5-v1.tif"/></fig><p>Because our comparison between the top and stable variants yielded no significant functional enrichment of one over the other, we investigate whether external factors—for example, the posterior probability of the variants—might moderate the relative enrichment of one variant over the other (i.e., we perform <italic>trend analysis</italic>—see Section for a list of all moderators). Here, we find that all but one of the moderators considered—namely, the Posterior Probability of Top Variant (see <xref ref-type="table" rid="table2">Table 2</xref>)—did not produce any significant trends for Potential Set 1. There is a small but significant positive correlation (Pearson’s <inline-formula><alternatives><mml:math id="inf45"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft45">\begin{document}$r=0.05$\end{document}</tex-math></alternatives></inline-formula>) between the posterior probability of the top variant and the difference between the stable variant and top variant FIRE scores (<xref ref-type="bibr" rid="bib30">Ioannidis et al., 2017</xref>), and a small but significant negative correlation (Pearson’s <inline-formula><alternatives><mml:math id="inf46"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.07</mml:mn></mml:math><tex-math id="inft46">\begin{document}$r=-0.07$\end{document}</tex-math></alternatives></inline-formula>) between the posterior probability of the top variant and the difference between the stable variant and top variant Absolute Distance to Canonical TSS. (Details are reported in Appendix 8.)</p><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>List of six moderating factors considered.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Moderator</th><th align="left" valign="bottom">Quantity/statistic computed</th></tr></thead><tbody><tr><td align="left" valign="bottom">(1) Degree of Stability</td><td align="left" valign="bottom">No. subpopulations for which stable variant has positive probability</td></tr><tr><td align="left" valign="bottom">(2) Population Diversity</td><td align="left" valign="bottom">Maximum of pairwise allele frequency difference between subpopulations for which stable variant has positive posterior probability</td></tr><tr><td align="left" valign="bottom">(3) Population Differentiation</td><td align="left" valign="bottom">Maximum <inline-formula><alternatives><mml:math id="inf47"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft47">\begin{document}$F_{ST} $\end{document}</tex-math></alternatives></inline-formula> between subpopulations for<break/>which stable variant has positive posterior probability</td></tr><tr><td align="left" valign="bottom">(4) Inclusion of Distal Subpopulations (Top)</td><td align="left" valign="bottom">Whether or not the top variant also had positive probability in Yoruban subpopulation when the stability-guided approach was used</td></tr><tr><td align="left" valign="bottom">(5) Inclusion of Distal Subpopulations (Stable)</td><td align="left" valign="bottom">Whether or not the stable variant had positive probability in Yoruban subpopulation when the stability-guided approach was used</td></tr><tr><td align="left" valign="bottom">(6) Degree of Certainty of Causality Using Residualization Approach</td><td align="left" valign="bottom">Posterior probability of top variant</td></tr></tbody></table></table-wrap><sec id="s2-7-1"><title>Additional comparisons</title><p>We perform various <italic>conditional analyses</italic> to evaluate whether additional restrictions to characteristics of fine-mapping outputs may boost the power of either approach over the other. Such characteristics include (1) the positive posterior probability support (i.e., how many variants reported a positive posterior probability from fine-mapping), and (2) the posterior probability of the top or stable variant. Results from our conditional analysis of (1) are reported in Appendix 9. As an example, for (2) we further perform a comparison between top and stable variants, by restricting to genes where the posterior probability of the top variant or the stable variant exceeds 0.9. Such restriction of valid fine-mapped variant-gene pairs is useful in training variant effect prediction models requiring reliable positive and negative examples, as seen in <xref ref-type="bibr" rid="bib58">Wang et al., 2021</xref>. We find that for genes where the top variant posterior probability exceeds 0.9, there is no significant enrichment of the top variant over the stable variant across the 378 annotations considered (all BH-adjusted p-values exceed 0.05). Interestingly, when focusing on genes where the stable variant posterior probability exceeds 0.9, FIRE scores of the top variant are significantly larger than the stable variant (BH-adjusted p-value <inline-formula><alternatives><mml:math id="inf48"><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft48">\begin{document}$=1\times 10^{-6}$\end{document}</tex-math></alternatives></inline-formula>). Detailed results are reported in Appendix 10.</p></sec></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>We have shown that a stability-guided approach complements existing approaches to detect biologically meaningful variants in genetic fine-mapping. Through various statistical comparisons, we have found that prioritizing the agreement between existing approaches and a stability-guided approach enhances the functional impact of the fine-mapped variant. Incorporating stability into fine-mapping also provides an adjuvant approach that helps discover variants of potential functional impact in case standard approaches fail to pick up variants of functional significance. Our findings are consistent with earlier reports of stable discoveries having the tendency to capture actual physical or mechanistic relationships, potentially making such discoveries generalizable or portable (<xref ref-type="bibr" rid="bib8">Basu et al., 2018</xref>).</p><p>The link between stability and generalizability is not new. In the machine learning and statistics literature, it has been shown that stable algorithms provably lead to generalization errors that are well-controlled (<xref ref-type="bibr" rid="bib11">Bousquet and Elisseeff, 2002</xref>, Theorem 17, p. 510). Furthermore, in certain classes of algorithms (e.g., sparse regression), it has been demonstrated empirically that this mathematical relationship is explained precisely by the stable algorithm removing spurious discoveries (<xref ref-type="bibr" rid="bib39">Lim and Yu, 2016</xref>).</p><p>The stability-guided approach is also distinct from other subsampling approaches, such as the bootstrap (<xref ref-type="bibr" rid="bib18">Efron and Tibshirani, 1994</xref>). First, the bootstrap is more often deployed as a method for calibrating uncertainty surrounding a prediction, which is not the objective of our method. Next, in settings where the bootstrap is deployed as a type of perturbation against which a prediction or estimand is expected to be stable (<xref ref-type="bibr" rid="bib8">Basu et al., 2018</xref>), a stability threshold is implicitly needed and would require tuning to be chosen (e.g., in <xref ref-type="bibr" rid="bib8">Basu et al., 2018</xref> this threshold was chosen to be 50%). Here, we leverage interpretable existing external annotations to define the perturbation against which we expect the fine-mapped variant to be stable—in other words, the user relates what is meant by the fine-mapped variant being stable to a biologically meaningful concept like ‘portable across environmentally and LD pattern-wise heterogeneous populations’.</p><p>The last sentence in the preceding paragraph suggests there are teleological similarities between the stability-guided approach and meta-analysis approaches (<xref ref-type="bibr" rid="bib56">Turley et al., 2021</xref>). We emphasize that meta-analyses rely on already analyzed cohorts, thereby implicitly assuming that <italic>within-cohort</italic> heterogeneities have been sufficiently accounted for prior to the reporting of findings for that cohort. The stability-guided approach, however, is relevant to the <italic>cohort-specific analysis itself</italic>, where existing approaches may present methodological insufficiencies resulting in inflated false discoveries. In other words, whereas the goal of meta-analysis may be stated as identifying consistent hits across cohorts while also assuming that findings specific to each cohort are reliable, the goal of a stability-guided approach is to search for consistent signals despite the presence of potential confounders within a single cohort. Our analysis has focused on comparing the stability-guided approach against residualization, a convenient and popular approach to account for population structure, which reflects this difference in purpose.</p><p>Related to the previous point, while revising our work, we came across a few recent papers that developed multi-ancestry fine-mapping algorithms to investigate causal variant heterogeneity across ancestries. <xref ref-type="bibr" rid="bib22">Gao and Zhou, 2024</xref> and <xref ref-type="bibr" rid="bib64">Yuan et al., 2024</xref> found that complex traits have both shared and non-negligible ancestry-specific causal signals, although the proportion of trait-specific causal variants that are ancestry-specific is probably low (<xref ref-type="bibr" rid="bib27">Hu et al., 2025</xref>; <xref ref-type="bibr" rid="bib54">Shi et al., 2021</xref>). On the other hand, for molecular quantitative traits, which include RNA sequencing reads that we have studied in this work, <xref ref-type="bibr" rid="bib42">Lu et al., 2025</xref> reported highly correlated effect sizes in <italic>cis</italic> QTLs across ancestries, with effect heterogeneity concentrated at predicted loss-of-function-intolerant genes. These findings are all consistent with our goal of leveraging stability guidance to identify generalizable causal signals and also imply that stability guidance would be particularly useful for molecular quantitative trait fine-mapping.</p><p>There are several limitations to our present work. First, we have chosen to focus on a non-parametric approach to fine-mapping, but multiple parametric fine-mapping approaches have been extended to incorporate cross-population heterogeneity (e.g., <xref ref-type="bibr" rid="bib35">LaPierre et al., 2021</xref>; <xref ref-type="bibr" rid="bib41">Lu et al., 2022</xref>; <xref ref-type="bibr" rid="bib59">Wen et al., 2015</xref>). While our present work is a proof-of-concept of the applicability of the stability to genetic fine-mapping, we believe that future work focusing on comparing functional impact of variants prioritized by our stability-driven approach or these parametric methods will shed more light on the efficacy of the stability principle at detecting generalizable biological signals. Our results for SuSiE on simulated gene expression data are promising in this regard. Second, our analyses assume that there is at most one <italic>cis</italic> eQTL prioritized by each potential set of PICS. This is owing to the implicit assumption that all other variants in high LD with the causal variant are simply tagging it, though we acknowledge the possibility of relaxing this assumption. Understanding the impact of this relaxation requires extensive investigation; hence, we defer it to future work. Third, while we have relied on numerous computational and experimental annotations to evaluate the functional impact of our fine-mapped variants, some computational variant effect predictors may themselves be biased owing to the lack of ancestral diversity of training data. Recent work has found that deep learning-based gene expression predictors—despite outperforming traditional statistical models at inferring regulatory tracks—explain surprisingly little of gene expression variation across ancestries (<xref ref-type="bibr" rid="bib29">Huang et al., 2023</xref>) and yield limited power at predicting individual gene expression levels more broadly (<xref ref-type="bibr" rid="bib52">Sasse et al., 2023</xref>). While this complicates the use of such predictors in their present form as strong evidence of functional impact, it is more constructive to understand their shortcomings and develop strategies to fine-tune their predictions for future use as trustworthy functional impact metrics—an emerging desideratum in the era of computational biomedicine (<xref ref-type="bibr" rid="bib7">Barbadilla-Martínez et al., 2025</xref>; <xref ref-type="bibr" rid="bib9">Benegas et al., 2025</xref>; <xref ref-type="bibr" rid="bib32">Katsonis et al., 2022</xref>; <xref ref-type="bibr" rid="bib40">Livesey et al., 2025</xref>).</p><p>In closing, while our work explores stability to subpopulation perturbations where subpopulations are defined by ancestry, we emphasize that our stability-guided slicing methodology is applicable to all settings where meaningful external labels are available to the data analyst. For instance, environmental or geographical variables, which are well-recognized determinants of some health outcomes (<xref ref-type="bibr" rid="bib1">Abdellaoui et al., 2022</xref>; <xref ref-type="bibr" rid="bib20">Favé et al., 2018</xref>) and arguably better measure potential confounders than ancestry labels, can be the basis on which slices are defined in the stability-guided approach. As the barriers to access larger biobank-scale datasets, which contain such aforementioned variables, continue to be lowered, we expect stability-driven analyses conducted on such data and relying on carefully defined slices will help users better understand genetic drivers of complex traits. Given its utility in our present work, we believe that the stability approach in precision medicine may find uses beyond genetic fine-mapping and few other biological tasks previously studied, ultimately empowering the discovery of veridical effects not previously known.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><p>We use publicly available GEUVADIS B-lymphocyte RNA-seq measurements from 445 individuals (<xref ref-type="bibr" rid="bib36">Lappalainen et al., 2013</xref>), whose genotypes are also available from the 1000 Genomes Project (<xref ref-type="bibr" rid="bib3">Auton et al., 2015</xref>). The 445 individuals come from five populations: Tuscany Italian (TSI, <inline-formula><alternatives><mml:math id="inf49"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>91</mml:mn></mml:math><tex-math id="inft49">\begin{document}$N=91$\end{document}</tex-math></alternatives></inline-formula>), Great British (GBR, <inline-formula><alternatives><mml:math id="inf50"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>86</mml:mn></mml:math><tex-math id="inft50">\begin{document}$N=86$\end{document}</tex-math></alternatives></inline-formula>), Finnish (FIN, <inline-formula><alternatives><mml:math id="inf51"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>92</mml:mn></mml:math><tex-math id="inft51">\begin{document}$N=92$\end{document}</tex-math></alternatives></inline-formula>), Utah White American (CEU, <inline-formula><alternatives><mml:math id="inf52"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>89</mml:mn></mml:math><tex-math id="inft52">\begin{document}$N=89$\end{document}</tex-math></alternatives></inline-formula>), and Yoruban (YRI, <inline-formula><alternatives><mml:math id="inf53"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>87</mml:mn></mml:math><tex-math id="inft53">\begin{document}$N=87$\end{document}</tex-math></alternatives></inline-formula>). The mRNA measurements are normalized for library depth, expression frequency across individuals as well as PEER factors, as reported in the GEUVADIS Project (<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/biostudies/files/E-GEUV-1/E-GEUV-1/analysis_results/GeuvadisRNASeqAnalysisFiles_README.txt">here</ext-link>). Of the available normalized gene expression phenotypes, only those with non-empty sets of variants lying within 1 Mb upstream or downstream of the canonical transcription start site were kept for (<italic>cis</italic>) fine-mapping. This process yielded 22,664 genes used in all subsequent analyses involving the PICS fine-mapping algorithm.</p><sec id="s4-1"><title>Probabilistic identification of causal SNPs</title><p>We implement PICS (<xref ref-type="bibr" rid="bib19">Farh et al., 2015</xref>), which is based on an earlier genome-wide analysis identifying genetic and epigenetic maps of causal autoimmune disease variants (<xref ref-type="bibr" rid="bib19">Farh et al., 2015</xref>).</p><p>We assume that <inline-formula><alternatives><mml:math id="inf54"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft54">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf55"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow></mml:math><tex-math id="inft55">\begin{document}$\mathbf{y}$\end{document}</tex-math></alternatives></inline-formula> are the <inline-formula><alternatives><mml:math id="inf56"><mml:mi>N</mml:mi><mml:mo>×</mml:mo><mml:mi>P</mml:mi></mml:math><tex-math id="inft56">\begin{document}$N\times P$\end{document}</tex-math></alternatives></inline-formula> locus haplotype (or genotype) matrix and the <inline-formula><alternatives><mml:math id="inf57"><mml:mi>N</mml:mi><mml:mo>×</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft57">\begin{document}$N\times 1$\end{document}</tex-math></alternatives></inline-formula> vector of phenotype values, respectively, with <inline-formula><alternatives><mml:math id="inf58"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow></mml:math><tex-math id="inft58">\begin{document}$\mathbf{y}$\end{document}</tex-math></alternatives></inline-formula> possibly already adjusted by relevant covariates. For consistency of exposition, let the SNPs be named <inline-formula><alternatives><mml:math id="inf59"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft59">\begin{document}$A_{1},\ldots,A_{P}$\end{document}</tex-math></alternatives></inline-formula>. Recall the goal of fine-mapping is to return information about which SNP(s) in the set <inline-formula><alternatives><mml:math id="inf60"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft60">\begin{document}$\{A_{1},\ldots,A_{P}\}$\end{document}</tex-math></alternatives></inline-formula> is (are) causal, given the input pair <inline-formula><alternatives><mml:math id="inf61"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft61">\begin{document}$\{\mathbf{X},\mathbf{y}\}$\end{document}</tex-math></alternatives></inline-formula> and possibly other external information such as functional annotations or, more generally, prior biological knowledge.</p><sec id="s4-1-1"><title>Overview</title><p>PICS is a Bayesian, non-parametric approach to fine-mapping. Given the observed patterns of association at a locus, and furthermore not assuming a parametric model relating the causal variants to the trait itself, one can estimate the probability that any SNP is causal by performing permutations that preserve its marginal association with the trait as well as the LD patterns at the locus.</p><p>To see how this is accomplished, suppose that <inline-formula><alternatives><mml:math id="inf62"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft62">\begin{document}$A_{1}$\end{document}</tex-math></alternatives></inline-formula> is the lead SNP. We are interested in <inline-formula><alternatives><mml:math id="inf63"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft63">\begin{document}$\mathbb{P}(A_{i}^{\text{causal}}|A_{1}=A^{\text{lead}})$\end{document}</tex-math></alternatives></inline-formula>, the probability that <inline-formula><alternatives><mml:math id="inf64"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft64">\begin{document}$A_{i}$\end{document}</tex-math></alternatives></inline-formula> is causal, for each <inline-formula><alternatives><mml:math id="inf65"><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft65">\begin{document}$i\in[P]$\end{document}</tex-math></alternatives></inline-formula>. By Bayes’ theorem,<disp-formula id="equ1"><label>(1)</label><alternatives><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>∝</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msubsup><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>×</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="t1">\begin{document}$$\displaystyle  \mathbb{P}(A_{i}^{\text{causal}}|A_{1}=A^{\text{lead}}) \propto \mathbb{P}(A_{1}=A^{\text{lead}}|A_{i}^{\text{causal}}) \times \mathbb{P}(A_{i}^{\text{causal}}). $$\end{document}</tex-math></alternatives></disp-formula></p><p>Focusing on the focal SNP <inline-formula><alternatives><mml:math id="inf66"><mml:mi>i</mml:mi></mml:math><tex-math id="inft66">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>, permute the rows of <inline-formula><alternatives><mml:math id="inf67"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft67">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> such that the association between the focal SNP and the trait <inline-formula><alternatives><mml:math id="inf68"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow></mml:math><tex-math id="inft68">\begin{document}$\mathbf{y}$\end{document}</tex-math></alternatives></inline-formula> is invariant; see Appendix 1 for a concrete mathematical description of this set of constrained permutations. Then the first term on the RHS of <xref ref-type="disp-formula" rid="equ1">Equation 1</xref>, <inline-formula><alternatives><mml:math id="inf69"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft69">\begin{document}$\mathbb{P}(A_{1}=A^{\text{lead}}|A_{i}^{\text{causal}})$\end{document}</tex-math></alternatives></inline-formula>, is estimated by the proportion of all permutations where <inline-formula><alternatives><mml:math id="inf70"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft70">\begin{document}$A_{1}$\end{document}</tex-math></alternatives></inline-formula> emerges as the lead SNP. The second term, <inline-formula><alternatives><mml:math id="inf71"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft71">\begin{document}$\mathbb{P}(A_{i}^{\text{causal}})$\end{document}</tex-math></alternatives></inline-formula>, is the prior probability of the focal SNP being causal, which the user can choose based on prior knowledge. The default setting in PICS is <inline-formula><alternatives><mml:math id="inf72"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo>…</mml:mo><mml:mo>=</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>P</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft72">\begin{document}$\mathbb{P}(A_{1}^{\text{causal}})=\ldots=\mathbb{P}(A_{P}^{\text{causal}})$\end{document}</tex-math></alternatives></inline-formula>. <xref ref-type="fig" rid="fig6">Figure 6</xref> provides a visual summary of the method.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Visual summary of the PICS algorithm described in Probabilistic Identification of Causal SNPs.</title><p>(<bold>A</bold>) Breakdown of the calculation of the probability of a focal SNP <inline-formula><alternatives><mml:math id="inf73"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft73">\begin{document}$A_{i}$\end{document}</tex-math></alternatives></inline-formula> being causal. (<bold>B</bold>) Illustration of the permutation procedure used to generate the null distribution. An example <inline-formula><alternatives><mml:math id="inf74"><mml:mi>N</mml:mi><mml:mo>×</mml:mo><mml:mi>P</mml:mi></mml:math><tex-math id="inft74">\begin{document}$N\times P$\end{document}</tex-math></alternatives></inline-formula> genotype array with <inline-formula><alternatives><mml:math id="inf75"><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>6</mml:mn></mml:math><tex-math id="inft75">\begin{document}$N=P=6$\end{document}</tex-math></alternatives></inline-formula> is used, with two valid row shuffles, or permutations, of the original array shown. Entries affected by the shuffle are highlighted, as is the focal SNP (<inline-formula><alternatives><mml:math id="inf76"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft76">\begin{document}$A_{3}$\end{document}</tex-math></alternatives></inline-formula>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-fig6-v1.tif"/></fig><p>By running the permutation procedure across all SNPs in the locus, one obtains <inline-formula><alternatives><mml:math id="inf77"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft77">\begin{document}$\mathbb{P}(A_{i}^{\text{causal}}|A_{1}=A^{\text{lead}})$\end{document}</tex-math></alternatives></inline-formula> for each <inline-formula><alternatives><mml:math id="inf78"><mml:mi>i</mml:mi></mml:math><tex-math id="inft78">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>. These posterior probabilities are then normalized so that <inline-formula><alternatives><mml:math id="inf79"><mml:munderover><mml:mo>∑</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft79">\begin{document}$\sum_{i=1}^{N}\mathbb{P}(A_{i}^{\text{causal}}|A_{1}=A^{\text{lead}})=1$\end{document}</tex-math></alternatives></inline-formula>.</p><p>Finally, the above procedure is performed after restricting the set of all SNPs to only those with correlation magnitude <inline-formula><alternatives><mml:math id="inf80"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&gt;</mml:mo><mml:mn>0.5</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft80">\begin{document}$|r| \gt 0.5$\end{document}</tex-math></alternatives></inline-formula> to the lead SNP.</p></sec><sec id="s4-1-2"><title>Algorithm</title><p>Based on the computation of posterior probabilities described above (<xref ref-type="disp-formula" rid="equ1">Equation 1</xref>), the full PICS algorithm returns putatively causal variants as follows.</p><p>The last line of Algorithm 1 returns a list of posterior probability vectors corresponding to each potential set. For our work, we set <inline-formula><alternatives><mml:math id="inf81"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:math><tex-math id="inft81">\begin{document}$|r|=0.5$\end{document}</tex-math></alternatives></inline-formula>, <inline-formula><alternatives><mml:math id="inf82"><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math><tex-math id="inft82">\begin{document}$C=3$\end{document}</tex-math></alternatives></inline-formula>, <inline-formula><alternatives><mml:math id="inf83"><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mn>500</mml:mn></mml:math><tex-math id="inft83">\begin{document}$R=500$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf84"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mrow><mml:mi mathvariant="bold">p</mml:mi></mml:mrow><mml:mn>0</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft84">\begin{document}$\mathbf{p}_0=(1/P,...,1/P)$\end{document}</tex-math></alternatives></inline-formula> throughout implementations of Algorithm 1.</p><table-wrap id="inlinetable1" position="anchor"><table frame="hsides" rules="groups" id="AL1"><tbody><tr><td align="left" valign="bottom"><bold>Algorithm 1</bold>. PICS.</td></tr><tr><td align="left" valign="top">1: <bold>Input:</bold> Individual-by-genotype array <inline-formula><alternatives><mml:math id="inf85"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:munder><mml:mrow><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>×</mml:mo><mml:mi>P</mml:mi></mml:mrow></mml:munder></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft85">\begin{document}$ \underset{N\times P}{\mathbf{X}} $\end{document}</tex-math></alternatives></inline-formula>, phenotype array <inline-formula><alternatives><mml:math id="inf86"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:munder><mml:mrow><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>×</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munder></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft86">\begin{document}$\underset{N\times 1}{\mathbf{y}}$\end{document}</tex-math></alternatives></inline-formula>, LD threshold <inline-formula><alternatives><mml:math id="inf87"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>r</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft87">\begin{document}$|r|$\end{document}</tex-math></alternatives></inline-formula>, resampling number <inline-formula><alternatives><mml:math id="inf88"><mml:mi>R</mml:mi></mml:math><tex-math id="inft88">\begin{document}$R$\end{document}</tex-math></alternatives></inline-formula>, number of potential sets <inline-formula><alternatives><mml:math id="inf89"><mml:mi>C</mml:mi></mml:math><tex-math id="inft89">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula>, prior causal probabilities <inline-formula><alternatives><mml:math id="inf90"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:munder><mml:msub><mml:mrow><mml:mi mathvariant="bold">p</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mi>P</mml:mi><mml:mo>×</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munder></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft90">\begin{document}$\underset{P\times 1}{\mathbf{p}_{0}}$\end{document}</tex-math></alternatives></inline-formula>.<break/>2: <inline-formula><alternatives><mml:math id="inf91"><mml:mi>c</mml:mi><mml:mo stretchy="false">←</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft91">\begin{document}$c\leftarrow 1$\end{document}</tex-math></alternatives></inline-formula><break/>3: <bold>while</bold> <inline-formula><alternatives><mml:math id="inf92"><mml:mi>c</mml:mi><mml:mo>⩽</mml:mo><mml:mi>C</mml:mi></mml:math><tex-math id="inft92">\begin{document}$c\leqslant C$\end{document}</tex-math></alternatives></inline-formula> <bold>do</bold><break/>4:  <inline-formula><alternatives><mml:math id="inf93"><mml:mi>M</mml:mi><mml:mo stretchy="false">←</mml:mo><mml:mtext>No. columns of </mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft93">\begin{document}$M\leftarrow\text{No. columns of }\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula><break/>5:  Identify lead SNP, <inline-formula><alternatives><mml:math id="inf94"><mml:mi>ℓ</mml:mi><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft94">\begin{document}$\ell\in\{1,\ldots,M\}$\end{document}</tex-math></alternatives></inline-formula><break/>6:  Identify neighboring SNPs, <inline-formula><alternatives><mml:math id="inf95"><mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">N</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>ℓ</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mi>M</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>:</mml:mo><mml:mtext>cor</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>ℓ</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&gt;</mml:mo><mml:msup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft95">\begin{document}$\mathcal{N}^{c}_{\ell}=\{k\in[M]:\text{cor}(A_{k},A_{\ell})^{2} \gt r^{2}\}$\end{document}</tex-math></alternatives></inline-formula>.<break/>7:  <bold>for</bold> <inline-formula><alternatives><mml:math id="inf96"><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">N</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>ℓ</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft96">\begin{document}$j\in\mathcal{N}^{c}_{\ell}$\end{document}</tex-math></alternatives></inline-formula> <bold>do</bold><break/>8:   Compute <inline-formula><alternatives><mml:math id="inf97"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mover><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mtext>causal</mml:mtext></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>ℓ</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mrow><mml:mtext>lead</mml:mtext></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft97">\begin{document}$\widehat{p_{j}}=\mathbb{P}(A_{j}^{\text{causal}}|A_{\ell}=A^{\text{lead}})$\end{document}</tex-math></alternatives></inline-formula> as described below <xref ref-type="disp-formula" rid="equ1">Equation 1</xref>. Use <inline-formula><alternatives><mml:math id="inf98"><mml:mi>R</mml:mi></mml:math><tex-math id="inft98">\begin{document}$R$\end{document}</tex-math></alternatives></inline-formula> permutations.<break/>9:  <bold>end for</bold><break/>10:  Record vector of posterior probabilities, <inline-formula><alternatives><mml:math id="inf99"><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mover><mml:msub><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo>:</mml:mo><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">N</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>ℓ</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft99">\begin{document}$\overrightarrow{p^{c}}=[\widehat{p_{j}}:j\in\mathcal{N}^{c}_{\ell}]$\end{document}</tex-math></alternatives></inline-formula><break/>11:  <inline-formula><alternatives><mml:math id="inf100"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo stretchy="false">←</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo><mml:mo class="MJX-variant">∖</mml:mo><mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">N</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>ℓ</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft100">\begin{document}$\mathbf{X}\leftarrow\mathbf{X}[:,\{1,\ldots,M\}\setminus\mathcal{N}_{\ell}^{c}]$\end{document}</tex-math></alternatives></inline-formula><break/>12:  <inline-formula><alternatives><mml:math id="inf101"><mml:mi>c</mml:mi><mml:mo stretchy="false">←</mml:mo><mml:mi>c</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft101">\begin{document}$c\leftarrow c+1$\end{document}</tex-math></alternatives></inline-formula><break/>13: <bold>end while</bold><break/>14: <bold>Output:</bold> <inline-formula><alternatives><mml:math id="inf102"><mml:mo stretchy="false">[</mml:mo><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover><mml:mo>:</mml:mo><mml:mi>c</mml:mi><mml:mo>⩽</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft102">\begin{document}$[\overrightarrow{p^{c}}:c\leqslant C]$\end{document}</tex-math></alternatives></inline-formula> (a list, where component <inline-formula><alternatives><mml:math id="inf103"><mml:mi>c</mml:mi></mml:math><tex-math id="inft103">\begin{document}$c$\end{document}</tex-math></alternatives></inline-formula> corresponds to Potential Set <inline-formula><alternatives><mml:math id="inf104"><mml:mi>c</mml:mi></mml:math><tex-math id="inft104">\begin{document}$c$\end{document}</tex-math></alternatives></inline-formula>)<break/></td></tr></tbody></table></table-wrap><p>As mentioned in the Introduction, we apply two different approaches to implementing Algorithm 1. One is guided by stability and will be introduced shortly in Incorporating stability. The other, which we now describe, is based on regressing out confounders, i.e., residualization. In implementing the residualization approach, we regress the top five principal components, obtained from the genotype matrix <inline-formula><alternatives><mml:math id="inf105"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft105">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula>, from the gene expression phenotype, <inline-formula><alternatives><mml:math id="inf106"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow></mml:math><tex-math id="inft106">\begin{document}$\mathbf{y}$\end{document}</tex-math></alternatives></inline-formula>. We choose five principal components based on the elbow method, an approach described in <xref ref-type="bibr" rid="bib12">Brown et al., 2018</xref>. This yields residuals <inline-formula><alternatives><mml:math id="inf107"><mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>r</mml:mi></mml:mrow></mml:msup></mml:math><tex-math id="inft107">\begin{document}$\mathbf{y}^{r}$\end{document}</tex-math></alternatives></inline-formula> (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), which we use as the input phenotype array in Algorithm 1. We subsequently report the variant with the highest posterior probability in each potential set as the putatively causal variant for that potential set. This variant is referred to as the top variant.</p><p>We remark that there are other ways of using potential confounders in variable selection, such as inclusion as covariates in a linear model. Because our algorithm explicitly avoids assuming a linear model, we have chosen the residualization approach just described.</p></sec></sec><sec id="s4-2"><title>Incorporating stability</title><p>Our stability-guided approach to implementing PICS follows the steps outlined below. Let <inline-formula><alternatives><mml:math id="inf108"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft108">\begin{document}$(\mathbf{X},\mathbf{y})$\end{document}</tex-math></alternatives></inline-formula> be the genotype array and gene expression array. Assume that there are <inline-formula><alternatives><mml:math id="inf109"><mml:mi>K</mml:mi></mml:math><tex-math id="inft109">\begin{document}$K$\end{document}</tex-math></alternatives></inline-formula> subpopulations making up the dataset, and let <inline-formula><alternatives><mml:math id="inf110"><mml:msub><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft110">\begin{document}$E_{k}$\end{document}</tex-math></alternatives></inline-formula> denote the set of row indices of <inline-formula><alternatives><mml:math id="inf111"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft111">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> corresponding to individuals from subpopulation <inline-formula><alternatives><mml:math id="inf112"><mml:mi>k</mml:mi></mml:math><tex-math id="inft112">\begin{document}$k$\end{document}</tex-math></alternatives></inline-formula>. (In our present work, <inline-formula><alternatives><mml:math id="inf113"><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:mn>5</mml:mn></mml:math><tex-math id="inft113">\begin{document}$K=5$\end{document}</tex-math></alternatives></inline-formula>. The five subpopulations are Utahns (<inline-formula><alternatives><mml:math id="inf114"><mml:msub><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>CEU</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>89</mml:mn></mml:math><tex-math id="inft114">\begin{document}$N_{\text{CEU}}=89$\end{document}</tex-math></alternatives></inline-formula>), Finns (<inline-formula><alternatives><mml:math id="inf115"><mml:msub><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>FIN</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>92</mml:mn></mml:math><tex-math id="inft115">\begin{document}$N_{\text{FIN}}=92$\end{document}</tex-math></alternatives></inline-formula>), British (<inline-formula><alternatives><mml:math id="inf116"><mml:msub><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>86</mml:mn></mml:math><tex-math id="inft116">\begin{document}$N_{\text{GBR}}=86$\end{document}</tex-math></alternatives></inline-formula>), Toscani (<inline-formula><alternatives><mml:math id="inf117"><mml:msub><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>TSI</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>91</mml:mn></mml:math><tex-math id="inft117">\begin{document}$N_{\text{TSI}}=91$\end{document}</tex-math></alternatives></inline-formula>), and Yoruban (<inline-formula><alternatives><mml:math id="inf118"><mml:msub><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>87</mml:mn></mml:math><tex-math id="inft118">\begin{document}$N_{\text{YRI}}=87$\end{document}</tex-math></alternatives></inline-formula>).)</p><list list-type="order" id="list1"><list-item><p>Run Algorithm 1 on the pair <inline-formula><alternatives><mml:math id="inf119"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft119">\begin{document}$(\mathbf{X},\mathbf{y})$\end{document}</tex-math></alternatives></inline-formula>. Obtain a list of posterior probability vectors, <inline-formula><alternatives><mml:math id="inf120"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover><mml:mo>:</mml:mo><mml:mi>c</mml:mi><mml:mo>⩽</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft120">\begin{document}$\mathcal{L}=[\overrightarrow{p^{c}}:c\leqslant C]$\end{document}</tex-math></alternatives></inline-formula>. Recall that <inline-formula><alternatives><mml:math id="inf121"><mml:mi>C</mml:mi></mml:math><tex-math id="inft121">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula> is the number of potential sets.</p></list-item><list-item><p>For each subpopulation <inline-formula><alternatives><mml:math id="inf122"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>K</mml:mi></mml:math><tex-math id="inft122">\begin{document}$k=1,\ldots,K$\end{document}</tex-math></alternatives></inline-formula>, run Algorithm 1 on the pair <inline-formula><alternatives><mml:math id="inf123"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:msub><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:msub><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft123">\begin{document}$(\mathbf{X}|_{E_{k}},\mathbf{y}|_{E_{k}})$\end{document}</tex-math></alternatives></inline-formula>. Obtain <inline-formula><alternatives><mml:math id="inf124"><mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mover><mml:msubsup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>k</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>→</mml:mo></mml:mover><mml:mo>:</mml:mo><mml:mi>c</mml:mi><mml:mo>⩽</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft124">\begin{document}$\mathcal{L}^{(k)}=[\overrightarrow{p^{c}_{k}}:c\leqslant C]$\end{document}</tex-math></alternatives></inline-formula> for each <inline-formula><alternatives><mml:math id="inf125"><mml:mi>k</mml:mi></mml:math><tex-math id="inft125">\begin{document}$k$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item><list-item><p>(Stability-guided choice of putatively causal variants) Collect the probability vectors in lists <inline-formula><alternatives><mml:math id="inf126"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow></mml:math><tex-math id="inft126">\begin{document}$\mathcal{L}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf127"><mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:math><tex-math id="inft127">\begin{document}$\mathcal{L}^{(k)}$\end{document}</tex-math></alternatives></inline-formula> (<inline-formula><alternatives><mml:math id="inf128"><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mi>K</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft128">\begin{document}$k\in[K]$\end{document}</tex-math></alternatives></inline-formula>). Operationalizing the principle that a stable variant has positive probability across multiple slices, we pick causal variants as follows. For potential set <inline-formula><alternatives><mml:math id="inf129"><mml:mi>c</mml:mi></mml:math><tex-math id="inft129">\begin{document}$c$\end{document}</tex-math></alternatives></inline-formula>, pick the variants that have (1) positive probability in <inline-formula><alternatives><mml:math id="inf130"><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover></mml:math><tex-math id="inft130">\begin{document}$\overrightarrow{p^{c}}$\end{document}</tex-math></alternatives></inline-formula> in Step 1, and moreover (2) have positive probability in the most number of probability vectors; call this set of variants <inline-formula><alternatives><mml:math id="inf131"><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft131">\begin{document}$\mathcal{S}_{c}$\end{document}</tex-math></alternatives></inline-formula> (note <inline-formula><alternatives><mml:math id="inf132"><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft132">\begin{document}$\mathcal{S}_{c}$\end{document}</tex-math></alternatives></inline-formula> is a subset of the support of <inline-formula><alternatives><mml:math id="inf133"><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover></mml:math><tex-math id="inft133">\begin{document}$\overrightarrow{p^{c}}$\end{document}</tex-math></alternatives></inline-formula> for Potential Set <inline-formula><alternatives><mml:math id="inf134"><mml:mi>c</mml:mi></mml:math><tex-math id="inft134">\begin{document}$c$\end{document}</tex-math></alternatives></inline-formula>). Among members of <inline-formula><alternatives><mml:math id="inf135"><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft135">\begin{document}$\mathcal{S}_{c}$\end{document}</tex-math></alternatives></inline-formula>, select the variant that had the highest posterior probability in <inline-formula><alternatives><mml:math id="inf136"><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover></mml:math><tex-math id="inft136">\begin{document}$\overrightarrow{p^{c}}$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item></list><p>In other words, the stability-guided approach reports the variant that not only appears with positive probability in the most number of subsets including the pooled sample, but also has the largest probability in the pooled sample. This variant is referred to as the stable variant.</p><p>To illustrate the last step, suppose there are <inline-formula><alternatives><mml:math id="inf137"><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math><tex-math id="inft137">\begin{document}$K=2$\end{document}</tex-math></alternatives></inline-formula> subpopulations, <inline-formula><alternatives><mml:math id="inf138"><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft138">\begin{document}$C=1$\end{document}</tex-math></alternatives></inline-formula> potential set, and <inline-formula><alternatives><mml:math id="inf139"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mn>5</mml:mn></mml:math><tex-math id="inft139">\begin{document}$P=5$\end{document}</tex-math></alternatives></inline-formula> SNPs. Let the outputs from Steps 1 and 2 be<disp-formula id="equ2"><alternatives><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mi class="mathcal" mathvariant="script">L</mml:mi></mml:mrow><mml:mo>:</mml:mo></mml:mtd><mml:mtd><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mn>1</mml:mn></mml:msup><mml:mo>→</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0.45</mml:mn><mml:mo>,</mml:mo><mml:mn>0.43</mml:mn><mml:mo>,</mml:mo><mml:mn>0.02</mml:mn><mml:mo>,</mml:mo><mml:mn>0.10</mml:mn><mml:mo>,</mml:mo><mml:mn>0.00</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi class="mathcal" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>:</mml:mo></mml:mtd><mml:mtd><mml:mover><mml:msubsup><mml:mi>p</mml:mi><mml:mn>1</mml:mn><mml:mn>1</mml:mn></mml:msubsup><mml:mo>→</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0.00</mml:mn><mml:mo>,</mml:mo><mml:mn>0.40</mml:mn><mml:mo>,</mml:mo><mml:mn>0.10</mml:mn><mml:mo>,</mml:mo><mml:mn>0.25</mml:mn><mml:mo>,</mml:mo><mml:mn>0.25</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi class="mathcal" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo>:</mml:mo></mml:mtd><mml:mtd><mml:mover><mml:msubsup><mml:mi>p</mml:mi><mml:mn>2</mml:mn><mml:mn>1</mml:mn></mml:msubsup><mml:mo>→</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0.20</mml:mn><mml:mo>,</mml:mo><mml:mn>0.60</mml:mn><mml:mo>,</mml:mo><mml:mn>0.00</mml:mn><mml:mo>,</mml:mo><mml:mn>0.10</mml:mn><mml:mo>,</mml:mo><mml:mn>0.10</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="t2">\begin{document}$$\displaystyle  \begin{array}{ll}\mathcal{L}: &amp; \overrightarrow{p^1}=(0.45,0.43,0.02,0.10,0.00)\\ \mathcal{L}^{(1)}: &amp; \overrightarrow{p^1_1}=(0.00,0.40,0.10,0.25,0.25)\\ \mathcal{L}^{(2)}: &amp; \overrightarrow{p^1_2}=(0.20,0.60,0.00,0.10,0.10)\end{array}$$\end{document}</tex-math></alternatives></disp-formula></p><p>In this example, the second and fourth variants have positive probability in the most number of probability vectors (<inline-formula><alternatives><mml:math id="inf140"><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft140">\begin{document}$\mathcal{S}_{1}=\{A_{2},A_{4}\}$\end{document}</tex-math></alternatives></inline-formula>). Among <inline-formula><alternatives><mml:math id="inf141"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft141">\begin{document}$A_{2}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf142"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft142">\begin{document}$A_{4}$\end{document}</tex-math></alternatives></inline-formula>, <inline-formula><alternatives><mml:math id="inf143"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft143">\begin{document}$A_{2}$\end{document}</tex-math></alternatives></inline-formula> has a higher posterior probability (i.e., 0.43) in <inline-formula><alternatives><mml:math id="inf144"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mover><mml:msup><mml:mi>p</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>→</mml:mo></mml:mover></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft144">\begin{document}$\overrightarrow{p^{1}}$\end{document}</tex-math></alternatives></inline-formula>, so the stable variant reported is <inline-formula><alternatives><mml:math id="inf145"><mml:msub><mml:mi>A</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft145">\begin{document}$A_{2}$\end{document}</tex-math></alternatives></inline-formula>. As a side remark, the posterior probability vector <inline-formula><alternatives><mml:math id="inf146"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow></mml:math><tex-math id="inft146">\begin{document}$\mathcal{L}$\end{document}</tex-math></alternatives></inline-formula> does not generally agree with the posterior probability vector computed using the residualization approach, because the latter is computed by running Algorithm 1 on the pair <inline-formula><alternatives><mml:math id="inf147"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>r</mml:mi></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft147">\begin{document}$(\mathbf{X},\mathbf{y}^{r})$\end{document}</tex-math></alternatives></inline-formula> rather than <inline-formula><alternatives><mml:math id="inf148"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft148">\begin{document}$(\mathbf{X},\mathbf{y})$\end{document}</tex-math></alternatives></inline-formula>, as described earlier.</p><sec id="s4-2-1"><title>Stability-guided SuSiE</title><p>We apply the stability principle to SuSiE (<xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref>) in a similar manner as described above. Concretely, we run SuSiE on the pair <inline-formula><alternatives><mml:math id="inf149"><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft149">\begin{document}$(\mathbf{X},\mathbf{y})$\end{document}</tex-math></alternatives></inline-formula>, before running it again on each subpopulation <inline-formula><alternatives><mml:math id="inf150"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>K</mml:mi></mml:math><tex-math id="inft150">\begin{document}$k=1,\ldots,K$\end{document}</tex-math></alternatives></inline-formula>. Collecting the posterior inclusion probability (PIP) vectors across each subpopulation and the entire sample, we then keep only variants that have posterior probability at least <inline-formula><alternatives><mml:math id="inf151"><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mtext>no. variants included</mml:mtext><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft151">\begin{document}$1/(\text{no. variants included})$\end{document}</tex-math></alternatives></inline-formula>, owing to SuSiE tending to return nonzero probabilities to all variants included in the fine-mapping. These ‘good’ variants (i.e., variants with PIP at least as large as the prior inclusion probability) are analogous to the ‘variants with positive probability’ in PICS. The remaining steps are identical to stability-guided PICS.</p></sec></sec><sec id="s4-3"><title>Simulation study and evaluation details</title><sec id="s4-3-1"><title>Main simulation study</title><p>We simulate 2400 synthetic gene expression data, which vary by two parameters: <inline-formula><alternatives><mml:math id="inf152"><mml:mi>C</mml:mi></mml:math><tex-math id="inft152">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula>, the number of causal variants, and <inline-formula><alternatives><mml:math id="inf153"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft153">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>, the proportion of variance in <inline-formula><alternatives><mml:math id="inf154"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow></mml:math><tex-math id="inft154">\begin{document}$\mathbf{y}$\end{document}</tex-math></alternatives></inline-formula> explained by the genotypes <inline-formula><alternatives><mml:math id="inf155"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft155">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula>. Phenotypes are simulated as follows.</p><list list-type="order" id="list2"><list-item><p>For a gene, whenever its canonical transcription start site (TSS) is available, restrict the genotype matrix <inline-formula><alternatives><mml:math id="inf156"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft156">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> to only variants lying within 1 Mb upstream and downstream. Else, use variants spanning the entire autosome associated with that gene. This ensures that <italic>cis</italic> eQTLs are used in simulation, for the most part.</p></list-item><list-item><p>Sample the indices of the <inline-formula><alternatives><mml:math id="inf157"><mml:mi>C</mml:mi></mml:math><tex-math id="inft157">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula> causal variants, <inline-formula><alternatives><mml:math id="inf158"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="script">C</mml:mi></mml:mrow></mml:math><tex-math id="inft158">\begin{document}$\mathscr{C}$\end{document}</tex-math></alternatives></inline-formula>, uniformly at random from <inline-formula><alternatives><mml:math id="inf159"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft159">\begin{document}$\{1,\ldots,p\}$\end{document}</tex-math></alternatives></inline-formula>, where <inline-formula><alternatives><mml:math id="inf160"><mml:mi>p</mml:mi></mml:math><tex-math id="inft160">\begin{document}$p$\end{document}</tex-math></alternatives></inline-formula> is the number of variants included in <inline-formula><alternatives><mml:math id="inf161"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft161">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> from Step 1.</p></list-item><list-item><p>For each <inline-formula><alternatives><mml:math id="inf162"><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="script">C</mml:mi></mml:mrow></mml:math><tex-math id="inft162">\begin{document}$j\in\mathscr{C}$\end{document}</tex-math></alternatives></inline-formula>, independently draw <inline-formula><alternatives><mml:math id="inf163"><mml:msub><mml:mi>b</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mn>0.6</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft163">\begin{document}$b_{j}\sim N(0,0.6^{2})$\end{document}</tex-math></alternatives></inline-formula> and, for all <inline-formula><alternatives><mml:math id="inf164"><mml:mi>j</mml:mi><mml:mo>∉</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="script">C</mml:mi></mml:mrow></mml:math><tex-math id="inft164">\begin{document}$j\not\in\mathscr{C}$\end{document}</tex-math></alternatives></inline-formula>, set <inline-formula><alternatives><mml:math id="inf165"><mml:msub><mml:mi>b</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math><tex-math id="inft165">\begin{document}$b_{j}=0$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item><list-item><p>Set <inline-formula><alternatives><mml:math id="inf166"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft166">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> to achieve the desired proportion of variance explained <inline-formula><alternatives><mml:math id="inf167"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft167">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item><list-item><p>Generate the phenotype by drawing <inline-formula><alternatives><mml:math id="inf168"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">y</mml:mi></mml:mrow><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">I</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft168">\begin{document}$\mathbf{y}\sim N(\mathbf{X}\mathbf{b},\sigma^{2}\mathbf{I}_{n\times n})$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item></list><p>After gene expression data were simulated, we ran three versions of PICS: the stability-guided version, the residualization version, and an uncorrected version, where no correction for population stratification or stability guidance was applied. To investigate the broader applicability of the stability approach, we also ran two versions of SuSiE (<xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref>), a popular fine-mapping method relying on variational approximation techniques. The first version runs SuSiE on residualized phenotypes, while the second version incorporates stability in selecting the most likely causal variant(s). In both approaches, we rely on default parameters of SuSiE while setting the number of credible sets to return to 3 (i.e., L=3 in <monospace>susieR::susie</monospace>).</p></sec><sec id="s4-3-2"><title>Smaller simulation study</title><p>To investigate whether non-genetic confounding between ancestries in a multi-ancestry cohort can hinder the performance of fine-mapping algorithms that do not correct for potential confounding by ancestry, we simulate a smaller set of synthetic gene expression data. We select 10 genes from Chromosomes 1, 20, 21, and 22 and repeat the simulation steps as described above, up to Step 4. At Step 4, instead of choosing a single <inline-formula><alternatives><mml:math id="inf169"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft169">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula>, we allow <inline-formula><alternatives><mml:math id="inf170"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft170">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> to depend on the ancestry of the individual in order to capture environmental heterogeneity. We specifically consider six scenarios that we classify as (environmental) variance heterogeneity and mean heterogeneity.</p><list list-type="bullet" id="list3"><list-item><p><italic>Variance heterogeneity</italic>. Let <inline-formula><alternatives><mml:math id="inf171"><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>TSI</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>FIN</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>CEU</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft171">\begin{document}$\boldsymbol{\alpha}(t)=(\alpha_{\text{TSI}}(t),\alpha_{\text{GBR}}(t),\alpha_{ \text{FIN}}(t),\alpha_{\text{CEU}}(t),\alpha_{\text{YRI}}(t))$\end{document}</tex-math></alternatives></inline-formula> be a vector of exponentiated proportions, where we define the exponentiated proportion of a subpopulation POP (<inline-formula><alternatives><mml:math id="inf172"><mml:mtext>POP</mml:mtext><mml:mo>∈</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mtext>TSI</mml:mtext><mml:mo>,</mml:mo><mml:mtext>FIN</mml:mtext><mml:mo>,</mml:mo><mml:mtext>GBR</mml:mtext><mml:mo>,</mml:mo><mml:mtext>CEU</mml:mtext><mml:mo>,</mml:mo><mml:mtext>YRI</mml:mtext><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math><tex-math id="inft172">\begin{document}$\text{POP}\in\{\text{TSI},\text{FIN},\text{GBR},\text{CEU},\text{YRI}\}$\end{document}</tex-math></alternatives></inline-formula>) as <inline-formula><alternatives><mml:math id="inf173"><mml:msub><mml:mi>α</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>POP</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>POP</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>TSI</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>FIN</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>CEU</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>N</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft173">\begin{document}$\alpha_{\text{POP}}(t)=N_{\text{POP}}^{t}/(N_{\text{TSI}}^{t}+N_{\text{GBR}}^{ t}+N_{\text{FIN}}^{t}+N_{\text{CEU}}^{t}+N_{\text{YRI}}^{t})$\end{document}</tex-math></alternatives></inline-formula>. In Step 4, we define a vector of ancestry-specific exogenous variances, <inline-formula><alternatives><mml:math id="inf174"><mml:msup><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft174">\begin{document}$\boldsymbol{\sigma^{2}}(t)=5\sigma^{2}\boldsymbol{\alpha}(t)$\end{document}</tex-math></alternatives></inline-formula>, where <inline-formula><alternatives><mml:math id="inf175"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft175">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> is calculated as in Step 4 of the original simulation setup. We then generate phenotypes independently in Step 5 by conditioning on the individual’s population membership (POP): <inline-formula><alternatives><mml:math id="inf176"><mml:msub><mml:mi>y</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:msup><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft176">\begin{document}$y_{i}\sim N(\boldsymbol{x}_{i}\mathbf{b},\boldsymbol{\sigma^{2}}(t)(k))$\end{document}</tex-math></alternatives></inline-formula>, where <inline-formula><alternatives><mml:math id="inf177"><mml:mi>k</mml:mi></mml:math><tex-math id="inft177">\begin{document}$k$\end{document}</tex-math></alternatives></inline-formula> is the index that corresponds to the population membership of individual <inline-formula><alternatives><mml:math id="inf178"><mml:mi>i</mml:mi></mml:math><tex-math id="inft178">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf179"><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft179">\begin{document}$\boldsymbol{x}_{i}$\end{document}</tex-math></alternatives></inline-formula> denotes the genotypes of individual <inline-formula><alternatives><mml:math id="inf180"><mml:mi>i</mml:mi></mml:math><tex-math id="inft180">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>. For example, if <inline-formula><alternatives><mml:math id="inf181"><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math><tex-math id="inft181">\begin{document}$t=0$\end{document}</tex-math></alternatives></inline-formula>, <inline-formula><alternatives><mml:math id="inf182"><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>5</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>5</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>5</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>5</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>5</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft182">\begin{document}$\boldsymbol{\alpha}(t)=\boldsymbol{\alpha}(0)=(1/5,1/5,1/5,1/5,1/5)$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf183"><mml:msup><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn mathvariant="bold">2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft183">\begin{document}$\boldsymbol{\sigma^{2}}(t)=(\sigma^{2},\sigma^{2},\sigma^{2},\sigma^{2},\sigma ^{2})$\end{document}</tex-math></alternatives></inline-formula>, which reduces to our original simulation setup. The choice of <inline-formula><alternatives><mml:math id="inf184"><mml:mi>t</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn></mml:math><tex-math id="inft184">\begin{document}$t \gt 0$\end{document}</tex-math></alternatives></inline-formula> determines the heterogeneity of the exogenous variance among the subpopulations, with a larger <inline-formula><alternatives><mml:math id="inf185"><mml:mi>t</mml:mi></mml:math><tex-math id="inft185">\begin{document}$t$\end{document}</tex-math></alternatives></inline-formula> producing higher heterogeneity. Here, we consider four choices of <inline-formula><alternatives><mml:math id="inf186"><mml:mi>t</mml:mi></mml:math><tex-math id="inft186">\begin{document}$t$\end{document}</tex-math></alternatives></inline-formula>: 8, 16, 128, and 256.</p></list-item><list-item><p><italic>Mean heterogeneity</italic>. Step 5 of the original simulation setup is modified to include an ancestry-dependent mean vector. We arbitrarily let GBR be the ‘focal’ population and generate phenotypes in two ways.</p><list list-type="bullet" id="list3subList1"><list-item><p>‘Smooth mean’: After <inline-formula><alternatives><mml:math id="inf187"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft187">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> is calculated in Step 4 of the original simulation setup, define the mean exogenous noise vector as <inline-formula><alternatives><mml:math id="inf188"><mml:mi mathvariant="bold-italic">μ</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>TSI</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>FIN</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>CEU</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mi>σ</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>σ</mml:mi><mml:mo>,</mml:mo><mml:mi>σ</mml:mi><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>σ</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft188">\begin{document}$\boldsymbol{\mu}=(\mu_{\text{TSI}},\mu_{\text{GBR}},\mu_{\text{FIN}},\mu_{ \text{CEU}},\mu_{\text{YRI}})=(2\sigma,0,\sigma,\sigma,2\sigma)$\end{document}</tex-math></alternatives></inline-formula>. An individual <inline-formula><alternatives><mml:math id="inf189"><mml:mi>i</mml:mi></mml:math><tex-math id="inft189">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> has their phenotype simulated as <inline-formula><alternatives><mml:math id="inf190"><mml:msub><mml:mi>y</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">μ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft190">\begin{document}$y_{i}\sim N(\boldsymbol{x}_{i}\mathbf{b}+\boldsymbol{\mu}(k),\sigma^{2})$\end{document}</tex-math></alternatives></inline-formula>, where again <inline-formula><alternatives><mml:math id="inf191"><mml:mi>k</mml:mi></mml:math><tex-math id="inft191">\begin{document}$k$\end{document}</tex-math></alternatives></inline-formula> is the index that corresponds to the population membership of individual <inline-formula><alternatives><mml:math id="inf192"><mml:mi>i</mml:mi></mml:math><tex-math id="inft192">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>. This effectively increases the gene expression of all but GBR individuals in a positive direction with varying amounts of shift.</p> </list-item><list-item><p>‘Spiked mean’: After <inline-formula><alternatives><mml:math id="inf193"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft193">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> is calculated in Step 4 of the original simulation setup, define <inline-formula><alternatives><mml:math id="inf194"><mml:mi mathvariant="bold-italic">μ</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>σ</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft194">\begin{document}$\boldsymbol{\mu}=(0,2\sigma,0,0,0)$\end{document}</tex-math></alternatives></inline-formula> and simulate individual <inline-formula><alternatives><mml:math id="inf195"><mml:mi>i</mml:mi></mml:math><tex-math id="inft195">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>’s phenotype as <inline-formula><alternatives><mml:math id="inf196"><mml:msub><mml:mi>y</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">b</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mi mathvariant="bold-italic">μ</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft196">\begin{document}$y_{i}\sim N(\boldsymbol{x}_{i}\mathbf{b}+\boldsymbol{\mu}(k),\sigma^{2})$\end{document}</tex-math></alternatives></inline-formula>. This effectively increases the gene expression of GBR individuals only in a positive direction, while the other subpopulation individuals share the same zero mean (as in the original simulation setup).</p> </list-item></list></list-item></list><p>For brevity, we refer to these six scenarios by <monospace>t=8, t=16, t=128, t=256, |i-3|</monospace> and <monospace>i=3</monospace>.</p></sec><sec id="s4-3-3"><title>Evaluation</title><p>Several metrics have been used to evaluate Bayesian fine-mapping algorithms (see e.g., <xref ref-type="bibr" rid="bib10">Benner et al., 2016</xref>; <xref ref-type="bibr" rid="bib14">Cui et al., 2024</xref>; <xref ref-type="bibr" rid="bib26">Hormozdiari et al., 2015</xref>; <xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref>; <xref ref-type="bibr" rid="bib60">Wen et al., 2016</xref>; <xref ref-type="bibr" rid="bib62">Yang et al., 2023</xref>), many of which compute credible sets and analyze their coverage. A challenge in using credible sets for our study is that the posterior probabilities returned by the PICS algorithm are not posterior inclusion probabilities in a strict sense: they do not measure the frequency with which a variant should be included in a <italic>linear model</italic> (<xref ref-type="bibr" rid="bib24">Griffin and Steel, 2021</xref>; note PICS does not assume a model relating features to outcome). This complicates defining credible sets and coverage, so we follow <xref ref-type="bibr" rid="bib44">Mazumder, 2020</xref> instead and evaluate performance of all algorithms by computing the signal recovery probability as a function of the signal to background noise (SNR). By the definition of <inline-formula><alternatives><mml:math id="inf197"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft197">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> (see <xref ref-type="bibr" rid="bib57">Wang et al., 2020</xref> for details), we may compute SNR directly from <inline-formula><alternatives><mml:math id="inf198"><mml:mi>ϕ</mml:mi></mml:math><tex-math id="inft198">\begin{document}$\phi$\end{document}</tex-math></alternatives></inline-formula> as <inline-formula><alternatives><mml:math id="inf199"><mml:mtext>SNR</mml:mtext><mml:mo>=</mml:mo><mml:mi>ϕ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>ϕ</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft199">\begin{document}$\text{SNR}=\phi/(1-\phi)$\end{document}</tex-math></alternatives></inline-formula>.</p></sec></sec><sec id="s4-4"><title>Statistical comparison methodology</title><p>We rely on 378 external functional annotations to interpret biological significance of our variants, summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Appendix 5 provides interpretation for the functional annotations, while Appendix 6 describes in greater detail how we generate annotations from Enformer predictions.</p><p>Our comparison of the top and stable variants is twofold. First, we evaluate the relative significance of the stable variant against the top variant by running paired Wilcoxon two-sample tests across all pairs of top and stable variants across all GEUVADIS gene expression phenotypes. We compute one-directional p-values in either direction to check for significant depletion or enrichment of the stable variant with respect to a particular annotation. Raw p-values are adjusted for false discovery rate control by applying the standard BH procedure (R command <monospace>p.adjust(…,method=’BH’)</monospace>) to p-values across all 378 annotations and all potential sets.</p><p>To investigate whether various external factors might moderate the differences in functional annotations, we next perform trend tests. Basically, for some external factor <inline-formula><alternatives><mml:math id="inf200"><mml:mi>F</mml:mi></mml:math><tex-math id="inft200">\begin{document}$F$\end{document}</tex-math></alternatives></inline-formula>, we run a trend test to see if values of <inline-formula><alternatives><mml:math id="inf201"><mml:mi>F</mml:mi></mml:math><tex-math id="inft201">\begin{document}$F$\end{document}</tex-math></alternatives></inline-formula> are associated with attenuation or augmentation of differences in functional annotation of the top and stable variants. The list of all external factors <inline-formula><alternatives><mml:math id="inf202"><mml:mi>F</mml:mi></mml:math><tex-math id="inft202">\begin{document}$F$\end{document}</tex-math></alternatives></inline-formula> is provided in <xref ref-type="table" rid="table2">Table 2</xref>.</p><p>For (1) Degree of Stability, we perform a Jonckheere–Terpstra test (R command <monospace>clinfun::jonckheere.test(…,nperm=5000)</monospace>). For (2) Population Diversity, (3) Population Differentiation, and (6) Degree of Certainty of Causality Using Residualization Approach, we perform a correlation test (R command <monospace>cor.test(…)</monospace>). For (4) Inclusion of Distal Subpopulations (Top) and (5) Inclusion of Distal Subpopulations (Stable), we perform an unpaired Wilcoxon test (R command <monospace>wilcox.test(…)</monospace>). Finally, for each moderator, we again compute one-directional p-values in either direction, before applying the BH procedure to all p-values across annotations and potential sets for that moderator only.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>Lionel Chentian Jin is affiliated with McKinsey &amp; Company. The author has no other competing interests to declare</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Software, Formal analysis, Investigation, Visualization, Methodology, Writing – original draft</p></fn><fn fn-type="con" id="con2"><p>Software, Visualization</p></fn><fn fn-type="con" id="con3"><p>Supervision, Writing – review and editing, Conceptualization, Investigation, Methodology</p></fn><fn fn-type="con" id="con4"><p>Supervision, Funding acquisition, Writing – review and editing, Conceptualization, Investigation, Methodology</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-88039-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>We provide the following data and scripts on the GitHub repository <ext-link ext-link-type="uri" xlink:href="https://github.com/songlab-cal/StableFM">https://github.com/songlab-cal/StableFM</ext-link> (copy archived at <xref ref-type="bibr" rid="bib5">Aw, 2026</xref>): fine-mapped variants with their functional annotations and moderator variable quantities, code for reproducing Main Text figures in this manuscript and building our visualization app.</p><p>The following previously published dataset was used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset1"><person-group person-group-type="author"><name><surname>Dermitzakis</surname><given-names>E</given-names></name><name><surname>Kurbatova</surname><given-names>N</given-names></name><name><surname>Lappalainen</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2013">2013</year><data-title>RNA-sequencing of 465 lymphoblastoid cell lines from the 1000 Genomes Project (GEUVADIS)</data-title><source>ArrayExpress</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/biostudies/arrayexpress/studies/E-GEUV-1">E-GEUV-1</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>This research is supported in part by an NIH grant R35-GM134922 and grant number CZF2019-</p><p>002449 from the Chan Zuckerberg Initiative Foundation. We thank Carlos Albors, Gonzalo Benegas,</p><p>Ryan Chung, Ziyue Gao, Iain Mathieson, and Yutong Wang for helpful discussions; members of the</p><p>Yu Group at Berkeley for feedback on the work; and Aniketh Reddy for help with data processing.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abdellaoui</surname><given-names>A</given-names></name><name><surname>Dolan</surname><given-names>CV</given-names></name><name><surname>Verweij</surname><given-names>KJH</given-names></name><name><surname>Nivard</surname><given-names>MG</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Gene-environment correlations across geographic regions affect genome-wide association studies</article-title><source>Nature Genetics</source><volume>54</volume><fpage>1345</fpage><lpage>1354</lpage><pub-id pub-id-type="doi">10.1038/s41588-022-01158-0</pub-id><pub-id pub-id-type="pmid">35995948</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Adzhubei</surname><given-names>IA</given-names></name><name><surname>Schmidt</surname><given-names>S</given-names></name><name><surname>Peshkin</surname><given-names>L</given-names></name><name><surname>Ramensky</surname><given-names>VE</given-names></name><name><surname>Gerasimova</surname><given-names>A</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Kondrashov</surname><given-names>AS</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A method and server for predicting damaging missense mutations</article-title><source>Nature Methods</source><volume>7</volume><fpage>248</fpage><lpage>249</lpage><pub-id pub-id-type="doi">10.1038/nmeth0410-248</pub-id><pub-id pub-id-type="pmid">20354512</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Garrison</surname><given-names>EP</given-names></name><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Korbel</surname><given-names>JO</given-names></name><name><surname>Marchini</surname><given-names>JL</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><name><surname>McVean</surname><given-names>GA</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name><collab>1000 Genomes Project Consortium</collab></person-group><year iso-8601-date="2015">2015</year><article-title>A global reference for human genetic variation</article-title><source>Nature</source><volume>526</volume><fpage>68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature15393</pub-id><pub-id pub-id-type="pmid">26432245</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Avsec</surname><given-names>Ž</given-names></name><name><surname>Agarwal</surname><given-names>V</given-names></name><name><surname>Visentin</surname><given-names>D</given-names></name><name><surname>Ledsam</surname><given-names>JR</given-names></name><name><surname>Grabska-Barwinska</surname><given-names>A</given-names></name><name><surname>Taylor</surname><given-names>KR</given-names></name><name><surname>Assael</surname><given-names>Y</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Kelley</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Effective gene expression prediction from sequence by integrating long-range interactions</article-title><source>Nature Methods</source><volume>18</volume><fpage>1196</fpage><lpage>1203</lpage><pub-id pub-id-type="doi">10.1038/s41592-021-01252-x</pub-id><pub-id pub-id-type="pmid">34608324</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Aw</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2026">2026</year><data-title>StableFM</data-title><version designator="swh:1:rev:49b7f17640d96e3148ad54981c847272a7898bea">swh:1:rev:49b7f17640d96e3148ad54981c847272a7898bea</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:8fd96eb5631490d35e00714f9e33185fc2b5c5c5;origin=https://github.com/songlab-cal/StableFM;visit=swh:1:snp:5df7c652f2f78013716ef00b2e017736b79cc89d;anchor=swh:1:rev:49b7f17640d96e3148ad54981c847272a7898bea">https://archive.softwareheritage.org/swh:1:dir:8fd96eb5631490d35e00714f9e33185fc2b5c5c5;origin=https://github.com/songlab-cal/StableFM;visit=swh:1:snp:5df7c652f2f78013716ef00b2e017736b79cc89d;anchor=swh:1:rev:49b7f17640d96e3148ad54981c847272a7898bea</ext-link></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Fu</surname><given-names>Y</given-names></name><name><surname>Pawashe</surname><given-names>M</given-names></name><name><surname>McGillivray</surname><given-names>P</given-names></name><name><surname>Jin</surname><given-names>M</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Karczewski</surname><given-names>KJ</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Using ALoFT to determine the impact of putative loss-of-function variants in protein-coding genes</article-title><source>Nature Communications</source><volume>8</volume><elocation-id>382</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-017-00443-5</pub-id><pub-id pub-id-type="pmid">28851873</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barbadilla-Martínez</surname><given-names>L</given-names></name><name><surname>Klaassen</surname><given-names>N</given-names></name><name><surname>van Steensel</surname><given-names>B</given-names></name><name><surname>de Ridder</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Predicting gene expression from DNA sequence using deep learning models</article-title><source>Nature Reviews. Genetics</source><volume>26</volume><fpage>666</fpage><lpage>680</lpage><pub-id pub-id-type="doi">10.1038/s41576-025-00841-2</pub-id><pub-id pub-id-type="pmid">40360798</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basu</surname><given-names>S</given-names></name><name><surname>Kumbier</surname><given-names>K</given-names></name><name><surname>Brown</surname><given-names>JB</given-names></name><name><surname>Yu</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Iterative random forests to discover predictive and stable high-order interactions</article-title><source>PNAS</source><volume>115</volume><fpage>1943</fpage><lpage>1948</lpage><pub-id pub-id-type="doi">10.1073/pnas.1711236115</pub-id><pub-id pub-id-type="pmid">29351989</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benegas</surname><given-names>G</given-names></name><name><surname>Ye</surname><given-names>C</given-names></name><name><surname>Albors</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>JC</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Genomic language models: opportunities and challenges</article-title><source>Trends in Genetics</source><volume>41</volume><fpage>286</fpage><lpage>302</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2024.11.013</pub-id><pub-id pub-id-type="pmid">39753409</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benner</surname><given-names>C</given-names></name><name><surname>Spencer</surname><given-names>CCA</given-names></name><name><surname>Havulinna</surname><given-names>AS</given-names></name><name><surname>Salomaa</surname><given-names>V</given-names></name><name><surname>Ripatti</surname><given-names>S</given-names></name><name><surname>Pirinen</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>FINEMAP: efficient variable selection using summary data from genome-wide association studies</article-title><source>Bioinformatics</source><volume>32</volume><fpage>1493</fpage><lpage>1501</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btw018</pub-id><pub-id pub-id-type="pmid">26773131</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bousquet</surname><given-names>O</given-names></name><name><surname>Elisseeff</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Stability and generalization</article-title><source>The Journal of Machine Learning Research</source><volume>2</volume><fpage>499</fpage><lpage>526</lpage></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>BC</given-names></name><name><surname>Bray</surname><given-names>NL</given-names></name><name><surname>Pachter</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Expression reflects population structure</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007841</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007841</pub-id><pub-id pub-id-type="pmid">30566439</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cavazos</surname><given-names>TB</given-names></name><name><surname>Witte</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Inclusion of variants discovered from diverse populations improves polygenic risk score transferability</article-title><source>HGG Advances</source><volume>2</volume><elocation-id>100017</elocation-id><pub-id pub-id-type="doi">10.1016/j.xhgg.2020.100017</pub-id><pub-id pub-id-type="pmid">33564748</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname><given-names>R</given-names></name><name><surname>Elzur</surname><given-names>RA</given-names></name><name><surname>Kanai</surname><given-names>M</given-names></name><name><surname>Ulirsch</surname><given-names>JC</given-names></name><name><surname>Weissbrod</surname><given-names>O</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Neale</surname><given-names>BM</given-names></name><name><surname>Fan</surname><given-names>Z</given-names></name><name><surname>Finucane</surname><given-names>HK</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Improving fine-mapping by modeling infinitesimal effects</article-title><source>Nature Genetics</source><volume>56</volume><fpage>162</fpage><lpage>169</lpage><pub-id pub-id-type="doi">10.1038/s41588-023-01597-3</pub-id><pub-id pub-id-type="pmid">38036779</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cunningham</surname><given-names>F</given-names></name><name><surname>Allen</surname><given-names>JE</given-names></name><name><surname>Allen</surname><given-names>J</given-names></name><name><surname>Alvarez-Jarreta</surname><given-names>J</given-names></name><name><surname>Amode</surname><given-names>MR</given-names></name><name><surname>Armean</surname><given-names>IM</given-names></name><name><surname>Austine-Orimoloye</surname><given-names>O</given-names></name><name><surname>Azov</surname><given-names>AG</given-names></name><name><surname>Barnes</surname><given-names>I</given-names></name><name><surname>Bennett</surname><given-names>R</given-names></name><name><surname>Berry</surname><given-names>A</given-names></name><name><surname>Bhai</surname><given-names>J</given-names></name><name><surname>Bignell</surname><given-names>A</given-names></name><name><surname>Billis</surname><given-names>K</given-names></name><name><surname>Boddu</surname><given-names>S</given-names></name><name><surname>Brooks</surname><given-names>L</given-names></name><name><surname>Charkhchi</surname><given-names>M</given-names></name><name><surname>Cummins</surname><given-names>C</given-names></name><name><surname>Da Rin Fioretto</surname><given-names>L</given-names></name><name><surname>Davidson</surname><given-names>C</given-names></name><name><surname>Dodiya</surname><given-names>K</given-names></name><name><surname>Donaldson</surname><given-names>S</given-names></name><name><surname>El Houdaigui</surname><given-names>B</given-names></name><name><surname>El Naboulsi</surname><given-names>T</given-names></name><name><surname>Fatima</surname><given-names>R</given-names></name><name><surname>Giron</surname><given-names>CG</given-names></name><name><surname>Genez</surname><given-names>T</given-names></name><name><surname>Martinez</surname><given-names>JG</given-names></name><name><surname>Guijarro-Clarke</surname><given-names>C</given-names></name><name><surname>Gymer</surname><given-names>A</given-names></name><name><surname>Hardy</surname><given-names>M</given-names></name><name><surname>Hollis</surname><given-names>Z</given-names></name><name><surname>Hourlier</surname><given-names>T</given-names></name><name><surname>Hunt</surname><given-names>T</given-names></name><name><surname>Juettemann</surname><given-names>T</given-names></name><name><surname>Kaikala</surname><given-names>V</given-names></name><name><surname>Kay</surname><given-names>M</given-names></name><name><surname>Lavidas</surname><given-names>I</given-names></name><name><surname>Le</surname><given-names>T</given-names></name><name><surname>Lemos</surname><given-names>D</given-names></name><name><surname>Marugán</surname><given-names>JC</given-names></name><name><surname>Mohanan</surname><given-names>S</given-names></name><name><surname>Mushtaq</surname><given-names>A</given-names></name><name><surname>Naven</surname><given-names>M</given-names></name><name><surname>Ogeh</surname><given-names>DN</given-names></name><name><surname>Parker</surname><given-names>A</given-names></name><name><surname>Parton</surname><given-names>A</given-names></name><name><surname>Perry</surname><given-names>M</given-names></name><name><surname>Piližota</surname><given-names>I</given-names></name><name><surname>Prosovetskaia</surname><given-names>I</given-names></name><name><surname>Sakthivel</surname><given-names>MP</given-names></name><name><surname>Salam</surname><given-names>AIA</given-names></name><name><surname>Schmitt</surname><given-names>BM</given-names></name><name><surname>Schuilenburg</surname><given-names>H</given-names></name><name><surname>Sheppard</surname><given-names>D</given-names></name><name><surname>Pérez-Silva</surname><given-names>JG</given-names></name><name><surname>Stark</surname><given-names>W</given-names></name><name><surname>Steed</surname><given-names>E</given-names></name><name><surname>Sutinen</surname><given-names>K</given-names></name><name><surname>Sukumaran</surname><given-names>R</given-names></name><name><surname>Sumathipala</surname><given-names>D</given-names></name><name><surname>Suner</surname><given-names>M-M</given-names></name><name><surname>Szpak</surname><given-names>M</given-names></name><name><surname>Thormann</surname><given-names>A</given-names></name><name><surname>Tricomi</surname><given-names>FF</given-names></name><name><surname>Urbina-Gómez</surname><given-names>D</given-names></name><name><surname>Veidenberg</surname><given-names>A</given-names></name><name><surname>Walsh</surname><given-names>TA</given-names></name><name><surname>Walts</surname><given-names>B</given-names></name><name><surname>Willhoft</surname><given-names>N</given-names></name><name><surname>Winterbottom</surname><given-names>A</given-names></name><name><surname>Wass</surname><given-names>E</given-names></name><name><surname>Chakiachvili</surname><given-names>M</given-names></name><name><surname>Flint</surname><given-names>B</given-names></name><name><surname>Frankish</surname><given-names>A</given-names></name><name><surname>Giorgetti</surname><given-names>S</given-names></name><name><surname>Haggerty</surname><given-names>L</given-names></name><name><surname>Hunt</surname><given-names>SE</given-names></name><name><surname>IIsley</surname><given-names>GR</given-names></name><name><surname>Loveland</surname><given-names>JE</given-names></name><name><surname>Martin</surname><given-names>FJ</given-names></name><name><surname>Moore</surname><given-names>B</given-names></name><name><surname>Mudge</surname><given-names>JM</given-names></name><name><surname>Muffato</surname><given-names>M</given-names></name><name><surname>Perry</surname><given-names>E</given-names></name><name><surname>Ruffier</surname><given-names>M</given-names></name><name><surname>Tate</surname><given-names>J</given-names></name><name><surname>Thybert</surname><given-names>D</given-names></name><name><surname>Trevanion</surname><given-names>SJ</given-names></name><name><surname>Dyer</surname><given-names>S</given-names></name><name><surname>Harrison</surname><given-names>PW</given-names></name><name><surname>Howe</surname><given-names>KL</given-names></name><name><surname>Yates</surname><given-names>AD</given-names></name><name><surname>Zerbino</surname><given-names>DR</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Ensembl 2022</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D988</fpage><lpage>D995</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1049</pub-id><pub-id pub-id-type="pmid">34791404</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davenport</surname><given-names>EE</given-names></name><name><surname>Amariuta</surname><given-names>T</given-names></name><name><surname>Gutierrez-Arcelus</surname><given-names>M</given-names></name><name><surname>Slowikowski</surname><given-names>K</given-names></name><name><surname>Westra</surname><given-names>H-J</given-names></name><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Shen</surname><given-names>C</given-names></name><name><surname>Rao</surname><given-names>DA</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Pearson</surname><given-names>S</given-names></name><name><surname>von Schack</surname><given-names>D</given-names></name><name><surname>Beebe</surname><given-names>JS</given-names></name><name><surname>Bing</surname><given-names>N</given-names></name><name><surname>John</surname><given-names>S</given-names></name><name><surname>Vincent</surname><given-names>MS</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Raychaudhuri</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Discovering in vivo cytokine-eQTL interactions from a lupus clinical trial</article-title><source>Genome Biology</source><volume>19</volume><elocation-id>168</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-018-1560-8</pub-id><pub-id pub-id-type="pmid">30340504</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davydov</surname><given-names>EV</given-names></name><name><surname>Goode</surname><given-names>DL</given-names></name><name><surname>Sirota</surname><given-names>M</given-names></name><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Sidow</surname><given-names>A</given-names></name><name><surname>Batzoglou</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Identifying a high fraction of the human genome to be under selective constraint using GERP++</article-title><source>PLOS Computational Biology</source><volume>6</volume><elocation-id>e1001025</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1001025</pub-id><pub-id pub-id-type="pmid">21152010</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Efron</surname><given-names>B</given-names></name><name><surname>Tibshirani</surname><given-names>RJ</given-names></name></person-group><year iso-8601-date="1994">1994</year><source>An Introduction to the Bootstrap</source><publisher-name>CRC press</publisher-name></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farh</surname><given-names>KK-H</given-names></name><name><surname>Marson</surname><given-names>A</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Kleinewietfeld</surname><given-names>M</given-names></name><name><surname>Housley</surname><given-names>WJ</given-names></name><name><surname>Beik</surname><given-names>S</given-names></name><name><surname>Shoresh</surname><given-names>N</given-names></name><name><surname>Whitton</surname><given-names>H</given-names></name><name><surname>Ryan</surname><given-names>RJH</given-names></name><name><surname>Shishkin</surname><given-names>AA</given-names></name><name><surname>Hatan</surname><given-names>M</given-names></name><name><surname>Carrasco-Alfonso</surname><given-names>MJ</given-names></name><name><surname>Mayer</surname><given-names>D</given-names></name><name><surname>Luckey</surname><given-names>CJ</given-names></name><name><surname>Patsopoulos</surname><given-names>NA</given-names></name><name><surname>De Jager</surname><given-names>PL</given-names></name><name><surname>Kuchroo</surname><given-names>VK</given-names></name><name><surname>Epstein</surname><given-names>CB</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Hafler</surname><given-names>DA</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Genetic and epigenetic fine mapping of causal autoimmune disease variants</article-title><source>Nature</source><volume>518</volume><fpage>337</fpage><lpage>343</lpage><pub-id pub-id-type="doi">10.1038/nature13835</pub-id><pub-id pub-id-type="pmid">25363779</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Favé</surname><given-names>M-J</given-names></name><name><surname>Lamaze</surname><given-names>FC</given-names></name><name><surname>Soave</surname><given-names>D</given-names></name><name><surname>Hodgkinson</surname><given-names>A</given-names></name><name><surname>Gauvin</surname><given-names>H</given-names></name><name><surname>Bruat</surname><given-names>V</given-names></name><name><surname>Grenier</surname><given-names>J-C</given-names></name><name><surname>Gbeha</surname><given-names>E</given-names></name><name><surname>Skead</surname><given-names>K</given-names></name><name><surname>Smargiassi</surname><given-names>A</given-names></name><name><surname>Johnson</surname><given-names>M</given-names></name><name><surname>Idaghdour</surname><given-names>Y</given-names></name><name><surname>Awadalla</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Gene-by-environment interactions in urban populations modulate risk phenotypes</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>827</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-03202-2</pub-id><pub-id pub-id-type="pmid">29511166</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>Y</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Lou</surname><given-names>S</given-names></name><name><surname>Bedford</surname><given-names>J</given-names></name><name><surname>Mu</surname><given-names>XJ</given-names></name><name><surname>Yip</surname><given-names>KY</given-names></name><name><surname>Khurana</surname><given-names>E</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>FunSeq2: a framework for prioritizing noncoding regulatory variants in cancer</article-title><source>Genome Biology</source><volume>15</volume><elocation-id>480</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-014-0480-5</pub-id><pub-id pub-id-type="pmid">25273974</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>B</given-names></name><name><surname>Zhou</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>MESuSiE enables scalable and powerful multi-ancestry fine-mapping of causal variants in genome-wide association studies</article-title><source>Nature Genetics</source><volume>56</volume><fpage>170</fpage><lpage>179</lpage><pub-id pub-id-type="doi">10.1038/s41588-023-01604-7</pub-id><pub-id pub-id-type="pmid">38168930</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gong</surname><given-names>J</given-names></name><name><surname>Mei</surname><given-names>S</given-names></name><name><surname>Liu</surname><given-names>C</given-names></name><name><surname>Xiang</surname><given-names>Y</given-names></name><name><surname>Ye</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Feng</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>R</given-names></name><name><surname>Diao</surname><given-names>L</given-names></name><name><surname>Guo</surname><given-names>A-Y</given-names></name><name><surname>Miao</surname><given-names>X</given-names></name><name><surname>Han</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>PancanQTL: systematic identification of cis-eQTLs and trans-eQTLs in 33 cancer types</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>D971</fpage><lpage>D976</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx861</pub-id><pub-id pub-id-type="pmid">29036324</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Griffin</surname><given-names>JE</given-names></name><name><surname>Steel</surname><given-names>MF</given-names></name></person-group><year iso-8601-date="2021">2021</year><source>Adaptive Computational Methods for Bayesian Variable Selection, Handbook of Bayesian Variable Selection</source><publisher-name>Chapman and Hall/CRC</publisher-name><pub-id pub-id-type="doi">10.1201/9781003089018-5</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Han</surname><given-names>B</given-names></name><name><surname>Eskin</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Interpreting meta-analyses of genome-wide association studies</article-title><source>PLOS Genetics</source><volume>8</volume><elocation-id>e1002555</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1002555</pub-id><pub-id pub-id-type="pmid">22396665</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hormozdiari</surname><given-names>F</given-names></name><name><surname>Kichaev</surname><given-names>G</given-names></name><name><surname>Yang</surname><given-names>WY</given-names></name><name><surname>Pasaniuc</surname><given-names>B</given-names></name><name><surname>Eskin</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Identification of causal genes for complex traits</article-title><source>Bioinformatics</source><volume>31</volume><fpage>i206</fpage><lpage>i213</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv240</pub-id><pub-id pub-id-type="pmid">26072484</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname><given-names>S</given-names></name><name><surname>Ferreira</surname><given-names>LAF</given-names></name><name><surname>Shi</surname><given-names>S</given-names></name><name><surname>Hellenthal</surname><given-names>G</given-names></name><name><surname>Marchini</surname><given-names>J</given-names></name><name><surname>Lawson</surname><given-names>DJ</given-names></name><name><surname>Myers</surname><given-names>SR</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Fine-scale population structure and widespread conservation of genetic effect sizes between human groups across traits</article-title><source>Nature Genetics</source><volume>57</volume><fpage>379</fpage><lpage>389</lpage><pub-id pub-id-type="doi">10.1038/s41588-024-02035-8</pub-id><pub-id pub-id-type="pmid">39901012</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>YF</given-names></name><name><surname>Gulko</surname><given-names>B</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Fast, scalable prediction of deleterious noncoding variants from functional and population genomic data</article-title><source>Nature Genetics</source><volume>49</volume><fpage>618</fpage><lpage>624</lpage><pub-id pub-id-type="doi">10.1038/ng.3810</pub-id><pub-id pub-id-type="pmid">28288115</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>C</given-names></name><name><surname>Shuai</surname><given-names>RW</given-names></name><name><surname>Baokar</surname><given-names>P</given-names></name><name><surname>Chung</surname><given-names>R</given-names></name><name><surname>Rastogi</surname><given-names>R</given-names></name><name><surname>Kathail</surname><given-names>P</given-names></name><name><surname>Ioannidis</surname><given-names>NM</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Personal transcriptome variation is poorly explained by current genomic deep learning models</article-title><source>Nature Genetics</source><volume>55</volume><fpage>2056</fpage><lpage>2059</lpage><pub-id pub-id-type="doi">10.1038/s41588-023-01574-w</pub-id><pub-id pub-id-type="pmid">38036790</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ioannidis</surname><given-names>NM</given-names></name><name><surname>Davis</surname><given-names>JR</given-names></name><name><surname>DeGorter</surname><given-names>MK</given-names></name><name><surname>Larson</surname><given-names>NB</given-names></name><name><surname>McDonnell</surname><given-names>SK</given-names></name><name><surname>French</surname><given-names>AJ</given-names></name><name><surname>Battle</surname><given-names>AJ</given-names></name><name><surname>Hastie</surname><given-names>TJ</given-names></name><name><surname>Thibodeau</surname><given-names>SN</given-names></name><name><surname>Montgomery</surname><given-names>SB</given-names></name><name><surname>Bustamante</surname><given-names>CD</given-names></name><name><surname>Sieh</surname><given-names>W</given-names></name><name><surname>Whittemore</surname><given-names>AS</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>FIRE: functional inference of genetic variants that regulate gene expression</article-title><source>Bioinformatics</source><volume>33</volume><fpage>3895</fpage><lpage>3901</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btx534</pub-id><pub-id pub-id-type="pmid">28961785</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karollus</surname><given-names>A</given-names></name><name><surname>Mauermeier</surname><given-names>T</given-names></name><name><surname>Gagneur</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Current sequence-based models capture gene expression determinants in promoters but mostly ignore distal enhancers</article-title><source>Genome Biology</source><volume>24</volume><elocation-id>56</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-023-02899-9</pub-id><pub-id pub-id-type="pmid">36973806</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Katsonis</surname><given-names>P</given-names></name><name><surname>Wilhelm</surname><given-names>K</given-names></name><name><surname>Williams</surname><given-names>A</given-names></name><name><surname>Lichtarge</surname><given-names>O</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Genome interpretation using in silico predictors of variant impact</article-title><source>Human Genetics</source><volume>141</volume><fpage>1549</fpage><lpage>1577</lpage><pub-id pub-id-type="doi">10.1007/s00439-022-02457-6</pub-id><pub-id pub-id-type="pmid">35488922</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Keys</surname><given-names>KL</given-names></name><name><surname>Mak</surname><given-names>ACY</given-names></name><name><surname>White</surname><given-names>MJ</given-names></name><name><surname>Eckalbar</surname><given-names>WL</given-names></name><name><surname>Dahl</surname><given-names>AW</given-names></name><name><surname>Mefford</surname><given-names>J</given-names></name><name><surname>Mikhaylova</surname><given-names>AV</given-names></name><name><surname>Contreras</surname><given-names>MG</given-names></name><name><surname>Elhawary</surname><given-names>JR</given-names></name><name><surname>Eng</surname><given-names>C</given-names></name><name><surname>Hu</surname><given-names>D</given-names></name><name><surname>Huntsman</surname><given-names>S</given-names></name><name><surname>Oh</surname><given-names>SS</given-names></name><name><surname>Salazar</surname><given-names>S</given-names></name><name><surname>Lenoir</surname><given-names>MA</given-names></name><name><surname>Ye</surname><given-names>JC</given-names></name><name><surname>Thornton</surname><given-names>TA</given-names></name><name><surname>Zaitlen</surname><given-names>N</given-names></name><name><surname>Burchard</surname><given-names>EG</given-names></name><name><surname>Gignoux</surname><given-names>CR</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>On the cross-population generalizability of gene expression prediction models</article-title><source>PLOS Genetics</source><volume>16</volume><elocation-id>e1008927</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1008927</pub-id><pub-id pub-id-type="pmid">32797036</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kheradpour</surname><given-names>P</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Systematic discovery and characterization of regulatory motifs in ENCODE TF binding experiments</article-title><source>Nucleic Acids Research</source><volume>42</volume><fpage>2976</fpage><lpage>2987</lpage><pub-id pub-id-type="doi">10.1093/nar/gkt1249</pub-id><pub-id pub-id-type="pmid">24335146</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>LaPierre</surname><given-names>N</given-names></name><name><surname>Taraszka</surname><given-names>K</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name><name><surname>He</surname><given-names>R</given-names></name><name><surname>Hormozdiari</surname><given-names>F</given-names></name><name><surname>Eskin</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Identifying causal variants by fine mapping across multiple studies</article-title><source>PLOS Genetics</source><volume>17</volume><elocation-id>e1009733</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1009733</pub-id><pub-id pub-id-type="pmid">34543273</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lappalainen</surname><given-names>T</given-names></name><name><surname>Sammeth</surname><given-names>M</given-names></name><name><surname>Friedländer</surname><given-names>MR</given-names></name><name><surname>’t Hoen</surname><given-names>PAC</given-names></name><name><surname>Monlong</surname><given-names>J</given-names></name><name><surname>Rivas</surname><given-names>MA</given-names></name><name><surname>Gonzàlez-Porta</surname><given-names>M</given-names></name><name><surname>Kurbatova</surname><given-names>N</given-names></name><name><surname>Griebel</surname><given-names>T</given-names></name><name><surname>Ferreira</surname><given-names>PG</given-names></name><name><surname>Barann</surname><given-names>M</given-names></name><name><surname>Wieland</surname><given-names>T</given-names></name><name><surname>Greger</surname><given-names>L</given-names></name><name><surname>van Iterson</surname><given-names>M</given-names></name><name><surname>Almlöf</surname><given-names>J</given-names></name><name><surname>Ribeca</surname><given-names>P</given-names></name><name><surname>Pulyakhina</surname><given-names>I</given-names></name><name><surname>Esser</surname><given-names>D</given-names></name><name><surname>Giger</surname><given-names>T</given-names></name><name><surname>Tikhonov</surname><given-names>A</given-names></name><name><surname>Sultan</surname><given-names>M</given-names></name><name><surname>Bertier</surname><given-names>G</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>Lek</surname><given-names>M</given-names></name><name><surname>Lizano</surname><given-names>E</given-names></name><name><surname>Buermans</surname><given-names>HPJ</given-names></name><name><surname>Padioleau</surname><given-names>I</given-names></name><name><surname>Schwarzmayr</surname><given-names>T</given-names></name><name><surname>Karlberg</surname><given-names>O</given-names></name><name><surname>Ongen</surname><given-names>H</given-names></name><name><surname>Kilpinen</surname><given-names>H</given-names></name><name><surname>Beltran</surname><given-names>S</given-names></name><name><surname>Gut</surname><given-names>M</given-names></name><name><surname>Kahlem</surname><given-names>K</given-names></name><name><surname>Amstislavskiy</surname><given-names>V</given-names></name><name><surname>Stegle</surname><given-names>O</given-names></name><name><surname>Pirinen</surname><given-names>M</given-names></name><name><surname>Montgomery</surname><given-names>SB</given-names></name><name><surname>Donnelly</surname><given-names>P</given-names></name><name><surname>McCarthy</surname><given-names>MI</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Strom</surname><given-names>TM</given-names></name><collab>Geuvadis Consortium</collab><name><surname>Lehrach</surname><given-names>H</given-names></name><name><surname>Schreiber</surname><given-names>S</given-names></name><name><surname>Sudbrak</surname><given-names>R</given-names></name><name><surname>Carracedo</surname><given-names>A</given-names></name><name><surname>Antonarakis</surname><given-names>SE</given-names></name><name><surname>Häsler</surname><given-names>R</given-names></name><name><surname>Syvänen</surname><given-names>A-C</given-names></name><name><surname>van Ommen</surname><given-names>G-J</given-names></name><name><surname>Brazma</surname><given-names>A</given-names></name><name><surname>Meitinger</surname><given-names>T</given-names></name><name><surname>Rosenstiel</surname><given-names>P</given-names></name><name><surname>Guigó</surname><given-names>R</given-names></name><name><surname>Gut</surname><given-names>IG</given-names></name><name><surname>Estivill</surname><given-names>X</given-names></name><name><surname>Dermitzakis</surname><given-names>ET</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Transcriptome and genome sequencing uncovers functional variation in humans</article-title><source>Nature</source><volume>501</volume><fpage>506</fpage><lpage>511</lpage><pub-id pub-id-type="doi">10.1038/nature12531</pub-id><pub-id pub-id-type="pmid">24037378</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Sesia</surname><given-names>M</given-names></name><name><surname>Romano</surname><given-names>Y</given-names></name><name><surname>Candès</surname><given-names>E</given-names></name><name><surname>Sabatti</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Searching for robust associations with a multi-environment knockoff filter</article-title><source>Biometrika</source><volume>109</volume><fpage>611</fpage><lpage>629</lpage><pub-id pub-id-type="doi">10.1093/biomet/asab055</pub-id><pub-id pub-id-type="pmid">38633763</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname><given-names>Y</given-names></name><name><surname>Pividori</surname><given-names>M</given-names></name><name><surname>Manichaikul</surname><given-names>A</given-names></name><name><surname>Palmer</surname><given-names>AA</given-names></name><name><surname>Cox</surname><given-names>NJ</given-names></name><name><surname>Wheeler</surname><given-names>HE</given-names></name><name><surname>Im</surname><given-names>HK</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Polygenic transcriptome risk scores (PTRS) can improve portability of polygenic risk scores across ancestries</article-title><source>Genome Biology</source><volume>23</volume><elocation-id>23</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-021-02591-w</pub-id><pub-id pub-id-type="pmid">35027082</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lim</surname><given-names>C</given-names></name><name><surname>Yu</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Estimation stability with cross-validation (ESCV)</article-title><source>Journal of Computational and Graphical Statistics</source><volume>25</volume><fpage>464</fpage><lpage>492</lpage><pub-id pub-id-type="doi">10.1080/10618600.2015.1020159</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Livesey</surname><given-names>BJ</given-names></name><name><surname>Badonyi</surname><given-names>M</given-names></name><name><surname>Dias</surname><given-names>M</given-names></name><name><surname>Frazer</surname><given-names>J</given-names></name><name><surname>Kumar</surname><given-names>S</given-names></name><name><surname>Lindorff-Larsen</surname><given-names>K</given-names></name><name><surname>McCandlish</surname><given-names>DM</given-names></name><name><surname>Orenbuch</surname><given-names>R</given-names></name><name><surname>Shearer</surname><given-names>CA</given-names></name><name><surname>Muffley</surname><given-names>L</given-names></name><name><surname>Foreman</surname><given-names>J</given-names></name><name><surname>Glazer</surname><given-names>AM</given-names></name><name><surname>Lehner</surname><given-names>B</given-names></name><name><surname>Marks</surname><given-names>DS</given-names></name><name><surname>Roth</surname><given-names>FP</given-names></name><name><surname>Rubin</surname><given-names>AF</given-names></name><name><surname>Starita</surname><given-names>LM</given-names></name><name><surname>Marsh</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Guidelines for releasing a variant effect predictor</article-title><source>Genome Biology</source><volume>26</volume><elocation-id>97</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-025-03572-z</pub-id><pub-id pub-id-type="pmid">40234898</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname><given-names>Z</given-names></name><name><surname>Gopalan</surname><given-names>S</given-names></name><name><surname>Yuan</surname><given-names>D</given-names></name><name><surname>Conti</surname><given-names>DV</given-names></name><name><surname>Pasaniuc</surname><given-names>B</given-names></name><name><surname>Gusev</surname><given-names>A</given-names></name><name><surname>Mancuso</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Multi-ancestry fine-mapping improves precision to identify causal genes in transcriptome-wide association studies</article-title><source>American Journal of Human Genetics</source><volume>109</volume><fpage>1388</fpage><lpage>1404</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2022.07.002</pub-id><pub-id pub-id-type="pmid">35931050</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Carr</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>A</given-names></name><name><surname>Gazal</surname><given-names>S</given-names></name><name><surname>Mohammadi</surname><given-names>P</given-names></name><name><surname>Wu</surname><given-names>L</given-names></name><name><surname>Pirruccello</surname><given-names>J</given-names></name><name><surname>Kachuri</surname><given-names>L</given-names></name><name><surname>Gusev</surname><given-names>A</given-names></name><name><surname>Mancuso</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2025">2025</year><article-title>Improved multiancestry fine-mapping identifies cis-regulatory variants underlying molecular traits and disease risk</article-title><source>Nature Genetics</source><volume>57</volume><fpage>1881</fpage><lpage>1889</lpage><pub-id pub-id-type="doi">10.1038/s41588-025-02262-7</pub-id><pub-id pub-id-type="pmid">40691406</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Márquez-Luna</surname><given-names>C</given-names></name><name><surname>Loh</surname><given-names>PR</given-names></name><name><surname>Price</surname><given-names>AL</given-names></name><collab>Consortium SATDS</collab><collab>Consortium STD</collab></person-group><year iso-8601-date="2017">2017</year><article-title>Multiethnic polygenic risk scores improve risk prediction in diverse populations</article-title><source>Genetic Epidemiology</source><volume>41</volume><fpage>811</fpage><lpage>823</lpage><pub-id pub-id-type="doi">10.1002/gepi.22083</pub-id><pub-id pub-id-type="pmid">29110330</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mazumder</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Discussion of “best subset, forward stepwise or lasso? analysis and recommendations based on extensive comparisons”</article-title><source>Statistical Science</source><volume>35</volume><fpage>602</fpage><lpage>608</lpage><pub-id pub-id-type="doi">10.1214/20-STS807</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVicker</surname><given-names>G</given-names></name><name><surname>Gordon</surname><given-names>D</given-names></name><name><surname>Davis</surname><given-names>C</given-names></name><name><surname>Green</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Widespread genomic signatures of natural selection in hominid evolution</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000471</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000471</pub-id><pub-id pub-id-type="pmid">19424416</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Morris</surname><given-names>AP</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Transethnic meta-analysis of genomewide association studies</article-title><source>Genetic Epidemiology</source><volume>35</volume><fpage>809</fpage><lpage>822</lpage><pub-id pub-id-type="doi">10.1002/gepi.20630</pub-id><pub-id pub-id-type="pmid">22125221</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mostafavi</surname><given-names>H</given-names></name><name><surname>Harpak</surname><given-names>A</given-names></name><name><surname>Agarwal</surname><given-names>I</given-names></name><name><surname>Conley</surname><given-names>D</given-names></name><name><surname>Pritchard</surname><given-names>JK</given-names></name><name><surname>Przeworski</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Variable prediction accuracy of polygenic scores within an ancestry group</article-title><source>eLife</source><volume>9</volume><elocation-id>e48376</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.48376</pub-id><pub-id pub-id-type="pmid">31999256</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ng</surname><given-names>PC</given-names></name><name><surname>Henikoff</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>SIFT: Predicting amino acid changes that affect protein function</article-title><source>Nucleic Acids Research</source><volume>31</volume><fpage>3812</fpage><lpage>3814</lpage><pub-id pub-id-type="doi">10.1093/nar/gkg509</pub-id><pub-id pub-id-type="pmid">12824425</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pollard</surname><given-names>KS</given-names></name><name><surname>Hubisz</surname><given-names>MJ</given-names></name><name><surname>Rosenbloom</surname><given-names>KR</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Detection of nonneutral substitution rates on mammalian phylogenies</article-title><source>Genome Research</source><volume>20</volume><fpage>110</fpage><lpage>121</lpage><pub-id pub-id-type="doi">10.1101/gr.097857.109</pub-id><pub-id pub-id-type="pmid">19858363</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rentzsch</surname><given-names>P</given-names></name><name><surname>Witten</surname><given-names>D</given-names></name><name><surname>Cooper</surname><given-names>GM</given-names></name><name><surname>Shendure</surname><given-names>J</given-names></name><name><surname>Kircher</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>CADD: predicting the deleteriousness of variants throughout the human genome</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D886</fpage><lpage>D894</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1016</pub-id><pub-id pub-id-type="pmid">30371827</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rogers</surname><given-names>MF</given-names></name><name><surname>Shihab</surname><given-names>HA</given-names></name><name><surname>Mort</surname><given-names>M</given-names></name><name><surname>Cooper</surname><given-names>DN</given-names></name><name><surname>Gaunt</surname><given-names>TR</given-names></name><name><surname>Campbell</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>FATHMM-XF: accurate prediction of pathogenic point mutations via extended features</article-title><source>Bioinformatics</source><volume>34</volume><fpage>511</fpage><lpage>513</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btx536</pub-id><pub-id pub-id-type="pmid">28968714</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sasse</surname><given-names>A</given-names></name><name><surname>Ng</surname><given-names>B</given-names></name><name><surname>Spiro</surname><given-names>AE</given-names></name><name><surname>Tasaki</surname><given-names>S</given-names></name><name><surname>Bennett</surname><given-names>DA</given-names></name><name><surname>Gaiteri</surname><given-names>C</given-names></name><name><surname>De Jager</surname><given-names>PL</given-names></name><name><surname>Chikina</surname><given-names>M</given-names></name><name><surname>Mostafavi</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Benchmarking of deep neural networks for predicting personal gene expression from DNA sequence highlights shortcomings</article-title><source>Nature Genetics</source><volume>55</volume><fpage>2060</fpage><lpage>2064</lpage><pub-id pub-id-type="doi">10.1038/s41588-023-01524-6</pub-id><pub-id pub-id-type="pmid">38036778</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schaid</surname><given-names>DJ</given-names></name><name><surname>Chen</surname><given-names>W</given-names></name><name><surname>Larson</surname><given-names>NB</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>From genome-wide associations to candidate causal variants by statistical fine-mapping</article-title><source>Nature Reviews. Genetics</source><volume>19</volume><fpage>491</fpage><lpage>504</lpage><pub-id pub-id-type="doi">10.1038/s41576-018-0016-z</pub-id><pub-id pub-id-type="pmid">29844615</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname><given-names>H</given-names></name><name><surname>Gazal</surname><given-names>S</given-names></name><name><surname>Kanai</surname><given-names>M</given-names></name><name><surname>Koch</surname><given-names>EM</given-names></name><name><surname>Schoech</surname><given-names>AP</given-names></name><name><surname>Siewert</surname><given-names>KM</given-names></name><name><surname>Kim</surname><given-names>SS</given-names></name><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Amariuta</surname><given-names>T</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name><name><surname>Okada</surname><given-names>Y</given-names></name><name><surname>Raychaudhuri</surname><given-names>S</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name><name><surname>Price</surname><given-names>AL</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Population-specific causal disease effect sizes in functionally important regions impacted by selection</article-title><source>Nature Communications</source><volume>12</volume><elocation-id>1098</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-021-21286-1</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Taylor</surname><given-names>KE</given-names></name><name><surname>Ansel</surname><given-names>KM</given-names></name><name><surname>Marson</surname><given-names>A</given-names></name><name><surname>Criswell</surname><given-names>LA</given-names></name><name><surname>Farh</surname><given-names>KKH</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>PICS2: next-generation fine mapping via probabilistic identification of causal SNPs</article-title><source>Bioinformatics</source><volume>37</volume><fpage>3004</fpage><lpage>3007</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btab122</pub-id><pub-id pub-id-type="pmid">33624747</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Turley</surname><given-names>P</given-names></name><name><surname>Martin</surname><given-names>AR</given-names></name><name><surname>Goldman</surname><given-names>G</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Kanai</surname><given-names>M</given-names></name><name><surname>Walters</surname><given-names>RK</given-names></name><name><surname>Jala</surname><given-names>JB</given-names></name><name><surname>Lin</surname><given-names>K</given-names></name><name><surname>Millwood</surname><given-names>IY</given-names></name><name><surname>Carey</surname><given-names>CE</given-names></name><name><surname>Palmer</surname><given-names>DS</given-names></name><name><surname>Zacher</surname><given-names>M</given-names></name><name><surname>Atkinson</surname><given-names>EG</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Akiyama</surname><given-names>M</given-names></name><name><surname>Okada</surname><given-names>Y</given-names></name><name><surname>Kamatani</surname><given-names>Y</given-names></name><name><surname>Walters</surname><given-names>RG</given-names></name><name><surname>Callier</surname><given-names>S</given-names></name><name><surname>Laibson</surname><given-names>D</given-names></name><name><surname>Meyer</surname><given-names>MN</given-names></name><name><surname>Cesarini</surname><given-names>D</given-names></name><name><surname>Daly</surname><given-names>M</given-names></name><name><surname>Benjamin</surname><given-names>DJ</given-names></name><name><surname>Neale</surname><given-names>BM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Multi-ancestry meta-analysis yields novel genetic discoveries and ancestry-specific associations</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.04.23.441003</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>G</given-names></name><name><surname>Sarkar</surname><given-names>A</given-names></name><name><surname>Carbonetto</surname><given-names>P</given-names></name><name><surname>Stephens</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A simple new approach to variable selection in regression, with application to genetic fine mapping</article-title><source>Journal of the Royal Statistical Society. Series B, Statistical Methodology</source><volume>82</volume><fpage>1273</fpage><lpage>1300</lpage><pub-id pub-id-type="doi">10.1111/rssb.12388</pub-id><pub-id pub-id-type="pmid">37220626</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>QS</given-names></name><name><surname>Kelley</surname><given-names>DR</given-names></name><name><surname>Ulirsch</surname><given-names>J</given-names></name><name><surname>Kanai</surname><given-names>M</given-names></name><name><surname>Sadhuka</surname><given-names>S</given-names></name><name><surname>Cui</surname><given-names>R</given-names></name><name><surname>Albors</surname><given-names>C</given-names></name><name><surname>Cheng</surname><given-names>N</given-names></name><name><surname>Okada</surname><given-names>Y</given-names></name><name><surname>Aguet</surname><given-names>F</given-names></name><name><surname>Ardlie</surname><given-names>KG</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name><name><surname>Finucane</surname><given-names>HK</given-names></name><collab>Biobank Japan Project</collab></person-group><year iso-8601-date="2021">2021</year><article-title>Leveraging supervised learning for functionally informed fine-mapping of cis-eQTLs identifies an additional 20,913 putative causal eQTLs</article-title><source>Nature Communications</source><volume>12</volume><elocation-id>3394</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-021-23134-8</pub-id><pub-id pub-id-type="pmid">34099641</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname><given-names>X</given-names></name><name><surname>Luca</surname><given-names>F</given-names></name><name><surname>Pique-Regi</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Cross-population joint analysis of eQTLs: fine mapping and functional annotation</article-title><source>PLOS Genetics</source><volume>11</volume><elocation-id>e1005176</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1005176</pub-id><pub-id pub-id-type="pmid">25906321</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname><given-names>X</given-names></name><name><surname>Lee</surname><given-names>Y</given-names></name><name><surname>Luca</surname><given-names>F</given-names></name><name><surname>Pique-Regi</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Efficient integrative multi-snp association analysis via deterministic approximation of posteriors</article-title><source>American Journal of Human Genetics</source><volume>98</volume><fpage>1114</fpage><lpage>1129</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2016.03.029</pub-id><pub-id pub-id-type="pmid">27236919</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Willer</surname><given-names>CJ</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Abecasis</surname><given-names>GR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>METAL: fast and efficient meta-analysis of genomewide association scans</article-title><source>Bioinformatics</source><volume>26</volume><fpage>2190</fpage><lpage>2191</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq340</pub-id><pub-id pub-id-type="pmid">20616382</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>C</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Khan</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>A</given-names></name><name><surname>Vardarajan</surname><given-names>B</given-names></name><name><surname>Mayeux</surname><given-names>R</given-names></name><name><surname>Kiryluk</surname><given-names>K</given-names></name><name><surname>Ionita-Laza</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>CARMA is a new Bayesian model for fine-mapping in genome-wide association meta-analyses</article-title><source>Nature Genetics</source><volume>55</volume><fpage>1057</fpage><lpage>1065</lpage><pub-id pub-id-type="doi">10.1038/s41588-023-01392-0</pub-id><pub-id pub-id-type="pmid">37169873</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>B</given-names></name><name><surname>Kumbier</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Veridical data science</article-title><source>PNAS</source><volume>117</volume><fpage>3920</fpage><lpage>3929</lpage><pub-id pub-id-type="doi">10.1073/pnas.1901326117</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname><given-names>K</given-names></name><name><surname>Longchamps</surname><given-names>RJ</given-names></name><name><surname>Pardiñas</surname><given-names>AF</given-names></name><name><surname>Yu</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>T-T</given-names></name><name><surname>Lin</surname><given-names>S-C</given-names></name><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Lam</surname><given-names>M</given-names></name><name><surname>Liu</surname><given-names>R</given-names></name><name><surname>Xia</surname><given-names>Y</given-names></name><name><surname>Guo</surname><given-names>Z</given-names></name><name><surname>Shi</surname><given-names>W</given-names></name><name><surname>Shen</surname><given-names>C</given-names></name><collab>Schizophrenia Workgroup of Psychiatric Genomics Consortium</collab><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>Neale</surname><given-names>BM</given-names></name><name><surname>Feng</surname><given-names>Y-CA</given-names></name><name><surname>Lin</surname><given-names>Y-F</given-names></name><name><surname>Chen</surname><given-names>C-Y</given-names></name><name><surname>O’Donovan</surname><given-names>MC</given-names></name><name><surname>Ge</surname><given-names>T</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Fine-mapping across diverse ancestries drives the discovery of putative causal variants underlying human complex traits and diseases</article-title><source>Nature Genetics</source><volume>56</volume><fpage>1841</fpage><lpage>1850</lpage><pub-id pub-id-type="doi">10.1038/s41588-024-01870-z</pub-id><pub-id pub-id-type="pmid">39187616</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zaidi</surname><given-names>AA</given-names></name><name><surname>Mathieson</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Demographic history mediates the effect of stratification on polygenic scores</article-title><source>eLife</source><volume>9</volume><elocation-id>e61548</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.61548</pub-id><pub-id pub-id-type="pmid">33200985</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname><given-names>H</given-names></name><name><surname>Arapoglou</surname><given-names>T</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name><name><surname>Zheng</surname><given-names>X</given-names></name><name><surname>Moore</surname><given-names>J</given-names></name><name><surname>Asok</surname><given-names>A</given-names></name><name><surname>Kumar</surname><given-names>S</given-names></name><name><surname>Blue</surname><given-names>EE</given-names></name><name><surname>Buyske</surname><given-names>S</given-names></name><name><surname>Cox</surname><given-names>N</given-names></name><name><surname>Felsenfeld</surname><given-names>A</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name><name><surname>Kenny</surname><given-names>E</given-names></name><name><surname>Li</surname><given-names>B</given-names></name><name><surname>Matise</surname><given-names>T</given-names></name><name><surname>Philippakis</surname><given-names>A</given-names></name><name><surname>Rehm</surname><given-names>HL</given-names></name><name><surname>Sofia</surname><given-names>HJ</given-names></name><name><surname>Snyder</surname><given-names>G</given-names></name><collab>NHGRI Genome Sequencing Program Variant Functional Annotation Working Group</collab><name><surname>Weng</surname><given-names>Z</given-names></name><name><surname>Neale</surname><given-names>B</given-names></name><name><surname>Sunyaev</surname><given-names>SR</given-names></name><name><surname>Lin</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>FAVOR: functional annotation of variants online resource and annotator for variation across the human genome</article-title><source>Nucleic Acids Research</source><volume>51</volume><fpage>D1300</fpage><lpage>D1311</lpage><pub-id pub-id-type="doi">10.1093/nar/gkac966</pub-id><pub-id pub-id-type="pmid">36350676</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s8"><title>Permuting while preserving marginal correlation and LD</title><p>For a focal SNP <inline-formula><alternatives><mml:math id="inf203"><mml:mi>i</mml:mi></mml:math><tex-math id="inft203">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula>, its <inline-formula><alternatives><mml:math id="inf204"><mml:mi>N</mml:mi></mml:math><tex-math id="inft204">\begin{document}$N$\end{document}</tex-math></alternatives></inline-formula> entries <inline-formula><alternatives><mml:math id="inf205"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft205">\begin{document}$(x_{1i},\ldots,x_{Ni})$\end{document}</tex-math></alternatives></inline-formula> take on only <inline-formula><alternatives><mml:math id="inf206"><mml:mi>D</mml:mi></mml:math><tex-math id="inft206">\begin{document}$D$\end{document}</tex-math></alternatives></inline-formula> values, where <inline-formula><alternatives><mml:math id="inf207"><mml:mi>D</mml:mi></mml:math><tex-math id="inft207">\begin{document}$D$\end{document}</tex-math></alternatives></inline-formula> is the ploidy of the data (for human genotypes <inline-formula><alternatives><mml:math id="inf208"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math><tex-math id="inft208">\begin{document}$D=2$\end{document}</tex-math></alternatives></inline-formula> and for human haplotypes <inline-formula><alternatives><mml:math id="inf209"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math><tex-math id="inft209">\begin{document}$D=1$\end{document}</tex-math></alternatives></inline-formula>). Suppose <inline-formula><alternatives><mml:math id="inf210"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math><tex-math id="inft210">\begin{document}$D=2$\end{document}</tex-math></alternatives></inline-formula> for exposition. For <inline-formula><alternatives><mml:math id="inf211"><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:math><tex-math id="inft211">\begin{document}$d=0,1,2$\end{document}</tex-math></alternatives></inline-formula>, let <inline-formula><alternatives><mml:math id="inf212"><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>⊂</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math><tex-math id="inft212">\begin{document}$I_{d}\subset[N]$\end{document}</tex-math></alternatives></inline-formula> denote the indices of those elements of the allelic dosage vector <inline-formula><alternatives><mml:math id="inf213"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft213">\begin{document}$(x_{1i},\ldots,x_{Ni})$\end{document}</tex-math></alternatives></inline-formula> taking on value <inline-formula><alternatives><mml:math id="inf214"><mml:mi>d</mml:mi></mml:math><tex-math id="inft214">\begin{document}$d$\end{document}</tex-math></alternatives></inline-formula>. Then, any permutation <inline-formula><alternatives><mml:math id="inf215"><mml:mi>σ</mml:mi><mml:mo>∈</mml:mo><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="fraktur">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft215">\begin{document}$\sigma\in\mathfrak{S}_{N}$\end{document}</tex-math></alternatives></inline-formula> of the rows of <inline-formula><alternatives><mml:math id="inf216"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="bold">X</mml:mi></mml:mrow></mml:math><tex-math id="inft216">\begin{document}$\mathbf{X}$\end{document}</tex-math></alternatives></inline-formula> that preserves marginal correlation with the phenotype and LD maps indices from <inline-formula><alternatives><mml:math id="inf217"><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft217">\begin{document}$I_{d}$\end{document}</tex-math></alternatives></inline-formula> to <inline-formula><alternatives><mml:math id="inf218"><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft218">\begin{document}$I_{d}$\end{document}</tex-math></alternatives></inline-formula>, because</p><list list-type="bullet" id="list4"><list-item><p>Each distinct entry in the allelic dosage vector of the focal SNP is still assigned the same phenotype entry;</p></list-item><list-item><p>Row shuffles do not change column-column covariances, the latter of which determines LD.</p></list-item></list><p>To be precise, <inline-formula><alternatives><mml:math id="inf219"><mml:mi>σ</mml:mi><mml:mo>∈</mml:mo><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="fraktur">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>×</mml:mo><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="fraktur">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>×</mml:mo><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi mathvariant="fraktur">S</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:math><tex-math id="inft219">\begin{document}$\sigma\in\mathfrak{S}_{|I_{0}|}\times\mathfrak{S}_{|I_{1}|}\times\mathfrak{S}_ {|I_{2}|}$\end{document}</tex-math></alternatives></inline-formula>, a direct product of the three permutation groups corresponding to the different allelic dosages.</p><p>In practice, we approximate the permutation distribution by sampling permutations uniformly at random from the group defined above. We set the sampling number to 500 for all our fine-mapping experiments.</p></sec></app><app id="appendix-2"><title>Appendix 2</title><sec sec-type="appendix" id="s9"><title>Impact of including more potential sets on matching frequency and causal variant recovery</title><p>The number of credible or potential sets is a parameter in many fine-mapping algorithms. Focusing on stability-guided approaches, we consider how including more potential sets for stable fine-mapping algorithms affects both causal variant recovery and matching frequency in simulations. For the latter, we will specifically investigate whether including more potential sets in PICS and searching for matching variants across different potential sets for Top and Stable PICS will improve causal variant recovery.</p><sec sec-type="appendix" id="s9-1"><title>Causal variant recovery</title><p>We investigate both Stable PICS and Stable SuSiE. Focusing first on simulations with one causal variant, we observe a modest gain in causal variant recovery for both Stable PICS and Stable SuSiE, most noticeably when the number of sets was increased from 1 to 2 under the lowest signal-to-noise ratio setting (SNR = 0.053, see <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>). For example, for Stable PICS, we observed 4 genes—<monospace>ENSG00000203760.4 </monospace>in Chr6, <monospace>ENSG00000132912.8</monospace> in Chr5, and <monospace>ENSG00000168614.12</monospace> and <monospace>ENSG00000171502.10</monospace> in Chr1—where the most stable variant in Potential Set 2 was the causal variant, which we would not have recovered had we only told Stable PICS to return one potential set.</p><p>When we consider simulations with two and three causal variants, we observe modest gains when increasing from 1 to 2 sets (<xref ref-type="fig" rid="fig2s6">Figure 2—figure supplement 6</xref>) and from 2 to 3 sets (<xref ref-type="fig" rid="fig2s7">Figure 2—figure supplement 7</xref>). Bigger gains were observed for Stable SuSiE than for Stable PICS, especially from 1 to 2 sets, where Stable SuSiE saw larger increases in probability of recovering all causal variants across all SNR settings.</p></sec><sec sec-type="appendix" id="s9-2"><title>Matching frequency</title><p>In PICS, we assume that each potential set returns just one causal variant, because PICS explicitly groups variants with very high LD within the same potential set. Here, we ask whether non-matching variants between the same potential sets returned by Top and Stable PICS are owing to the lack of multiple causal variant modeling; in other words, whether non-matching variants would match across <italic>different</italic> Top PICs and Stable PICS potential sets.</p><p>Computing fractions of non-matching variants for which the stable variant matched the top variant of another potential set, we observed generally small matching fractions (see <xref ref-type="table" rid="app12table4">Appendix 12—table 4</xref>). The mean matching fraction between Potential Set 1 stable variant and a top variant in Potential Set 2 or 3 across all simulations was 11%, with larger matching fractions occurring between Potential Set 1 stable and Potential Set 2 top variants. Mean matching fractions for Potential Sets 2 and 3 stable variant and a different potential set top variant were 23% and 14%, with larger matching fractions occurring between Potential Set 2 stable and Potential Set 3 top variants and Potential Set 3 stable and Potential Set 2 top variants, respectively.</p><p>We next checked if these ‘off-diagonal’ matching variants are enriched in causal variants. Across all simulations, an off-diagonal matching Potential Set 1 stable variant recovered a causal variant 54% of the time, while for Potential Set 2 stable variant and Potential Set 3 stable variant the recovery percentages were 16% and 8%. These fractions are larger than if the stable variant did not match any top variant (24%, 5%, and 3% for Potential Sets 1, 2, and 3, respectively). However, these fractions are generally smaller when compared against causal variant recovery percentages for matching variants: we observed smaller fractions for Potential Set 1 (54% &lt; 77%) and Potential Set 2 (16% &lt; 20%), and only for Potential Set 3 did we observe a slightly larger fraction (8% &gt; 6%).</p><p>Taken together, these findings demonstrate that it is helpful to condition on matching variants between ‘off-diagonal’ potential sets to recover causal variants in case the top and stable variants in the same potential set do not match, but such events occur infrequently and do not enrich for causal variants as much as do matching variants from the same potential sets. Therefore, in our real data analysis, we proceed with comparing matching variants from the same potential sets.</p></sec></sec></app><app id="appendix-3"><title>Appendix 3</title><sec sec-type="appendix" id="s10"><title>Difference in posterior probability of non-matching variants</title><p>We investigate whether there are differences in posterior probabilities between non-matching variants in simulations. Focusing first on Potential Set 1 non-matching variants, we observe that posterior probabilities tend to be clustered around two regions: (1) similar posterior probabilities that are less than 0.5; and (2) low stable variant posterior probabilities (<xref ref-type="fig" rid="fig2s9">Figure 2—figure supplement 9</xref>). This clustering pattern differs from matching variants, which tend to have more similar and larger posterior probabilities (<xref ref-type="fig" rid="fig2s8">Figure 2—figure supplement 8</xref>). Moving down to Potential Sets 2 and 3, we observe a similar clustering around region (2) for non-matching variants, with the remaining points distributed more randomly within the rectangle [0,1] × [0,1] (<xref ref-type="fig" rid="fig2s9">Figure 2—figure supplement 9</xref>). Matching variants for these two potential sets remain more similar (lying close to <inline-formula><alternatives><mml:math id="inf220"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>x</mml:mi></mml:math><tex-math id="inft220">\begin{document}$y=x$\end{document}</tex-math></alternatives></inline-formula> diagonal of the rectangle, see <xref ref-type="fig" rid="fig2s8">Figure 2—figure supplement 8</xref>), suggesting that a salient feature of non-matching variants is a small stable variant posterior probability (despite possibly large top variant posterior probability).</p></sec></app><app id="appendix-4"><title>Appendix 4</title><sec sec-type="appendix" id="s11"><title>Comparing SuSiE to stable PICS</title><p>We investigate how PICS performs relative to SuSiE, a widely used fine-mapping algorithm, in simulations. Our comparison is between Stable PICS and Top SuSiE, the latter of which performs residualization on gene expression phenotypes to remove potential confounding by ancestry. Starting with simulations involving one causal variant, we observe slightly better causal variant recovery frequency for SuSiE at larger SNR, although the opposite is true at smaller SNR (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). Notably, PICS always recovers causal variants with probability at least 0.5. Moving onto simulations with two causal variants (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>) and three causal variants (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>), we observe similar patterns, where PICS recovers more causal variants at small SNR and SuSiE recovers more causal variants at large SNR. (Comparisons between Stable PICS and Stable SuSiE are summarized in <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplements 5–7</xref>.)</p></sec></app><app id="appendix-5"><title>Appendix 5</title><sec sec-type="appendix" id="s12"><title>List of annotations</title><p>We measure the biological significance of a variant using a wide range of available functional annotations (see <xref ref-type="table" rid="app12table6">Appendix 12—table 6</xref>). The annotations cover potential biological activity in the local vicinity of the variant position along the genome, estimated model-based biological quantities (e.g., selection and conservation scores), and predicted effects of mutagenesis from the reference to the alternate allele at the variant.</p></sec></app><app id="appendix-6"><title>Appendix 6</title><sec sec-type="appendix" id="s13"><title>Generating annotations from enformer predictions</title><p>The Enformer (<xref ref-type="bibr" rid="bib4">Avsec et al., 2021</xref>) is a sequence-based prediction model, which leverages the attention mechanism in a transformer neural network to capture long-range effects on gene regulation.</p><p>For Enformer predictions, we subset the 5313 ChIP-seq, DNase-seq, ATAC-seq, and CAGE tracks to only those relevant to the lymphoblastoid cell line (GM12878), the cell line with respect to which GEUVADIS gene expression phenotypes are measured. We also restrict to genes whose top and stable variants are within the 196,608 bp input length limit of the Enformer from the corresponding GEUVADIS gene’s TSS. This restriction allows us to obtain Enformer predictions on three sequences: a <italic>null sequence</italic>, a sequence where the REF allele is replaced with the ALT allele at the top variant (<italic>top sequence</italic>), and a sequence where the REF allele is replaced with the ALT allele at the stable variant (<italic>stable sequence</italic>).</p><p>To compare the top and the stable variants with respect to a particular track, we take the predictions obtained from the three input sequences (null prediction, top prediction, stable prediction—a <italic>triplet</italic>), and compute the magnitude of change in predictions between both the top and null and the stable and null. We use the magnitude of change here, rather than the change itself, to capture the impact the mutation associated with the variant has on the track. Our Enformer functional annotations can thus be interpreted as measures of mutagenic impact on a track’s prediction, without considering the directionality of the impact. We refer to these annotations as <italic>perturbation scores</italic>.</p><p>We compute perturbation scores in two ways. The first way is to center all input sequences on the transcription start site of the gene (occasionally referred to as ‘TSS’ in our work), thus allowing the model to measure the impact of mutagenesis at the (top or stable) variant on predicted gene expression profile for the sequence. The second way is to average the perturbation scores over three triplets, one centered on the gene TSS, and two centered on the flanking positions of the gene TSS (i.e., the two neighboring bins). The second way (occasionally referred to as ‘AVE’ in our work) accounts for errors arising from imprecise TSS positioning and the instability of the model to small changes in input—a technique used in <xref ref-type="bibr" rid="bib31">Karollus et al., 2023</xref> and by the Enformer team (<xref ref-type="bibr" rid="bib4">Avsec et al., 2021</xref>).</p></sec></app><app id="appendix-7"><title>Appendix 7</title><sec sec-type="appendix" id="s14"><title>Matching vs non-matching variant results</title><p>As described in the main text, we compare between one set of genes where the top and stable variants match and another set of genes where the top and stable variants disagree; comparisons are made between matching variants and one of the non-matching sets of variants (top or stable).</p><p>Concretely, for Potential Set 1, we compare between <inline-formula><alternatives><mml:math id="inf221"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>12743</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft221">\begin{document}$N_{\text{match}}=12743$\end{document}</tex-math></alternatives></inline-formula> genes with matching variants and <inline-formula><alternatives><mml:math id="inf222"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>non-match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>9921</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft222">\begin{document}$N_{\text{non-match}}=9921$\end{document}</tex-math></alternatives></inline-formula> genes with non-matching variants; for Potential Set 2, we compare between <inline-formula><alternatives><mml:math id="inf223"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>8197</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft223">\begin{document}$N_{\text{match}}=8197$\end{document}</tex-math></alternatives></inline-formula> genes with matching variants and <inline-formula><alternatives><mml:math id="inf224"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>non-match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>14452</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft224">\begin{document}$N_{\text{non-match}}=14452$\end{document}</tex-math></alternatives></inline-formula> genes with non-matching variants; and for Potential Set 3, we compare between <inline-formula><alternatives><mml:math id="inf225"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>5807</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft225">\begin{document}$N_{\text{match}}=5807$\end{document}</tex-math></alternatives></inline-formula> genes with matching variants and <inline-formula><alternatives><mml:math id="inf226"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>non-match</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>16812</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft226">\begin{document}$N_{\text{non-match}}=16812$\end{document}</tex-math></alternatives></inline-formula> genes with non-matching variants. Below, we summarize our findings.</p><sec sec-type="appendix" id="s14-1"><title>Matching variants vs non-matching top variants</title><p>We observe the following significant trends: as mentioned in the main text, for Potential Set 1361 functional annotations reported significantly higher scores for the matching variants, with all BH-adjusted p-values less than 0.05. The remaining 17 = 378 − 361 annotations <italic>not</italic> exhibiting significantly higher scores are: Distance to Canonical TSS, CTCF Binding Enrichment, Enhancer Enrichment, Open Chromatin Enrichment, TF Binding Enrichment, Promoter Flanking Enrichment, CADD raw score, PHRED-normalized CADD score, SIFTVal, priPhyloP, mamPhyloP, verPhyloP, GerpS, LINSIGHT, Funseq2, ALoft, and FIRE.</p><p>For Potential Set 2, 132 functional annotations reported significantly higher scores for the matching variants. For Potential Set 3, 9 functional annotations reported significantly higher scores for the matching variants: B Statistic, percent GC content, and Enformer tracks <monospace>ENCFF776DPQ, ENCFF831ZHL, ENCFF601YET, ENCFF945XXY, ENCFF676UXN, ENCFF151LGF, ENCFF876DXW</monospace>. For all three potential sets, there were no functional annotations reporting significantly lower scores for the matching variants, even for annotations where a low score implies greater biological functionality (e.g., B Statistic).</p></sec><sec sec-type="appendix" id="s14-2"><title>Matching variants vs non-matching stable variants</title><p>We observe the following significant trends: as mentioned in the main text, for Potential Set 1, 363 functional annotations reported significantly higher scores for the matching variants, with all BH-adjusted p-values less than 0.05. The remaining 15 annotations <italic>not</italic> exhibiting significantly higher scores are: Distance to Canonical TSS, CTCF Binding Enrichment, Enhancer Enrichment, Open Chromatin Enrichment, TF Binding Enrichment, Promoter Flanking Enrichment, CADD raw score, PHRED-normalized CADD score, SIFTVal, mamPhyloP, verPhyloP, GerpS, LINSIGHT, ALoft, and FIRE.</p><p>For Potential Set 2, 359 functional annotations reported significantly higher scores for the matching variants (all BH-adjusted p-values &lt;0.05), while only one functional annotation—Distance to Canonical TSS—reported significantly lower scores for the matching variants (BH-adjusted p-value <inline-formula><alternatives><mml:math id="inf227"><mml:mo>=</mml:mo><mml:mn>8</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft227">\begin{document}$=8\times 10^{-4}$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 3, 259 functional annotations reported significantly higher scores for the matching variants (all BH-adjusted p-values &lt;0.05), while only one functional annotation—Distance to Canonical TSS—reported significantly lower scores for the matching variants (BH-adjusted p-value <inline-formula><alternatives><mml:math id="inf228"><mml:mo>=</mml:mo><mml:mn>8</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft228">\begin{document}$=8\times 10^{-4}$\end{document}</tex-math></alternatives></inline-formula>).</p></sec></sec></app><app id="appendix-8"><title>Appendix 8</title><sec sec-type="appendix" id="s15"><title>Top variant vs stable variant results</title><p>As described in the main text, we compare, across all genes, the annotations of the top and the stable variant. We reported no significant differences in enrichment or trends in enrichment differences driven by moderators for Potential Set 1. For completeness, we report here significant results for all three potential sets.</p><sec sec-type="appendix" id="s15-1"><title>Paired analysis</title><p>After BH adjustment, there are no significant differences across all three potential sets using a threshold of <inline-formula><alternatives><mml:math id="inf229"><mml:mi>α</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft229">\begin{document}$\alpha=0.05$\end{document}</tex-math></alternatives></inline-formula>.</p></sec><sec sec-type="appendix" id="s15-2"><title>Trend analysis</title><p>For the following moderators, we observe some significant trends.</p><list list-type="bullet" id="list5"><list-item><p>Inclusion of Distal Subpopulations (Top)</p><p>For Potential Set 3, FIRE scores exhibited a greater difference between stable and top variant scores (i.e., <inline-formula><alternatives><mml:math id="inf230"><mml:mtext>stable score</mml:mtext><mml:mo>−</mml:mo><mml:mtext>top score</mml:mtext></mml:math><tex-math id="inft230">\begin{document}$\text{stable score}-\text{top score}$\end{document}</tex-math></alternatives></inline-formula>) for genes with top variants not discovered in YRI than those with top variants discovered in YRI (one-sided BH-adjusted <inline-formula><alternatives><mml:math id="inf231"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.03</mml:mn></mml:math><tex-math id="inft231">\begin{document}$p=0.03$\end{document}</tex-math></alternatives></inline-formula>). So did TSS-centered perturbation scores for 121 Enformer tracks and AVE perturbation scores for 121 Enformer tracks. (These tracks and their corresponding BH-adjusted p-values are available <ext-link ext-link-type="uri" xlink:href="https://github.com/songlab-cal/StableFM/blob/master/data/results_with_moderators/p_values/mod_Top%20SNV%20Discovered%20in%20YRI.csv">here</ext-link>.)</p></list-item><list-item><p>Posterior Probability of Top Variant</p><p>For all potential sets, there are small but significant positive correlations between the posterior probability of the top variant and the difference between the stable variant FIRE score and the top variant FIRE score (Potential Set 1: Pearson’s <inline-formula><alternatives><mml:math id="inf232"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft232">\begin{document}$r=0.05$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf233"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft233">\begin{document}$p=3\times 10^{-5}$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 2: Pearson’s <inline-formula><alternatives><mml:math id="inf234"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.07</mml:mn></mml:math><tex-math id="inft234">\begin{document}$r=0.07$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf235"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>14</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft235">\begin{document}$p=3\times 10^{-14}$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 3: Pearson’s <inline-formula><alternatives><mml:math id="inf236"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft236">\begin{document}$r=0.05$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf237"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft237">\begin{document}$p=1\times 10^{-7}$\end{document}</tex-math></alternatives></inline-formula>). For all potential sets, there are small but significant negative correlations between the posterior probability of the top variant and the difference between the stable variant Absolute Distance to TSS and the top variant Absolute Distance to TSS (Potential Set 1: Pearson’s <inline-formula><alternatives><mml:math id="inf238"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.07</mml:mn></mml:math><tex-math id="inft238">\begin{document}$r=-0.07$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf239"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft239">\begin{document}$p=1\times 10^{-10}$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 2: Pearson’s <inline-formula><alternatives><mml:math id="inf240"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.05</mml:mn></mml:math><tex-math id="inft240">\begin{document}$r=-0.05$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf241"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>9</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft241">\begin{document}$p=9\times 10^{-7}$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 3: Pearson’s <inline-formula><alternatives><mml:math id="inf242"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.06</mml:mn></mml:math><tex-math id="inft242">\begin{document}$r=-0.06$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf243"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>12</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft243">\begin{document}$p=2\times 10^{-12}$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 2, there are significant negative correlations between the posterior probability of the top variant and the difference between the stable variant and top variant Percent GC (Pearson’s <inline-formula><alternatives><mml:math id="inf244"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.04</mml:mn></mml:math><tex-math id="inft244">\begin{document}$r=-0.04$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf245"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft245">\begin{document}$p=5\times 10^{-5}$\end{document}</tex-math></alternatives></inline-formula>) and FATHMM-XF scores (Pearson’s <inline-formula><alternatives><mml:math id="inf246"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.04</mml:mn></mml:math><tex-math id="inft246">\begin{document}$r=-0.04$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf247"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>7</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft247">\begin{document}$p=7\times 10^{-4}$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 3, there is a significant negative correlation between the posterior probability of the top variant and the difference between the stable variant and top variant FATHMM-XF scores (Pearson’s <inline-formula><alternatives><mml:math id="inf248"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.04</mml:mn></mml:math><tex-math id="inft248">\begin{document}$r=-0.04$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf249"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft249">\begin{document}$p=2\times 10^{-5}$\end{document}</tex-math></alternatives></inline-formula>).</p></list-item></list></sec></sec></app><app id="appendix-9"><title>Appendix 9</title><sec sec-type="appendix" id="s16"><title>Impact of positive posterior probability support on results</title><p>One key consideration in statistical fine-mapping is the number of variants possessing positive posterior probability, which we refer to as the <italic>positive posterior probability support</italic> (or <italic>support</italic>, for short). Indeed, a larger support may indicate greater uncertainty in the assignment of causal variant to the allele with largest posterior probability. To evaluate the potential utility of the stability-guided approach in settings where the residualization approach leads to large support, we repeat our analysis on two restricted sets of genes, namely (1) those genes where the top variant has support greater than 10; or (2) those genes where the top variant has support greater than 50.</p><sec sec-type="appendix" id="s16-1"><title>Paired analysis</title><p><italic>Gene Set (1)</italic>. For all Potential Sets, Distance to TSS is significantly smaller for top variant (all three BH-adjusted p-values <inline-formula><alternatives><mml:math id="inf250"><mml:mo>&lt;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft250">\begin{document}$ \lt 10^{-6}$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 3, FATHMM-XF score is significantly larger for stable variant (BH-adjusted <inline-formula><alternatives><mml:math id="inf251"><mml:mi>p</mml:mi><mml:mo>&lt;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft251">\begin{document}$p \lt 10^{-6}$\end{document}</tex-math></alternatives></inline-formula>). For all Potential Sets, FIRE score is significantly larger for top variant (BH-adjusted p-values: Potential Set 1 = 0.02, Potential Set 2 <inline-formula><alternatives><mml:math id="inf252"><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft252">\begin{document}$=2\times 10^{-10}$\end{document}</tex-math></alternatives></inline-formula>, Potential Set 3 <inline-formula><alternatives><mml:math id="inf253"><mml:mo>=</mml:mo><mml:mn>1.7</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft253">\begin{document}$=1.7\times 10^{-4}$\end{document}</tex-math></alternatives></inline-formula>). <italic>Gene Set (2)</italic>. For all Potential Sets 2 and 3, FIRE score is significantly larger for top variant (both BH-adjusted p-values are 0.04).</p></sec><sec sec-type="appendix" id="s16-2"><title>Trend analysis</title><p><italic>Gene Set (1)</italic>. For the following moderators, we observe some significant trends.</p><list list-type="bullet" id="list6"><list-item><p>Posterior Probability of Top Variant</p><p>For Potential Sets 2 and 3, there is a significant negative correlation between the posterior probability of the top variant and the difference between the stable variant FIRE score and the top variant FIRE score (Potential Set 2: Pearson’s <inline-formula><alternatives><mml:math id="inf254"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.08</mml:mn></mml:math><tex-math id="inft254">\begin{document}$r=0.08$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf255"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>1.9</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft255">\begin{document}$p=1.9\times 10^{-5}$\end{document}</tex-math></alternatives></inline-formula>; Potential Set 3: Pearson’s <inline-formula><alternatives><mml:math id="inf256"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>0.06</mml:mn></mml:math><tex-math id="inft256">\begin{document}$r=0.06$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf257"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.003</mml:mn></mml:math><tex-math id="inft257">\begin{document}$p=0.003$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 2, there is a significant positive correlation between the posterior probability of the top variant and the difference between the stable variant’s FATHMM-XF score and the top variant’s FATHMM-XF score (Pearson’s <inline-formula><alternatives><mml:math id="inf258"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.06</mml:mn></mml:math><tex-math id="inft258">\begin{document}$r=-0.06$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf259"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:math><tex-math id="inft259">\begin{document}$p=0.01$\end{document}</tex-math></alternatives></inline-formula>).</p></list-item></list><p><italic>Gene Set (2)</italic>. For the following moderators, we observe some significant trends.</p><list list-type="bullet" id="list7"><list-item><p>Posterior Probability of Top Variant</p><p>For Potential Set 2, TSS-centered perturbation scores for 8 Enformer tracks exhibited a positive correlation between the posterior probability of the top variant and the difference between stable and top variant perturbation scores. Track names (with empirical Pearson’s <inline-formula><alternatives><mml:math id="inf260"><mml:mi>r</mml:mi></mml:math><tex-math id="inft260">\begin{document}$r$\end{document}</tex-math></alternatives></inline-formula>) are <monospace>ENCFF915DFR (-0.40), ENCFF107LDM (-0.41), ENCFF821PRO (-0.42), ENCFF171MDW (-0.43), ENCFF170NTY (-0.43), ENCFF935KTD (-0.41), ENCFF676GTP (-0.41), ENCFF013ZOI (-0.42)</monospace>. Averaged perturbation scores for two Enformer tracks also exhibited negative correlation; these two tracks are <monospace>ENCFF107LDM (-0.40)</monospace> and <monospace>ENCFF170NTY (-0.40)</monospace>. All BH-adjusted p-values are 0.0495.</p></list-item></list></sec></sec></app><app id="appendix-10"><title>Appendix 10</title><sec sec-type="appendix" id="s17"><title>Impact of posterior probability on results</title><p>We perform a comparison between top and stable variants, by restricting to genes where the posterior probability of the top variant or the stable variant exceeds 0.9. Specifically, we repeat our analysis on two restricted sets of genes, namely (1) those genes where the top variant reported a posterior probability exceeding 0.9; or (2) those genes where the stable variant reported a posterior probability exceeding 0.9. For reference, we plot in <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref> the joint distribution of posterior probabilities of top and stable variants, across all genes for which the two fine-mapping approaches returned distinct variants.</p><sec sec-type="appendix" id="s17-1"><title>Paired analysis</title><p><italic>Gene Set (1)</italic>. For Potential Set 3, Distance to TSS is significantly smaller for stable variant (BH-adjusted <inline-formula><alternatives><mml:math id="inf261"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.01</mml:mn></mml:math><tex-math id="inft261">\begin{document}$p=0.01$\end{document}</tex-math></alternatives></inline-formula>). <italic>Gene Set (2)</italic>. For Potential Sets 1 and 2, FIRE scores of the top variant are significantly larger than the stable variant (BH-adjusted p-values: Potential Set 1 <inline-formula><alternatives><mml:math id="inf262"><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft262">\begin{document}$=1\times 10^{-6}$\end{document}</tex-math></alternatives></inline-formula>, Potential Set 2 = 0.02). For Potential Set 2, the distance to TSS of the top variant is significantly smaller (BH-adjusted <inline-formula><alternatives><mml:math id="inf263"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>−</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft263">\begin{document}$p=3\times 10^{-4}$\end{document}</tex-math></alternatives></inline-formula>). For Potential Set 3, FATHMM-XF scores of the stable variant are significantly larger (BH-adjusted <inline-formula><alternatives><mml:math id="inf264"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.006</mml:mn></mml:math><tex-math id="inft264">\begin{document}$p=0.006$\end{document}</tex-math></alternatives></inline-formula>).</p></sec><sec sec-type="appendix" id="s17-2"><title>Trend analysis</title><p><italic>Gene Set (1)</italic>. For the following moderators, we observe some significant trends.</p><list list-type="bullet" id="list8"><list-item><p>Inclusion of Distal Subpopulations (Top)</p><p>For Potential Set 3, TSS-centered perturbation scores for nine Enformer tracks exhibited a higher difference between stable and top variant perturbation score for genes with top variants not discovered in YRI vs those discovered in YRI. Track names (with BH-adjusted unpaired Wilcoxon test p-values) are <monospace>ENCFF279CYY (0.03), ENCFF417WYL (0.04), ENCFF782WWH (0.03), ENCFF629RRF (0.03), ENCFF676GTP (0.03), ENCFF984HLU (0.03), ENCFF700YOH (0.03), ENCFF848LJL (0.0477), ENCFF038IYA (0.03). Averaged perturbation scores for 10 Enformer tracks also a higher difference; these are ENCFF776DPQ (0.02), ENCFF279CYY (0.02), ENCFF629RRF (0.03), ENCFF319YAI (0.04), ENCFF676GTP (0.046), ENCFF367WTF (0.04), ENCFF984HLU (0.04), ENCFF917YSR (0.03), ENCFF613CYH (0.04), and ENCFF700YOH (0.04)</monospace>.</p></list-item></list><p><italic>Gene Set (2)</italic>. For the following moderators, we observe some significant trends.</p><list list-type="bullet" id="list9"><list-item><p>Posterior Probability of Top Variant</p><p>For Potential Set 3, there is a significant negative correlation between the posterior probability of the top variant and the difference between the stable variant Distance to TSS and the top variant Distance to TSS (Pearson’s <inline-formula><alternatives><mml:math id="inf265"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mn>0.09</mml:mn></mml:math><tex-math id="inft265">\begin{document}$r=-0.09$\end{document}</tex-math></alternatives></inline-formula>, BH-adjusted <inline-formula><alternatives><mml:math id="inf266"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mn>0.004</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft266">\begin{document}$p=0.004$\end{document}</tex-math></alternatives></inline-formula>).</p></list-item></list></sec></sec></app><app id="appendix-11"><title>Appendix 11</title><sec sec-type="appendix" id="s18"><title>On matching variants with very low (stable) posterior probability</title><p>In both simulations and analysis of GEUVADIS data, some matching variants have low stable posterior probability. To better interpret such variants, we look at allele frequency heterogeneity by ancestry, posterior probability support size, and (in Stable PICS) number of slices containing the variant. For simulated data, we additionally consider if such low posterior probabilities can be predictive of causal variant recovery. For GEUVADIS data, we compare functional enrichments of such variants relative to other matching variants.</p><sec sec-type="appendix" id="s18-1"><title>Simulated gene expression</title><p>By defining a very low stable posterior probability as having a probability &lt;0.01, we found two expression phenotypes that returned matching top and stable variants in Potential Set 1: <monospace>rs8008094 </monospace>(using TSS metadata for gene <monospace>ENSG00000151413.12</monospace>) and <monospace>rs4758290 </monospace>(using TSS metadata for gene <monospace>ENSG00000258659.1</monospace>). Both simulations involved two causal variants and &gt;1000 background variants. Allele frequencies for both variants did not differ substantially across ancestry slices: <inline-formula><alternatives><mml:math id="inf267"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mtext>YRI</mml:mtext></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mtext>TSI</mml:mtext></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mtext>GBR</mml:mtext></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mtext>FIN</mml:mtext></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mtext>CEU</mml:mtext></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0.408</mml:mn><mml:mo>,</mml:mo><mml:mn>0.302</mml:mn><mml:mo>,</mml:mo><mml:mn>0.198</mml:mn><mml:mo>,</mml:mo><mml:mn>0.25</mml:mn><mml:mo>,</mml:mo><mml:mn>0.208</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math><tex-math id="inft267">\begin{document}$(f_{\text{YRI}},f_{\text{TSI}},f_{\text{GBR}},f_{\text{FIN}},f_{\text{CEU}})= \linebreak(0.408,0.302,0.198,0.25,0.208)$\end{document}</tex-math></alternatives></inline-formula> for <monospace>rs8008094 </monospace>with pooled allele frequency <inline-formula><alternatives><mml:math id="inf268"><mml:mi>f</mml:mi><mml:mo>=</mml:mo><mml:mn>0.273</mml:mn></mml:math><tex-math id="inft268">\begin{document}$f=0.273$\end{document}</tex-math></alternatives></inline-formula>, while <inline-formula><alternatives><mml:math id="inf269"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>TSI</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>FIN</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>CEU</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0.540</mml:mn><mml:mo>,</mml:mo><mml:mn>0.527</mml:mn><mml:mo>,</mml:mo><mml:mn>0.523</mml:mn><mml:mo>,</mml:mo><mml:mn>0.571</mml:mn><mml:mo>,</mml:mo><mml:mn>0.528</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft269">\begin{document}$(f_{\text{YRI}},f_{\text{TSI}},f_{\text{GBR}},f_{\text{FIN}},f_{\text{CEU}})=(0.540,0.527,0.523,0.571,0.528)$\end{document}</tex-math></alternatives></inline-formula> for <monospace>rs4758290 </monospace>with pooled allele frequency <inline-formula><alternatives><mml:math id="inf270"><mml:mi>f</mml:mi><mml:mo>=</mml:mo><mml:mn>0.538</mml:mn></mml:math><tex-math id="inft270">\begin{document}$f=0.538$\end{document}</tex-math></alternatives></inline-formula>. However, for <monospace>rs8008094, </monospace>the positive posterior probability support size was 47, whereas for <monospace>rs4758290, </monospace>the positive posterior probability support size was 7. Additionally, Top PICS returned a posterior probability of 0.359 for <monospace>rs8008094 </monospace>and 0.846 for <monospace>rs4758290</monospace>. The latter variant also appeared in five slices for Stable PICS, while the former appeared in three slices. It turns out that <monospace>rs4758290 </monospace>was one of the causal variants in the simulations whereas <monospace>rs8008094 </monospace>was not.</p><p>Moving onto Potential Sets 2 and 3 matching variants, using the same cutoff, we identified five (<monospace>rs4687770, rs2834344, rs1125036, rs7963386, rs74080151</monospace>) and one (<monospace>rs1404862</monospace>) variant(s), respectively, that had low stable posterior probability. All but two simulations involved three causal variants—<monospace>rs4687770 </monospace>and <monospace>rs1404862 </monospace>involved two causal variants. Inspecting Top PICS posterior probabilities for these variants, we observed a modest spread of values (all contained in [0.33, 0.85]). Each variant was also contained in at least three slices in Stable PICS (three were contained in five slices), while posterior probability support sizes ranged from 6 to 57. Apart from <monospace>rs4687770</monospace>, which had a considerable difference in allele frequency between YRI and GBR (<inline-formula><alternatives><mml:math id="inf271"><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>YRI</mml:mtext></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>GBR</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.36</mml:mn></mml:math><tex-math id="inft271">\begin{document}$f_{\text{YRI}}-f_{\text{GBR}}=0.36$\end{document}</tex-math></alternatives></inline-formula>), all other variants did not report substantial difference in allele frequencies by ancestry. None of these variants were causal variants used in simulations.</p><p>While we cannot derive any generalizable conclusions from such a small number of observations, this analysis suggests that matching variants with very low stable posterior probability are largely depleted in causal variants. However, at least for Potential Set 1, other factors, such as the number of slices including the stable variant as well as the top variant posterior probability, may still be useful for causal variant enrichment.</p></sec><sec sec-type="appendix" id="s18-2"><title>GEUVADIS</title><p>Again using a very low stable posterior probability cutoff of 0.01, we found 17 matching variants with very low stable posterior probability across all three potential sets. These are listed in <xref ref-type="table" rid="app12table5">Appendix 12—table 5</xref>. Briefly, the Top PICS posterior probabilities ranged from 0.23 to 1, with Stable PICS support sizes ranging from as small as 4 to as large as 45. The number of slices containing the stable variant (including the ALL slice) ranged from 3 to 6, while the maximum allele frequency difference between any pair of ancestry slices ranged from 0.084 to 0.43. Comparing these quantities potential set by potential set, allele frequencies tend to be less homogeneous and support sizes tend to be in the upper half of the distribution across the complementary sets of matching variants with stable posterior probability ≥0.01. Notably, all eight Potential Set 1 variants had support sizes lying above the third quartile of the distribution of support sizes for Potential Set 1 matching variants with stable posterior probability ≥0.01. While we cannot derive any generalizable conclusions from such a small number of observations, this suggests that a larger support size could result in stable variants with low stable posterior probability.</p><p>Because we do not know the causal variants, we performed functional enrichment analyses of the 17 matching variants. Similar to the above analysis, we cannot make generalizable claims owing to small sample sizes, but we can nonetheless report some patterns. By inspecting each functional annotation, we did not notice systematic patterns in either extreme direction relative to matching variants with stable posterior probability ≥0.01. However, we occasionally found variants that stand out for having impactful functional annotation scores. We list one below for each potential set.</p><list list-type="bullet" id="list10"><list-item><p>Potential Set 1 reported the variant <monospace>rs12224894 </monospace>from fine-mapping <monospace>ENSG00000255284.1 </monospace>(accession code <italic>AP006621.3</italic>) in Chromosome 11. This variant stood out for lying in the promoter flanking region of multiple cell types and being relatively enriched for GC content within a 75 bp flanking region. This variant has been reported as a cis eQTL for <italic>AP006621</italic> (using whole blood gene expression, rather than lymphoblastoid cell line gene expression in this study) in a clinical trial study of patients with systemic lupus erythematosus (<xref ref-type="bibr" rid="bib16">Davenport et al., 2018</xref>). Its nearest gene is <italic>GATD1</italic>, a ubiquitously expressed gene that codes for a protein and is predicted to regulate enzymatic and catabolic activity. This variant appeared in all 6 slices, with a moderate support size of 23.</p></list-item><list-item><p>Potential Set 2 reported the variant <monospace>rs9912201 </monospace>from fine-mapping <monospace>ENSG00000108592.9</monospace> (mapped to <italic>FTSJ3</italic>) in Chromosome 17. Its FIRE score is 0.976, which is close to the maximum FIRE score reported across all Potential Set 2 matching variants. This variant has been reported as an SNP in high LD to a GWAS hit SNP <monospace>rs7223966 </monospace>in a pan-cancer study (<xref ref-type="bibr" rid="bib23">Gong et al., 2018</xref>). This variant appeared in all six slices, with a moderate support size of 32.</p></list-item><list-item><p>Potential Set 3 reported the variant <monospace>rs625750 </monospace>from fine-mapping <monospace>ENSG00000254614.1 </monospace>(mapped to <italic>CAPN1-AS1</italic>, an RNA gene) in Chromosome 11. Its FIRE score is 0.971 and its B statistic is 0.405 (region under selection), which lie at the extreme quantiles of the distributions of these scores for Potential Set 3 matching variants with stable posterior probability ≥0.01. Its associated mutation has been predicted to affect transcription factor binding, as computed using several position weight matrices (<xref ref-type="bibr" rid="bib34">Kheradpour and Kellis, 2014</xref>). This variant appeared in just three slices, possibly owing to the considerable allele frequency difference between ancestries (maximum AF difference = 0.22). However, it has a small support size of 4, with moderately high Top PICS posterior probability of 0.64.</p></list-item></list><p>To summarize, our analysis of GEUVADIS fine-mapped variants demonstrates that matching variants with very low stable posterior probability could still be functionally important, even for lower potential sets, conditional on supportive scores in interpretable features such as the number of slices containing the stable variant and the posterior probability support size. It would be interesting to scale up such analyses in future work to investigate the generalizability of these conclusions beyond just 17 data points.</p></sec></sec></app><app id="appendix-12"><title>Appendix 12</title><sec sec-type="appendix" id="s19"><title>Supplementary tables</title><table-wrap id="app12table1" position="float"><label>Appendix 12—table 1.</label><caption><title>Plain and Stable PICS matching frequencies.</title><p>Below reports the frequencies with which Plain and Stable PICS have matching variants for the same potential set. The numbers of matching variants for each SNR scenario are reported in the parentheses. The bottom two rows show matching frequencies when results are stratified by posterior probability (PP) of the Plain PICS variant. The numbers of matching variants for each PP stratum are reported in the parentheses.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top" colspan="4">Stratified by signal-to-noise ratio (SNR) of simulations</th></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top">SNR = 0.053</td><td align="char" char="." valign="top">0.736 (265)</td><td align="char" char="." valign="top">0.803 (289)</td><td align="char" char="." valign="top">0.797 (287)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="char" char="." valign="top">0.775 (279)</td><td align="char" char="." valign="top">0.753 (271)</td><td align="char" char="." valign="top">0.758 (273)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="char" char="." valign="top">0.903 (325)</td><td align="char" char="." valign="top">0.714 (257)</td><td align="char" char="." valign="top">0.728 (262)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="char" char="." valign="top">0.906 (326)</td><td align="char" char="." valign="top">0.753 (271)</td><td align="char" char="." valign="top">0.744 (268)</td></tr><tr><th align="left" valign="bottom" colspan="4">Stratified by posterior probability (PP) of plain PICS variant</th></tr><tr><td align="left" valign="bottom">p &gt; 0.9</td><td align="left" valign="bottom">0.978 (441)</td><td align="left" valign="bottom">0.899 (286)</td><td align="left" valign="bottom">0.927 (307)</td></tr><tr><td align="left" valign="top">p ≤ 0.9</td><td align="left" valign="top">0.762 (754)</td><td align="left" valign="top">0.715 (802)</td><td align="left" valign="top">0.706 (783)</td></tr></tbody></table></table-wrap><table-wrap id="app12table2" position="float"><label>Appendix 12—table 2.</label><caption><title>Stable and Top PICS matching frequencies.</title><p>Below reports the frequencies with which Stable and Top PICS have matching variants for the same potential set. The numbers of matching variants for each SNR/‘No. Causal Variants’ scenario are reported in the parentheses.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top" colspan="5">Stratified by signal-to-noise ratio (SNR) and No. Causal Variants (<italic>S</italic>) in simulations</th></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top" rowspan="4">One causal variant (<italic>S</italic> = 1)</td><td align="left" valign="top">SNR = 0.053</td><td align="char" char="." valign="top">0.695 (139)</td><td align="char" char="." valign="top">0.41 (82)</td><td align="char" char="." valign="top">0.22 (44)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="char" char="." valign="top">0.75 (150)</td><td align="char" char="." valign="top">0.405 (81)</td><td align="char" char="." valign="top">0.24 (48)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="char" char="." valign="top">0.825 (165)</td><td align="char" char="." valign="top">0.45 (90)</td><td align="char" char="." valign="top">0.26 (52)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="char" char="." valign="top">0.895 (179)</td><td align="char" char="." valign="top">0.405 (81)</td><td align="char" char="." valign="top">0.275 (55)</td></tr><tr><td align="left" valign="top" rowspan="4">Two causal variants (<italic>S</italic> = 2)</td><td align="left" valign="top">SNR = 0.053</td><td align="char" char="." valign="top">0.545 (109)</td><td align="char" char="." valign="top">0.36 (72)</td><td align="char" char="." valign="top">0.225 (45)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="char" char="." valign="top">0.68 (136)</td><td align="char" char="." valign="top">0.38 (76)</td><td align="char" char="." valign="top">0.215 (43)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="char" char="." valign="top">0.79 (158)</td><td align="char" char="." valign="top">0.435 (87)</td><td align="char" char="." valign="top">0.27 (54)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="char" char="." valign="top">0.78 (156)</td><td align="char" char="." valign="top">0.41 (82)</td><td align="char" char="." valign="top">0.26 (52)</td></tr><tr><td align="left" valign="top" rowspan="4">Three causal variants (<italic>S</italic> = 3)</td><td align="left" valign="top">SNR = 0.053</td><td align="char" char="." valign="top">0.565 (113)</td><td align="char" char="." valign="top">0.37 (74)</td><td align="char" char="." valign="top">0.245 (49)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="char" char="." valign="top">0.655 (131)</td><td align="char" char="." valign="top">0.36 (72)</td><td align="char" char="." valign="top">0.265 (53)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="char" char="." valign="top">0.72 (144)</td><td align="char" char="." valign="top">0.39 (78)</td><td align="char" char="." valign="top">0.22 (44)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="char" char="." valign="top">0.785 (157)</td><td align="char" char="." valign="top">0.48 (96)</td><td align="char" char="." valign="top">0.255 (51)</td></tr></tbody></table></table-wrap><table-wrap id="app12table3" position="float"><label>Appendix 12—table 3.</label><caption><title>Stable and Top SuSiE matching frequencies.</title><p>Below reports the frequencies with which Stable and Top SuSiE have matching variants for the same potential set. The numbers of matching variants for each SNR/‘No. Causal Variants’ scenario are reported in the parentheses.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top" colspan="5">Stratified by signal-to-noise ratio (SNR) and No. Causal Variants (<italic>S</italic>) in simulations</th></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top" rowspan="4">One causal variant (<italic>S</italic> = 1)</td><td align="left" valign="top">SNR = 0.053</td><td align="left" valign="top">0.55 (110)</td><td align="left" valign="top">0.115 (23)</td><td align="left" valign="top">0.095 (19)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="left" valign="top">0.725 (145)</td><td align="left" valign="top">0.14 (28)</td><td align="left" valign="top">0.095 (19)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="left" valign="top">0.84 (168)</td><td align="left" valign="top">0.145 (29)</td><td align="left" valign="top">0.17 (34)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="left" valign="top">0.875 (175)</td><td align="left" valign="top">0.14 (28)</td><td align="left" valign="top">0.19 (38)</td></tr><tr><td align="left" valign="top" rowspan="4">Two causal variants (<italic>S</italic> = 2)</td><td align="left" valign="top">SNR = 0.053</td><td align="left" valign="top">0.505 (101)</td><td align="left" valign="top">0.095 (19)</td><td align="left" valign="top">0.065 (13)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="left" valign="top">0.73 (146)</td><td align="left" valign="top">0.14 (28)</td><td align="left" valign="top">0.095 (19)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="left" valign="top">0.875 (175)</td><td align="left" valign="top">0.33 (66)</td><td align="left" valign="top">0.105 (21)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="left" valign="top">0.875 (175)</td><td align="left" valign="top">0.425 (85)</td><td align="left" valign="top">0.115 (23)</td></tr><tr><td align="left" valign="top" rowspan="4">Three causal variants (<italic>S</italic> = 3)</td><td align="left" valign="top">SNR = 0.053</td><td align="left" valign="top">0.345 (69)</td><td align="left" valign="top">0.075 (15)</td><td align="left" valign="top">0.085 (17)</td></tr><tr><td align="left" valign="top">SNR = 0.111</td><td align="left" valign="top">0.68 (136)</td><td align="left" valign="top">0.23 (46)</td><td align="left" valign="top">0.145 (29)</td></tr><tr><td align="left" valign="top">SNR = 0.25</td><td align="left" valign="top">0.795 (159)</td><td align="left" valign="top">0.37 (74)</td><td align="left" valign="top">0.185 (37)</td></tr><tr><td align="left" valign="top">SNR = 0.667</td><td align="left" valign="top">0.85 (170)</td><td align="left" valign="top">0.585 (117)</td><td align="left" valign="top">0.25 (50)</td></tr></tbody></table></table-wrap><table-wrap id="app12table4" position="float"><label>Appendix 12—table 4.</label><caption><title>Off-diagonal matching frequencies and causal variant recovery.</title><p>Below reports the number of Stable and Top PICS non-matching variants that match across different, or ‘off-diagonal’, potential sets. Frequencies are computed across simulations with the same number of causal variants (<italic>S</italic> = 1, 2, or 3), with numbers along the yellow-shaded diagonal reporting the number of non-matching variants between the same potential sets. Each off-diagonal element reports both the number of matching variants for the pair of potential sets listed as well as the percentage of these matches that also correspond to the causal variant.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top" colspan="5">Simulations with one causal variant</th></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top" colspan="3">Top PICS potential set compared against</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top" rowspan="3">Stable PICS potential set</td><td align="left" valign="top">Potential Set 1</td><td style="background-color: #FFF176;">167</td><td align="char" char="." valign="top">5 (60%)</td><td align="char" char="." valign="top">3 (67%)</td></tr><tr><td align="left" valign="top">Potential Set 2</td><td align="char" char="." valign="top">4 (25%)</td><td style="background-color: #FFF176;">466</td><td align="char" char="." valign="top">104 (0.96%)</td></tr><tr><td align="left" valign="top">Potential Set 3</td><td align="char" char="." valign="top">0</td><td align="char" char="." valign="top">94 (0%)</td><td style="background-color: #FFF176;">601</td></tr><tr><th align="left" valign="top" colspan="5">Simulations with two causal variants</th></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top" colspan="3">Top PICS potential set compared against</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top" rowspan="3">Stable PICS potential set</td><td align="left" valign="top">Potential Set 1</td><td style="background-color: #FFF176;">241</td><td align="left" valign="top">24 (46%)</td><td align="left" valign="top">5 (60%)</td></tr><tr><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">29 (52%)</td><td style="background-color: #FFF176;">483</td><td align="left" valign="top">88 (10%)</td></tr><tr><td align="left" valign="top">Potential Set 3</td><td align="left" valign="top">9 (11%)</td><td align="left" valign="top">84 (13%)</td><td style="background-color: #FFF176;">606</td></tr><tr><th align="left" valign="top" colspan="5">Simulations with three causal variants</th></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top" colspan="3">Top PICS potential set compared against</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Potential Set 1</td><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">Potential Set 3</td></tr><tr><td align="left" valign="top" rowspan="3">Stable PICS potential set</td><td align="left" valign="top">Potential Set 1</td><td style="background-color: #FFF176;">255</td><td align="left" valign="top">29 (66%)</td><td align="left" valign="top">7 (14%)</td></tr><tr><td align="left" valign="top">Potential Set 2</td><td align="left" valign="top">30 (40%)</td><td style="background-color: #FFF176;">480</td><td align="left" valign="top">79 (18%)</td></tr><tr><td align="left" valign="top">Potential Set 3</td><td align="left" valign="top">4 (0%)</td><td align="left" valign="top">65 (12%)</td><td style="background-color: #FFF176;">603</td></tr></tbody></table></table-wrap><table-wrap id="app12table5" position="float"><label>Appendix 12—table 5.</label><caption><title>List of matching variants with low stable posterior probability.</title><p>Below summarizes the genes and potential sets for which Stable and Top PICS returned matching variants, along with SNP-level and fine-mapping features for interpretation. Five statistics are reported: posterior probability of the stable variant (Stable PP); posterior probability of the top variant (Top PP); posterior probability support size, defined as the number variants with positive probability (Support Size); the number of ancestry slices, including the ALL slice, for which the stable variant had positive posterior probability from running Stable PICS (Number of Slices); the maximum difference in allele frequency between any pair of subpopulations among YRI, TSI, GBR, FIN, and CEU (Max AF Difference).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="top">Potential Set</th><th align="left" valign="top">Gene</th><th align="left" valign="top">Matching variant</th><th align="left" valign="top">Stable PP</th><th align="left" valign="top">Top PP</th><th align="left" valign="top">Support size</th><th align="left" valign="top">Number of slices</th><th align="left" valign="top">Max AF Difference</th></tr></thead><tbody><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000134762.11</td><td align="left" valign="top">rs61731921</td><td align="char" char="." valign="top">0.0028</td><td align="char" char="." valign="top">0.76</td><td align="char" char="." valign="top">23</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.22</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000197847.8</td><td align="left" valign="top">rs7130955</td><td align="char" char="." valign="top">0.0075</td><td align="char" char="." valign="top">0.23</td><td align="char" char="." valign="top">45</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">0.18</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000255284.1</td><td align="left" valign="top">rs12224894</td><td align="char" char="." valign="top">0.0067</td><td align="char" char="." valign="top">0.65</td><td align="char" char="." valign="top">23</td><td align="char" char="." valign="top">6</td><td align="char" char="." valign="top">0.14</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000104442.5</td><td align="left" valign="top">rs6995242</td><td align="char" char="." valign="top">0.0092</td><td align="char" char="." valign="top">0.31</td><td align="char" char="." valign="top">42</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.34</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000146733.9</td><td align="left" valign="top">rs10239528</td><td align="char" char="." valign="top">0.0031</td><td align="char" char="." valign="top">0.53</td><td align="char" char="." valign="top">27</td><td align="char" char="." valign="top">5</td><td align="char" char="." valign="top">0.24</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000248468.1</td><td align="left" valign="top">rs9853505</td><td align="char" char="." valign="top">0.0099</td><td align="char" char="." valign="top">0.29</td><td align="char" char="." valign="top">39</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">0.43</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000122224.10</td><td align="left" valign="top">rs57449</td><td align="char" char="." valign="top">0.0089</td><td align="char" char="." valign="top">0.50</td><td align="char" char="." valign="top">25</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.31</td></tr><tr><td align="char" char="." valign="top">1</td><td align="left" valign="top">ENSG00000134262.8</td><td align="left" valign="top">rs17464525</td><td align="char" char="." valign="top">0.0030</td><td align="char" char="." valign="top">0.45</td><td align="char" char="." valign="top">23</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.15</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">ENSG00000216522.3</td><td align="left" valign="top">rs5751902</td><td align="char" char="." valign="top">0.0052</td><td align="char" char="." valign="top">1</td><td align="char" char="." valign="top">7</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">0.16</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">ENSG00000108592.9</td><td align="left" valign="top">rs9912201</td><td align="char" char="." valign="top">0.0022</td><td align="char" char="." valign="top">0.32</td><td align="char" char="." valign="top">32</td><td align="char" char="." valign="top">6</td><td align="char" char="." valign="top">0.27</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">ENSG00000134551.7</td><td align="left" valign="top">rs7315843</td><td align="char" char="." valign="top">0.0019</td><td align="char" char="." valign="top">0.58</td><td align="char" char="." valign="top">10</td><td align="char" char="." valign="top">5</td><td align="char" char="." valign="top">0.22</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">ENSG00000221947.3</td><td align="left" valign="top">rs3103860</td><td align="char" char="." valign="top">0.0018</td><td align="char" char="." valign="top">0.99</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.084</td></tr><tr><td align="char" char="." valign="top">2</td><td align="left" valign="top">ENSG00000081791.4</td><td align="left" valign="top">rs2270113</td><td align="char" char="." valign="top">0.0059</td><td align="char" char="." valign="top">0.77</td><td align="char" char="." valign="top">11</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.33</td></tr><tr><td align="char" char="." valign="top">3</td><td align="left" valign="top">ENSG00000140368.8</td><td align="left" valign="top">rs62027296</td><td align="char" char="." valign="top">0.0069</td><td align="char" char="." valign="top">0.29</td><td align="char" char="." valign="top">21</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.15</td></tr><tr><td align="char" char="." valign="top">3</td><td align="left" valign="top">ENSG00000254614.1</td><td align="left" valign="top">rs625750</td><td align="char" char="." valign="top">0.0017</td><td align="char" char="." valign="top">0.64</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">0.22</td></tr><tr><td align="char" char="." valign="top">3</td><td align="left" valign="top">ENSG00000133835.9</td><td align="left" valign="top">rs2451818</td><td align="char" char="." valign="top">0.0036</td><td align="char" char="." valign="top">0.40</td><td align="char" char="." valign="top">32</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">0.41</td></tr><tr><td align="char" char="." valign="top">3</td><td align="left" valign="top">ENSG00000158234.8</td><td align="left" valign="top">rs693293</td><td align="char" char="." valign="top">0.0069</td><td align="char" char="." valign="top">0.55</td><td align="char" char="." valign="top">24</td><td align="char" char="." valign="top">4</td><td align="char" char="." valign="top">0.13</td></tr></tbody></table></table-wrap><table-wrap id="app12table6" position="float"><label>Appendix 12—table 6.</label><caption><title>List of variant annotations with interpretations.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Functional annotation</th><th align="left" valign="bottom">Interpretation</th></tr></thead><tbody><tr><td align="left" valign="bottom">Distance to Canonical Transcription Start Site (TSS)</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">Percent CpG in 75 bp window centered on variant position</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">Percent GC in 75 bp window centered on variant position</td><td align="left" valign="bottom">-</td></tr><tr><td align="left" valign="bottom">CTCF Binding Enrichment</td><td align="left" valign="bottom">Whether the variant lies within a CTCF binding site region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">Enhancer Enrichment</td><td align="left" valign="bottom">Whether the variant lies within an enhancer region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">Open Chromatin Enrichment</td><td align="left" valign="bottom">Whether the variant lies within an open chromatin region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">Promoter Enrichment</td><td align="left" valign="bottom">Whether the variant lies within a promoter region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">TF Binding Enrichment</td><td align="left" valign="bottom">Whether the variant lies within a TF-binding site region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">Promoter Flanking Enrichment</td><td align="left" valign="bottom">Whether the variant lies within a promoter flanking region as predicted by Ensembl</td></tr><tr><td align="left" valign="bottom">CADD (2 scores)</td><td align="left" valign="bottom">Whether the variant is likely to be simulated or not, and hence likely deleterious or not. One score is raw while the other is rank-normalized</td></tr><tr><td align="left" valign="bottom">SIFTVal</td><td align="left" valign="bottom">Whether the variant affects protein function, and hence deleterious</td></tr><tr><td align="left" valign="bottom">Polyphen2</td><td align="left" valign="bottom">Posterior probability that the variant is damaging</td></tr><tr><td align="left" valign="bottom">LINSIGHT</td><td align="left" valign="bottom">Probability that the variant site is under selection, thus having functional consequence</td></tr><tr><td align="left" valign="bottom">PhyloP (3 scores)</td><td align="left" valign="bottom">Substitution rates measuring cross-species evolutionary conservation at the site of the variant. Each score is computed with respect to a clade (vertebrate, mammal, primate)</td></tr><tr><td align="left" valign="bottom">GerpN</td><td align="left" valign="bottom">Estimated neutral substitution rate at variant position, with higher value implying greater conservation</td></tr><tr><td align="left" valign="bottom">GerpS</td><td align="left" valign="bottom">Estimated rejected substitution rate at variant position, with positive value implying a deficit in substitutions</td></tr><tr><td align="left" valign="bottom">B Statistic</td><td align="left" valign="bottom">Background selection at variant position, with smaller value indicating larger impact of selection</td></tr><tr><td align="left" valign="bottom">FATHMM-XF</td><td align="left" valign="bottom">Integrative score measuring deleteriousness of the variant</td></tr><tr><td align="left" valign="bottom">Funseq2</td><td align="left" valign="bottom">Integrative score measuring deleteriousness of the variant</td></tr><tr><td align="left" valign="bottom">ALoft</td><td align="left" valign="bottom">Integrative score measuring loss of function associated with the variant</td></tr><tr><td align="left" valign="bottom">FIRE</td><td align="left" valign="bottom">Integrative score measuring deleteriousness of the variant</td></tr><tr><td align="left" valign="bottom">Magnitude of Effect on Enformer Track Prediction (177 tracks)</td><td align="left" valign="bottom">Change in prediction of a gene regulatory track when performing in silico mutagenesis on the variant in a 196,608 bp sequence</td></tr></tbody></table></table-wrap></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88039.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Young</surname><given-names>Alexander</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>University of California, Los Angeles</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Convincing</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Important</kwd></kwd-group></front-stub><body><p>This <bold>important</bold> study presents a methodologically rigorous framework for stability-guided fine-mapping, extending PICS and generalizing to methods such as SuSiE, supported by comprehensive simulations and functional enrichment analyses. The evidence is now <bold>convincing</bold>, demonstrating improved causal variant recovery and offering a robust alternative for cross-population fine-mapping. The approach will be of particular interest to statistical geneticists, computational biologists, and biomedical researchers who rely on fine-mapping to interpret genetic association signals.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88039.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Aw et al. have proposed that utilizing stability analysis can be useful for fine-mapping of cross populations. In addition, the authors have performed extensive analyses to understand the cases where the top eQTL and stable eQTL are the same or different via functional data.</p><p>Comments on revisions:</p><p>The authors have answered all my concerns.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88039.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Aw et al presents a new stability-guided fine-mapping method by extending the previously proposed PICS method. They applied their stability-based method to fine-map cis-eQTLs in the GEUVADIS dataset and compared it against residualization-based approaches. They evaluated the performance of the proposed method using publicly available functional annotations and demonstrated that the variants identified by their stability-based method show enrichment for these functional annotations.</p><p>The authors have substantially strengthened the manuscript by addressing the major concerns raised in the initial review. I acknowledge that they have conducted comprehensive simulation studies to show the performance of their proposed approach and that they have extended their approach to SuSiE (&quot;Stable SuSiE&quot;) to demonstrate the broader applicability of the stability-guided principle beyond PICS.</p><p>One remaining question is the interpretation of matching variants with very low stable posterior probabilities (~0), which the authors have analyzed in detail but without fully conclusive findings. I agree with the authors that this event is relatively rare and the current sample size is limited but this might be something to keep in mind for future studies.</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88039.3.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Aw</surname><given-names>Alan J</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Jin</surname><given-names>Lionel Chentian</given-names></name><role specific-use="author">Author</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01gmv5d77</institution-id><institution>McKinsey &amp; Company (United States)</institution></institution-wrap><addr-line><named-content content-type="city">New York</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Ioannidis</surname><given-names>Nilah</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Song</surname><given-names>Yun S</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the latest reviews:</p><disp-quote content-type="editor-comment"><p>&quot;One remaining question is the interpretation of matching variants with very low stable posterior probabilities (~0), which the authors have analyzed in detail but without fully conclusive findings. I agree with the authors that this event is relatively rare and the current sample size is limited but this might be something to keep in mind for future studies.&quot;</p></disp-quote><p><bold>Fine-mapping stability</bold> – <bold>on matching variants with very low stable posterior probability</bold></p><p>We thank Reviewer 2 for encouraging us to think more about how low stable posterior probability matching variants can be interpreted. We describe a few plausible interpretations, even though – as Reviewer 2 and we have both acknowledged – our present experiments do not point to a clear and conclusive account.</p><p>One explanation is that the locus captured by the variant might not be well-resolved, in the sense that many correlated variants exist around the locus. Thus, the variant itself is unlikely causal, but the set of variants in high LD with it may contain the true causal variant, or it's possible that the causal variant itself was not sequenced but lies in that locus. A comparison of LD patterns across ancestries at the locus would be helpful here.</p><p>Another explanation rests on the following observation. For a variant to be matching between top and stable PICS and to also have very small stable PP, it has to have the largest PP <italic>after residualization</italic> on the ALL slice but also have positive PP with gene expression on many other slices. In other words, failing to control for potential confounders shrinks the PP. If one assumes that the matching variant is truly causal, then our observation points to an example of negative confounding (aka <italic>suppressor effect</italic>). This can occur when the confounders (PCs) are correlated with allele dosage at the causal variant in a different direction than their correlation with gene expression, so that the crude association between unresidualized gene expression and causal variant allele dosage is biased toward 0.</p><p>Although our present study does not allow us to systematically confirm either interpretation – since we found that matching variants were depleted in causal variants in our simulations, violating the second argument, but we also found functional enrichment in analyses of GEUVADIS data though only 17 matching variants with low stable PP were reported – we believe a larger-scale study using larger cohort sizes (at least 1000 individuals per ancestry) and many more simulations (to increase yield of such cases) would be insightful.</p><p>———</p><p>The following is the authors’ response to the original reviews:</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1:</bold></p><p>Major comments:</p><p>(1) It would be interesting to see how much fine-mapping stability can improve the fine-mapping results in cross-population. One can simulate data using true genotype data and quantify the amount the fine-mapping methods improve utilizing the stability idea.</p></disp-quote><p>We agree, and have performed simulation studies where we assume that causal variants are shared across populations. Specifically, by mirroring the simulation approach described in Wang et al. (2020), we generated 2,400 synthetic gene expression phenotypes across 22 autosomes, using GEUVADIS gene expression metadata (i.e., gene transcription start site) to ensure largely <italic>cis</italic> expression phenotypes were simulated. We additionally generated 1,440 synthetic gene expression phenotypes that incorporate environmental heterogeneity, to motivate our pursuit of fine-mapping stability in the first place (see Response to Reviewer 2, Comment 6). These are described in Results section “Simulation study”:</p><p>We evaluated the performance of the PICS algorithm, specifically comparing the approach incorporating stability guidance against the residualization approach that is more commonly used — similar to our application to the real GEUVADIS data. We additionally investigated two ways of “combining” the residualization and stability guidance approaches: (1) running stability-guided PICS on residualized phenotypes; (2) prioritizing matching variants returned by both approaches. See Response to Reviewer 2, Comment 5.</p><disp-quote content-type="editor-comment"><p>(2) I would be very interested to see how other fine-mapping methods (FINEMAP, SuSiE, and CAVIAR) perform via the stability idea.</p></disp-quote><p>Thank you for this valuable comment. We ran SuSiE on the same set of simulated datasets. Specifically, we ran a version that uses residualized phenotypes (supposedly removing the effects of population structure), and also a version that incorporates stability. The second version is similar to how we incorporate stability in PICS. We investigated the performance of Stable SuSiE in a similar manner to our investigation of PICS. First we compared the performance relative to SuSiE that was run on residualized phenotypes. Motivated by our finding in PICS that prioritizing matching variants improves causal variant recovery, we did the same analysis for SuSiE. This analysis is described in Results section “Stability guidance improves causal variant recovery in SuSiE.”</p><p>We reported overall matching frequencies and causal variant recovery rates of top and stable variants for SuSiE in Figures 2C&amp;D.</p><p>Frequencies with which Stable and Top SuSiE variants match, stratified by the simulation parameters, are summarized in Supplementary File 2C (reproduced for convenience in Response to Reviewer 2, Comment 3). Causal variant recovery rates split by the number of causal variants simulated, and stratified by both signal-to-noise ratio and the number of credible sets included, are reported in Figure 2—figure supplements 16-18. We reproduce Figure 2—figure supplement 18 (three causal variants scenario) below for convenience. Analogous recovery rates for matching versus non-matching top or stable variants are reported in Figure 2—figure supplements 19, 21 and 23.</p><disp-quote content-type="editor-comment"><p>(3) I am a little bit concerned about the PICS's assumption about one causal variant. The authors mentioned this assumption as one of their method limitations. However, given the utility of existing fine-mapping methods (FINEMAP and SuSiE), it is worth exploring this domain.</p></disp-quote><p>Thank you for raising this fair concern. We explored this domain, by considering simulations that include two and three causal variants (see Response to Reviewer 2, Comment 3). We looked at how well PICS recovers causal variants, and found that each potential set largely does not contain more than one causal variant (Figure 2—figure supplements 20 and 22). This can be explained by the fact that PICS potential sets are constructed from variants with a minimum linkage disequilibrium to a focal variant. On the other hand, in SuSiE, we observed multiple causal variants appearing in lower credible sets when applying stability guidance (Figure 2—figure supplements 21 and 23). A more extensive study involving more fine-mapping methods and metrics specific to violation of the one causal variant assumption could be pursued in future work.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2:</bold></p><p>Aw et al. presents a new stability-guided fine-mapping method by extending the previously proposed PICS method. They applied their stability-based method to fine-map cis-eQTLs in the GEUVADIS dataset and compared it against what they call residualization-based method. They evaluated the performance of the proposed method using publicly available functional annotations and claimed the variants identified by their proposed stability-based method are more enriched for these functional annotations.</p><p>While the reviewer acknowledges the contribution of the present work, there are a couple of major concerns as described below.</p><p>Major:</p><p>(1) It is critical to evaluate the proposed method in simulation settings, where we know which variants are truly causal. While I acknowledge their empirical approach using the functional annotations, a more unbiased, comprehensive evaluation in simulations would be necessary to assess its performance against the existing methods.</p></disp-quote><p>Thank you for this point. We agree. We have performed a simulation study where we assume that causal variants are shared across populations (see response to Reviewer 1, Comment 1). Specifically, by mirroring the simulation approach described in Wang et al. (2020), we generated 2,400 synthetic gene expression phenotypes across 22 autosomes, using GEUVADIS gene expression metadata (i.e., gene transcription start site) to ensure <italic>cis</italic> expression phenotypes were simulated.</p><disp-quote content-type="editor-comment"><p>(2) Also, simulations would be required to assess how the method is sensitive to different parameters, e.g., LD threshold, resampling number, or number of potential sets.</p></disp-quote><p>Thank you for raising this point. The underlying PICS algorithm was not proposed by us, so we followed the default parameters set (LD threshold, r<sup>2</sup> = 0.5; see Taylor et al., 2021 Bioinformatics) to focus on how stability considerations will impact the existing fine-mapping algorithm. We attempted to derive the asymptotic joint distribution of the p-values, but it was too difficult. Hence, we used 500 permutations because such a large number would allow large-sample asymptotics to kick in. However, following your critical suggestion we varied the number of potential sets in our analyses of simulated data. We briefly mention this in the Results.</p><p>“In the Supplement, we also describe findings from investigations into the impact of including more potential sets on matching frequency and causal variant recovery…”</p><p>A detailed write-up is provided in Supplementary File 1 Section S2 (p.2):</p><p>“The number of credible or potential sets is a parameter in many fine-mapping algorithms. Focusing on stability-guided approaches, we consider how including more potential sets for stable fine-mapping algorithms affects both causal variant recovery and matching frequency in simulations…</p><p>Causal variant recovery. We investigate both Stable PICS and Stable SuSiE. Focusing first on simulations with one causal variant, we observe a modest gain in causal variant recovery for both Stable PICS and Stable SuSiE, most noticeably when the number of sets was increased from 1 to 2 under the lowest signal-to-noise ratio setting…”</p><p>We observed that increasing the number of potential sets helps with recovering causal variants for Stable PICS (Figure 2—figure supplements 13-15). This observation also accounts for the comparable power that Stable PICS has with SuSiE in simulations with low signal-to-noise ratio (SNR), when we increase the number of credible sets or potential sets (Figure 2—figure supplements 10-12).</p><disp-quote content-type="editor-comment"><p>(3) Given the previous studies have identified multiple putative causal variants in both GWAS and eQTL, I think it's better to model multiple causal variants in any modern fine-mapping methods. At least, a simulation to assess its impact would be appreciated.</p></disp-quote><p>We agree. In our simulations we considered up to three causal variants in <italic>cis</italic>, and evaluated how well the top three Potential Sets recovered all causal variants (Figure 2—figure supplements 13-15; Figure 2—figure supplement 15). We also reported the frequency of variant matches between Top and Stable PICS stratified by the number of causal variants simulated in Supplementary File 2B and 2C. Note Supplementary File 2C is for results from SuSiE fine-mapping; see Response to Reviewer 1, Comment 2.</p><p>Supplementary File 2B. Frequencies with which Stable and Top PICS have matching variants for the same potential set. For each SNR/ “No. Causal Variants” scenario, the number of matching variants is reported in parentheses.</p><p>Supplementary File 2C. Frequencies with which Stable and Top SuSiE have matching variants for the same credible set. For each SNR/ “No. Causal Variants” scenario, the number of matching variants is reported in parentheses.</p><disp-quote content-type="editor-comment"><p>(4) Relatedly, I wonder what fraction of non-matching variants are due to the lack of multiple causal variant modeling.</p></disp-quote><p>PICS handles multiple causal variants by including more potential sets to return, owing to the important caveat that causal variants in high LD cannot be statistically distinguished. For example, if one believes there are three causal variants that are not too tightly linked, one could make PICS return three potential sets rather than just one. To answer the question using our simulation study, we subsetted our results to just scenarios where the top and stable variants do not match. This mimics the exact scenario of having modeled multiple causal variants but still not yielding matching variants, so we can investigate whether these non-matching variants are in fact enriched in the true causal variants.</p><p>Because we expect causal variants to appear in some potential set, we specifically considered whether these non-matching causal variants might match along different potential sets across the different methods. In other words, we compared the stable variant with the top variant from another potential set for the other approach (e.g., Stable PICS Potential Set 1 variant vs Top PICS Potential Set 2 variant). First, we computed the frequency with which such pairs of variants match. A high frequency would demonstrate that, even if the corresponding potential sets do not have a variant match, there could still be a match between non-corresponding potential sets across the two approaches, which shows that multiple causal variant modeling boosts identification of matching variants between both approaches — regardless of whether the matching variant is in fact causal.</p><p>Low frequencies were observed. For example, when restricting to simulations where Top and Stable PICS Potential Set 1 variants did not match, about 2-3% of variants matched between the Potential Set 1 variant in Stable PICS and Potential Sets 2 and 3 variants in Top PICS; or between the Potential Set 1 variant in Top PICS and Potential Sets 2 and 3 variants in Stable PICS (Supplementary File 2D). When looking at non-matching Potential Set 2 or Potential Set 3 variants, we do see an increase in matching frequencies (between 10-20%) between Potential Set 2 variants and other potential set variants between the different approaches. However, these percentages are still small compared to the matching frequencies we observed between corresponding potential sets (e.g., for simulations with one causal variant this was 70-90% between Top and Stable PICS Potential Set 1, and for simulations with two and three causal variants this was 55-78% and 57-79% respectively).</p><p>We next checked whether these “off-diagonal” matching variants corresponded to the true causal variants simulated. Here we find that the causal variant recovery rate is mostly less than the corresponding rate for diagonally matching variants, which together with the low matching frequency suggests that the enrichment of causal variants of “off-diagonal” matching variants is much weaker than in the diagonally matching approach. In other words, the fraction of non-matching (causal) variants due to the lack of multiple causal variant modeling is low.</p><p>We discuss these findings in Supplementary File 1 Section S2 (bottom of p.2).</p><disp-quote content-type="editor-comment"><p>(5) I wonder if you can combine the stability-based and the residualization-based approach, i.e., using the residualized phenotypes for the stability-based approach. Would that further improve the accuracy or not?</p></disp-quote><p>This is a good idea, thank you for suggesting it. We pursued this combined approach on simulated gene expression phenotypes, but did not observe significant gains in causal variant recovery (Figure 2B; Figure 2—figure supplements 2, 13 and 15). We reported this Results “Searching for matching variants between Top PICS and Stable PICS improves causal variant Recovery.”</p><p>“We thus explore ways to combine the residualization and stability-driven approaches, by considering (i) combining them into a single fine-mapping algorithm (we call the resulting procedure Combined PICS); and (ii) prioritizing matching variants between the two algorithms. Comparing the performance of Combined PICS against both Top and Stable PICS, however, we find no significant difference in its ability to recover causal variants (Figure 2B)...”</p><p>However, we also confirmed in our simulations that prioritizing matching variants between the two approaches led to gains in causal variant recovery (Figure 2D; Figure 2—figure supplements 4, 19, 20 and 22). We reported this Results “Searching for matching variants between Top PICS and Stable PICS improves causal variant Recovery.”</p><p>“On the other hand, matching variants between Top and Stable PICS are significantly more likely to be causal. Across all simulations, a matching variant in Potential Set 1 is 2.5X as likely to be causal than either a non-matching top or stable variant (Figure 2D) — a result that was qualitatively consistent even when we stratified simulations by SNR and number of causal variants simulated (Figure 2—figure supplements 19, 20 and 22)...”</p><p>This finding is consistent with our analysis of real GEUVADIS gene expression data, where we reported larger functional significance of matching variants relative to non-matching variants returned by either Top of Stable PICS.</p><disp-quote content-type="editor-comment"><p>(6) The authors state that confounding in cohorts with diverse ancestries poses potential difficulties in identifying the correct causal variants. However, I don't see that they directly address whether the stability approach is mitigating this. It is hard to say whether the stability approach is helping beyond what simpler post-hoc QC (e.g., thresholding) can do.</p></disp-quote><p>Thank you for raising this fair point. Here is a model we have in mind. Gene expression phenotypes (Y) can be explained by both genotypic effects (G, as in genotypic allelic dosage) and the environment (E): Y = G + E. However, both G and E depend on ancestry (A), so that Y = G|A+E|A. Suppose that the causal variants are shared across ancestries, so that (G|A=<italic>a</italic>)=G for all ancestries <italic>a</italic>. Suppose however that environments are heterogeneous by ancestry: (E|A=<italic>a</italic>) = e(<italic>a</italic>) for some function e that depends non-trivially on <italic>a</italic>. This would violate the exchangeability of exogenous E in the full sample, but by performing fine-mapping on each ancestry stratum, the exchangeability of exogenous E is preserved. This provides theoretical justification for the stability approach.</p><p>We next turned to simulations, where we investigated 1,440 simulated gene expression phenotypes capturing various ways in which ancestry induces heterogeneity in the exogenous E variable (simulation details in Lines 576-610 of Materials and Methods). We ran Stable PICS, as well as a version of PICS that did not residualize phenotypes or apply the stability principle. We observed that (i) causal variant recovery performance was not significantly different between the two approaches (Figure 2—figure supplements 24-32); but (ii) disagreement between the approaches can be considerable, especially when the signal-to-noise ratio is low (Supplementary File 2A). For example, in a set of simulations with three causal variants, with SNR = 0.11 and E heterogeneous by ancestry by letting E be drawn from <italic>N</italic>(2σ,σ<sup>2</sup>) for only GBR individuals (rest are <italic>N</italic>(0,σ<sup>2</sup>)), there was disagreement between Potential Set 1 and 2 variants in 25% of simulations — though recovery rates were similar (Probability of recovering at least one causal variant: 75% for Plain PICS and 80% for Stable PICS). These points suggest that confounding in cohorts can reduce power in methods not adjusting or accounting for ancestral heterogeneity, but can be remedied by approaches that do so. We report this analysis in Results “Simulations justify exploration of stability guidance”</p><p>In the current version of our work, we have evaluated, using both simulations and empirical evidence, different ways to combine approaches to boost causal variant recovery. Our simulation study shows that prioritizing matching variants across multiple methods improves causal variant recovery. On GEUVADIS data, where we might not know which variants are causal, we already demonstrated that matching variants are enriched for functional annotations. Therefore, our analyses justify that the adverse consequence of confounding on reducing fine-mapping accuracy can be mitigated by prioritizing matching variants between algorithms including those that account for stability.</p><disp-quote content-type="editor-comment"><p>(7) For non-matching variants, I wonder what the difference of posterior probabilities is between the stable and top variants in each method. If the difference is small, maybe it is due to noise rather than signal.</p></disp-quote><p>We have reported differences in posterior probabilities returned by Stable and Top PICS for GEUVADIS data; see Figure 3—figure supplement 1. For completeness, we compute the differences in posterior probabilities and summarize these differences both as histograms and as numerical summary statistics.</p><p>Potential Set 1</p><p>- Number of non-matching variants = 9,921</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table1" position="float"><label>Author response table 1.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-0.999</td><td align="char" char="." valign="bottom">-0.342</td><td align="char" char="." valign="bottom">-0.107</td><td align="char" char="." valign="bottom">-0.117</td><td align="char" char="." valign="bottom">0.067</td><td align="char" char="." valign="bottom">0.949</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig1-v1.tif"/></fig><p>Potential Set 2</p><p>- Number of non-matching variants = 14,454</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table2" position="float"><label>Author response table 2.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-1.000</td><td align="char" char="." valign="bottom">-0.334</td><td align="char" char="." valign="bottom">-0.0741</td><td align="char" char="." valign="bottom">-0.0778</td><td align="char" char="." valign="bottom">0.162</td><td align="char" char="." valign="bottom">0.976</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig2" position="float"><label>Author response image 2.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig2-v1.tif"/></fig><p>Potential Set 3</p><p>- Number of non-matching variants = 16,814</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table3" position="float"><label>Author response table 3.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-0.998</td><td align="char" char="." valign="bottom">-0.327</td><td align="char" char="." valign="bottom">-0.0564</td><td align="char" char="." valign="bottom">-0.0629</td><td align="char" char="." valign="bottom">0.191</td><td align="char" char="." valign="bottom">0.968</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig3" position="float"><label>Author response image 3.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig3-v1.tif"/></fig><p>We also compared the difference in posterior probabilities between non-matching variants returned by Stable PICS and Top PICS for our 2,400 simulated gene expression phenotypes. Focusing on just Potential Set 1 variants, we find two equally likely scenarios, as demonstrated by two distinct clusters of points in a “posterior probability-posterior probability” plot. The first is, as pointed out, a small difference in posterior probability (points lying close to y=x). The second, however, reveals stable variants with very small posterior probability (of order 4 x 10<sup>–5</sup> to 0.05) but with a non-matching top variant taking on posterior probability well distributed along [0,1]. Moving down to Potential Sets 2 and 3, the distribution of pairs of posterior probabilities appears less clustered, indicating less tendency for posterior probability differences to be small (Figure 2—figure supplement 8).</p><p>Here are the histograms and numerical summary statistics.</p><p>Potential Set 1</p><p>- Number of non-matching variants = 663 (out of 2,400)</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table4" position="float"><label>Author response table 4.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-0.994</td><td align="char" char="." valign="bottom">-0.266</td><td align="char" char="." valign="bottom">-0.00645</td><td align="char" char="." valign="bottom">-0.109</td><td align="char" char="." valign="bottom">0.049</td><td align="char" char="." valign="bottom">0.924</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig4" position="float"><label>Author response image 4.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig4-v1.tif"/></fig><p>Potential Set 2</p><p>Number of non-matching variants = 1,429 (out of 2,400)</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table5" position="float"><label>Author response table 5.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-0.998</td><td align="char" char="." valign="bottom">-0.376</td><td align="char" char="." valign="bottom">-0.0944</td><td align="char" char="." valign="bottom">-0.121</td><td align="char" char="." valign="bottom">0.115</td><td align="char" char="." valign="bottom">0.903</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig5" position="float"><label>Author response image 5.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig5-v1.tif"/></fig><p>Potential Set 3</p><p>- Number of non-matching variants = 1,810 (out of 2,400)</p><p>- Table of Summary Statistics of (Stable Posterior Probability – Top Posterior Probability)</p><table-wrap id="sa3table6" position="float"><label>Author response table 6.</label><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Min</th><th valign="bottom">1st Qu.</th><th valign="bottom">Median</th><th valign="bottom">Mean</th><th valign="bottom">3rd Qu.</th><th valign="bottom">Max</th></tr></thead><tbody><tr><td align="char" char="." valign="bottom">-0.998</td><td align="char" char="." valign="bottom">-0.389</td><td align="char" char="." valign="bottom">-0.100</td><td align="char" char="." valign="bottom">-0.114</td><td align="char" char="." valign="bottom">0.146</td><td align="char" char="." valign="bottom">0.915</td></tr></tbody></table></table-wrap><p>- Histogram of (Stable Posterior Probability – Top Posterior Probability)</p><fig id="sa3fig6" position="float"><label>Author response image 6.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88039-sa3-fig6-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(8) It's a bit surprising that you observed matching variants with (stable) posterior probability ~ 0 (SFig. 1). What are the interpretations for these variants? Do you observe functional enrichment even for low posterior probability matching variants?</p></disp-quote><p>Thank you for this question. We have performed a thorough analysis of matching variants with very low stable posterior probability, which we define as having a posterior probability &lt; 0.01 (Supplementary File 1 Section S11). Here, we briefly summarize the analysis and key findings.</p><p>Analysis</p><p>First, such variants occur very rarely — only 8 across all three potential sets in simulations, and 17 across all three potential sets for GEUVADIS (the latter variants are listed in Supplementary 2E). We begin interpreting these variants by looking at allele frequency heterogeneity by ancestry, support size — defined as the number of variants with positive posterior probability in the ALL slice* — and the number of slices including the stable variant (i.e., the stable variant reported positive posterior probability for the slice).</p><p>*Note that the stable variant posterior probability need not be at least 1/(Support Size). This is because the algorithm may have picked a SNP that has a lower posterior probability in the ALL slice (i.e., not the top variant) but happens to appear in the most number of other slices (i.e., a stable variant).</p><p>For variants arising from simulations, because we know the true causal variants, we check if these variants are causal. For GEUVADIS fine-mapped variants, we rely on functional annotations to compare their relative enrichment against other matching variants that did not have very low stable posterior probability.</p><p>Findings</p><p>While we caution against generalizing from observations reported here, which are based on very small sample sizes, we noticed the following. In simulations, matching variants with very low stable posterior probability are largely depleted in causal variants, although factors such as the number of slices including the stable variant may still be useful. In GEUVADIS, however, these variants can still be functionally enriched. We reported three examples in Supplementary File 1 Section S11 (pp. 8-9 of Supplement), where the variants were enriched in either VEP or biologically interpretable functional annotations, and were also reported in earlier studies. We partially reproduce our report below for convenience.</p><p>“However, we occasionally found variants that stand out for having large functional annotation scores. We list one below for each potential set.</p><p>- Potential Set 1 reported the variant rs12224894 from fine-mapping ENSG00000255284.1 (accession code <italic>AP006621.3</italic>) in Chromosome 11. This variant stood out for lying in the promoter flanking region of multiple cell types and being relatively enriched for GC content with a 75bp flanking region. This variant has been reported as a cis eQTL for AP006632 (using whole blood gene expression, rather than lymphoblastoid cell line gene expression in this study) in a clinical trial study of patients with systemic lupus erythematosus (Davenport et al., 2018). Its nearest gene is GATD1, a ubiquitously expressed gene that codes for a protein and is predicted to regulate enzymatic and catabolic activity. This variant appeared in all 6 slices, with a moderate support size of 23.</p><p>- Potential Set 2 reported the variant rs9912201 from fine-mapping ENSG00000108592.9 (mapped to <italic>FTSJ3</italic>) in Chromosome 17. Its FIRE score is 0.976, which is close to the maximum FIRE score reported across all Potential Set 2 matching variants. This variant has been reported as a SNP in high LD to a GWAS hit SNP rs7223966 in a pan-cancer study (Gong et al., 2018). This variant appeared in all 6 slices, with a moderate support size of 32.</p><p>- Potential Set 3 reported the variant rs625750 from fine-mapping ENSG00000254614.1 (mapped to <italic>CAPN1-AS1</italic>, an RNA gene) in Chromosome 11. Its FIRE score is 0.971 and its B statistic is 0.405 (region under selection), which lie at the extreme quantiles of the distributions of these scores for Potential Set 3 matching variants with stable posterior probability at least 0.01. Its associated mutation has been predicted to affect transcription factor binding, as computed using several position weight matrices (Kheradpour and Kellis, 2014). This variant appeared in just 3 slices, possibly owing to the considerable allele frequency difference between ancestries (maximum AF difference = 0.22). However, it has a small support size of 4 and a moderately high Top PICS posterior probability of 0.64.</p><p>To summarize, our analysis of GEUVADIS fine-mapped variants demonstrates that matching variants with very low stable posterior probability could still be functionally important, even for lower potential sets, conditional on supportive scores in interpretable features such as the number of slices containing the stable variant and the posterior probability support size…”</p></body></sub-article></article>