<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">61812</article-id><article-id pub-id-type="doi">10.7554/eLife.61812</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Tools and Resources</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Medicine</subject></subj-group></article-categories><title-group><article-title>Reproducible analysis of disease space via principal components using the novel R package syndRomics</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-201094"><name><surname>Torres-Espín</surname><given-names>Abel</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9787-8738</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund7"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201796"><name><surname>Chou</surname><given-names>Austin</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-180305"><name><surname>Huie</surname><given-names>J Russell</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201797"><name><surname>Kyritsis</surname><given-names>Nikos</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-201798"><name><surname>Upadhyayula</surname><given-names>Pavan S</given-names></name><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-116435"><name><surname>Ferguson</surname><given-names>Adam R</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7102-1608</contrib-id><email>adam.ferguson@ucsf.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Weill Institute for Neurosciences, Brain and Spinal Injury Center (BASIC), University of California, San Francisco (UCSF)</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Neurological Surgery, University of California San Francisco (UCSF)</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>Zuckerberg San Francisco General Hospital and Trauma Center</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution>School of Medicine, University of California San Diego (UCSD)</institution><addr-line><named-content content-type="city">San Diego</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution>San Francisco VA Health Care System</institution><addr-line><named-content content-type="city">San Francisco</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Zaidi</surname><given-names>Mone</given-names></name><role>Reviewing Editor</role><aff><institution>Icahn School of Medicine at Mount Sinai</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Barton</surname><given-names>Matthias</given-names></name><role>Senior Editor</role><aff><institution>University of Zurich</institution><country>Switzerland</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>14</day><month>01</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e61812</elocation-id><history><date date-type="received" iso-8601-date="2020-08-05"><day>05</day><month>08</month><year>2020</year></date><date date-type="accepted" iso-8601-date="2021-01-13"><day>13</day><month>01</month><year>2021</year></date></history><permissions><copyright-statement>© 2021, Torres-Espín et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Torres-Espín et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-61812-v2.pdf"/><abstract><p>Biomedical data are usually analyzed at the univariate level, focused on a single primary outcome measure to provide insight into systems biology, complex disease states, and precision medicine opportunities. More broadly, these complex biological and disease states can be detected as common factors emerging from the relationships among measured variables using multivariate approaches. ‘Syndromics’ refers to an analytical framework for measuring disease states using principal component analysis and related multivariate statistics as primary tools for extracting underlying disease patterns. A key part of the syndromic workflow is the interpretation, the visualization, and the study of robustness of the main components that characterize the disease space. We present a new software package, <italic>syndRomics</italic>, an open-source R package with utility for component visualization, interpretation, and stability for syndromic analysis. We document the implementation of <italic>syndRomics</italic> and illustrate the use of the package in case studies of neurological trauma data.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>syndromics</kwd><kwd>disease pattern discovery</kwd><kwd>principal component analysis pca</kwd><kwd>nonlinear PCA</kwd><kwd>R package</kwd><kwd>multivariate analysis</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>NS106899</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>NS088475</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000738</institution-id><institution>Department of Veterans Affairs</institution></institution-wrap></funding-source><award-id>I01RX02245</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000738</institution-id><institution>Department of Veterans Affairs</institution></institution-wrap></funding-source><award-id>I01RX002787</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100005191</institution-id><institution>Craig H. Neilsen Foundation</institution></institution-wrap></funding-source><award-id>Special Project</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100008191</institution-id><institution>Wings for Life</institution></institution-wrap></funding-source><award-id>Special Project</award-id><principal-award-recipient><name><surname>Ferguson</surname><given-names>Adam R</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100008191</institution-id><institution>Wings for Life</institution></institution-wrap></funding-source><award-id>Individual Grant</award-id><principal-award-recipient><name><surname>Torres-Espín</surname><given-names>Abel</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>A tutorial and open-source software to aid in reproducible disease pattern detection using principal component analysis.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The goal of the burgeoning field of precision medicine is to understand complex disease states and provide opportunities for deep patient phenotyping and highly targeted therapeutics. Precision medicine requires an understanding of multidimensional disease states. Yet, the analysis of biomedical data remains largely univariate, with response variables considered individually and reports involving several distinct analyses. This analytical approach limits our interpretation of the complexity of a disease by not considering the shared information across variables and potentially contributing to irreproducibility due to statistical limitations of multiple comparison testing. Understanding the full set of interrelated disease features through multivariate statistics is the goal of the growing domain of 'syndromics' (<xref ref-type="bibr" rid="bib23">Ferguson et al., 2011</xref>). In particular, principal component analysis (PCA) and related multivariate statistics such as nonlinear PCA or factor analysis have been proposed as tools for extracting underlying factors or patterns (principal components [PCs]) reflecting disease states (<xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>; <xref ref-type="bibr" rid="bib31">Haefeli et al., 2017a</xref>; <xref ref-type="bibr" rid="bib32">Haefeli et al., 2017b</xref>; <xref ref-type="bibr" rid="bib49">Kutcher et al., 2013</xref>; <xref ref-type="bibr" rid="bib61">Nielson et al., 2014</xref>; <xref ref-type="bibr" rid="bib62">Nielson et al., 2015</xref>; <xref ref-type="bibr" rid="bib65">Panaretos et al., 2017</xref>; <xref ref-type="bibr" rid="bib69">Rosenzweig et al., 2010</xref>; <xref ref-type="bibr" rid="bib70">Rosenzweig et al., 2018</xref>; <xref ref-type="bibr" rid="bib71">Rosenzweig et al., 2019</xref>; <xref ref-type="bibr" rid="bib90">Zhang and Castelló, 2017</xref>). There are several other multivariate methods that could be used for multivariate pattern detection: other ordination and dimension reduction techniques, cluster analysis, discrimination analysis, or the plethora of more recent machine learning methods. The use of any of these methods has its advantages and pitfalls (<xref ref-type="bibr" rid="bib21">Everitt and Hothorn, 2011</xref>). We focus on PCA as being one of the most widely used method for pattern detection. PCA is a multivariate statistical procedure that allows for the generation of new uncorrelated variables, called PCs, as a weighted combination of the original variables (<xref ref-type="bibr" rid="bib1">Abdi and Williams, 2010</xref>; <xref ref-type="bibr" rid="bib37">Hotelling, 1933</xref>; <xref ref-type="bibr" rid="bib44">Jolliffe and Cadima, 2016</xref>). These components are ordered such that the first component explains the major source of variance in the data, the second component the second largest source of variance, etc. The extracted components reflect the interrelation between all the original variables or features, allowing for disease pattern detection, guiding in the interpretation of disease complex space and overcoming univariate analysis limitations.</p><p>Despite the extensive use of PCA in some subfields of biological research and the increasing use of PCA for disease pattern discovery, there is very limited information in the literature that can guide applied biomedical researchers about its implementation and interpretation. Here, we offer a practical guide to the application of PCA for the extraction of disease patterns that conform the disease space, with focus on reproducibility. By no means can we cover the extensive field of PCA in the present document. Rather, we aim to provide an introductory manual to extraction of reproducible disease patterns using multidimensional analytics, directed to biomedical researcher practitioners while pointing to additional relevant sources of information. We introduce a software package for the R programming language called <italic>syndRomics</italic>, implementing some of the tools described here. We will illustrate the analysis workflow and the use of the package in experimental data from case studies in neurotrauma.</p><p>The key steps in disease pattern detection by PCA are shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>. The <italic>syndRomics</italic> package offers functionalities that aid in these steps, building on the extensive PCA framework developed by the R open-source community. The package implements a novel visualization tool, the syndromic plot first published by <xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>, as well as functions to quickly generate two other publication-ready visualizations (a heatmap and a barmap). In addition, the package implements resampling strategies, providing data-driven approaches to analytical decision-making aimed to reduce researcher subjectivity and increase reproducibility. In particular, the package offers a function to extract metrics for component and variable significance by using nonparametric permutation methods (<xref ref-type="bibr" rid="bib50">Landgrebe et al., 2002</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>; <xref ref-type="bibr" rid="bib66">Peres-Neto et al., 2003</xref>), to inform component selection and component interpretation. Finally, the package incorporates functions to study component stability toward understanding the generalizability and robustness of the analysis (<xref ref-type="bibr" rid="bib14">Cattell and Baggaley, 1960</xref>; <xref ref-type="bibr" rid="bib13">Cattell et al., 1969</xref>; <xref ref-type="bibr" rid="bib57">Lorenzo-Seva and ten Berge, 2006</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Summary of the syndromic framework and analysis steps.</title><p>(<bold>A</bold>) The theoretical framework of syndromic analysis. The intersection between different outcome measures can create a multivariate measure (principal component if PCA is used) to explain different patterns of variance in the data. The conceptual union of three van diagram forms the core of the syndromic plot symbolizing the multidimensional measure. (<bold>B</bold>) The different steps of the workflow to using PCA such as for disease pattern analysis.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig1-v2.tif"/></fig></sec><sec id="s2" sec-type="results"><title>Results</title><p>We will describe the general steps to use PCA for syndromic analysis and illustrate the use of the <italic>syndRomic</italic> package along the analytical steps with two case studies of neurotrauma data. Details of the usage and implementation of the package and functions are described in the Materials and methods section. The full code reproducing the analysis can be found in the supplementary material. Code boxes in the text provide snippets illustrating the main sections of the code. The first case study is used as a tutorial to illustrate the steps of analysis; the second case study is discussed at the end of the results section. For the first case, we used a publicly available preclinical dataset on the Open Data Commons for Spinal Cord Injury (<ext-link ext-link-type="uri" xlink:href="http://odc-sci.org/">odc-sci.org</ext-link>) (<xref ref-type="bibr" rid="bib10">Callahan et al., 2017</xref>; <xref ref-type="bibr" rid="bib26">Fouad et al., 2019</xref>). We selected a subset of the dataset with accession number ODC-SCI: 26 (<xref ref-type="bibr" rid="bib25">Ferguson et al., 2018</xref>) that has been previously used for deriving the so-called spinal cord injury (SCI) syndromics (<xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>). The dataset contains 159 subjects (rats) that have been studied on different motor functional outcomes across time after cervical spinal cord injury. The subset chosen for the present analysis consists of 18 outcome variables measured at 6 weeks after injury. The included variables for this analysis are shown in <xref ref-type="table" rid="table1">Table 1</xref>. For additional details of these variables, see <xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>.</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>List of variables included in the first case study.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="top">Variable</th><th valign="top">Definition</th></tr></thead><tbody><tr><td>wtChng</td><td>Change of animal weight (grams) from day of Injury to 6 weeks post-injury</td></tr><tr><td>RFSL</td><td>CATWALK SYSTEM RightForelimb StrideLength at 6 weeks post-injury</td></tr><tr><td>LFSL</td><td>CATWALK SYSTEM LeftForelimb StrideLength at 6 weeks post-injury</td></tr><tr><td>RHSL</td><td>CATWALK SYSTEM RightHindlimb StrideLength at 6 weeks post-injury</td></tr><tr><td>LHSL</td><td>CATWALK SYSTEM LeftHindlimb StrideLength at 6 weeks post-injury</td></tr><tr><td>RFPA</td><td>CATWALK SYSTEM RightForelimb PrintArea at 6 weeks post-injury</td></tr><tr><td>LFPA</td><td>CATWALK SYSTEM LeftForelimb PrintArea at 6 weeks post-injury</td></tr><tr><td>RHPA</td><td>CATWALK SYSTEM RightHindlimb PrintArea at 6 weeks post-injury</td></tr><tr><td>LHPA</td><td>CATWALK SYSTEM LeftHindlimb PrintArea at 6 weeks post-injury</td></tr><tr><td>StepDistRF</td><td>CATWALK SYSTEM RightForelimb Step Distribution Deviation from 25% at 6 weeks post-injury</td></tr><tr><td>StepDistLF</td><td>CATWALK SYSTEM LeftForelimb Step Distribution Deviation from 25% at 6 weeks post-injury</td></tr><tr><td>StepDistRH</td><td>CATWALK SYSTEM RightHindlimb Step Distribution Deviation from 25% at 6 weeks post-injury</td></tr><tr><td>StepDistLH</td><td>CATWALK SYSTEM LeftHindlimb Step Distribution Deviation from 25% at 6 weeks post-injury</td></tr><tr><td>TotalSubscore</td><td>Total BBB Subscore at 6 weeks post-injury</td></tr><tr><td>BBB FergTrans</td><td>BBB Ferguson Transformation score 6 weeks post-injury</td></tr><tr><td>Groom</td><td>Grooming Score 6 weeks post-injury</td></tr><tr><td>PawPL</td><td>PawPlacement score 6 weeks post-injury</td></tr><tr><td>ForelimbOpenField</td><td>Forelimb openfield score at 6 weeks post-injury</td></tr></tbody></table></table-wrap><sec id="s2-1"><title>Step 1: Extracting PCA solution from the data</title><p>There is extensive literature on performing PCA (<xref ref-type="bibr" rid="bib1">Abdi and Williams, 2010</xref>; <xref ref-type="bibr" rid="bib44">Jolliffe and Cadima, 2016</xref>; <xref ref-type="bibr" rid="bib90">Zhang and Castelló, 2017</xref>). As a consideration, biomedical data aiming to capture the multivariate disease space usually contains variables of different types (i.e. categorical, continuous, etc.) and scales, known as ‘mixed-type’ data. Moreover, missing data is a common problem in biomedicine (<xref ref-type="bibr" rid="bib34">Hollestein and Carpenter, 2017</xref>; <xref ref-type="bibr" rid="bib46">Kaushal, 2014</xref>; <xref ref-type="bibr" rid="bib64">Nielson et al., 2020</xref>) that needs to be solved to be able to apply most standard PCA algorithms. Therefore, some pre-processing transformations are usually applied before performing PCA. For example, linear PCA is sensitive to the scale of variables, thus when applying a linear PCA to continuous variables of different units or scales, a common practice is to scale the data to unit variance first (i.e. equivalent to performing the PCA on the correlation matrix). The use of the package to conduct syndromics analysis from linear PCA is illustrated on the first case study. In cases of datasets with mixed data types and/or non-linear relationships between variables, nonlinear PCA with optimal scaling transformation (<xref ref-type="bibr" rid="bib54">Linting et al., 2007a</xref>; <xref ref-type="bibr" rid="bib58">Mair and Leeuw, 2019</xref>) has been previously used for disease pattern analysis (<xref ref-type="bibr" rid="bib70">Rosenzweig et al., 2018</xref>; <xref ref-type="bibr" rid="bib71">Rosenzweig et al., 2019</xref>). We used the syndRomics package to analyze patterns from a nonlinear PCA in the second case study. In cases with missing data, strategies such as data imputation or the use of PCA algorithms allowing missing values might be needed (<xref ref-type="bibr" rid="bib17">Dray and Josse, 2015</xref>). While missing values analysis and dealing with missingness is an extensive topic that is not covered in detail here (<xref ref-type="bibr" rid="bib72">Rubin, 1976</xref>; <xref ref-type="bibr" rid="bib8">Buuren, 2018</xref>), the chosen case studies do contain missing values and illustrate how the package can help to determine the stability of the PCs when imputing missing values (see component stability section).</p><p>Another consideration is selecting which variables to include in the analysis. For PCA of experimental data where there are stratifying factors (e.g. control vs. treatment), it is important to leave out variables that directly capture the variance of these factors, which would bias PCA results toward separating the experimental groups. This bias is problematic since in syndromic analysis, the goal is to find the relationship between variables describing different diseases states in an unsupervised (i.e. not guided by our design) manner. For instance, if treatment indicators are included and the variance between treatment groups is high, the PCA solution would directly capture the experimental design and confound the multivariate patterns.</p><p>The disease components can be used in subsequent analysis as multivariate outcomes or predictor indicators (<xref ref-type="bibr" rid="bib31">Haefeli et al., 2017a</xref>; <xref ref-type="bibr" rid="bib62">Nielson et al., 2015</xref>; <xref ref-type="bibr" rid="bib70">Rosenzweig et al., 2018</xref>; <xref ref-type="bibr" rid="bib71">Rosenzweig et al., 2019</xref>). PCA is used to extract the correlation structure between variables, generating new independent variables as linear combinations. Beyond the use of PCs as proxies for disease patterns, the PCs can help mitigate issues that might appear when analyzing several variables such as multicollinearity, overfitting, and multiple testing (<xref ref-type="bibr" rid="bib2">Altman and Krzywinski, 2018</xref>; <xref ref-type="bibr" rid="bib43">Johnson et al., 1973</xref>; <xref ref-type="bibr" rid="bib52">Lever et al., 2017</xref>).</p><p>The reader is referred to some materials of interest on considerations and limitations when conducting PCA and related methods for biomedical research (<xref ref-type="bibr" rid="bib42">Jiang and Eskridge, 2000</xref>; <xref ref-type="bibr" rid="bib47">Konishi, 2015</xref>; <xref ref-type="bibr" rid="bib60">Nguyen and Holmes, 2019</xref>; <xref ref-type="bibr" rid="bib90">Zhang and Castelló, 2017</xref>).</p><p>Case study: In the first case study, the goal is to run a linear PCA to study the motor function components 6 weeks after cervical spinal cord injury. This will summarize all motor function variables as a small set of independent components explaining different aspects of the motor behavior after an SCI. The data contains missing values (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>), and therefore we performed missing values analysis before continuing with the workflow. Typically, the first step in missing values analysis is to determine patterns of missingness and classify missing values as missing completely at random (MCAR), missing at random (MAR) or missing not at random (MNAR) (<xref ref-type="bibr" rid="bib72">Rubin, 1976</xref>). The type of missingness will guide the decision on which is an acceptable procedure to deal with missing values. For instance, deleting all subjects that contain at least one missing observation is common practice (aka listwise deletion or complete-case analysis), but it is only acceptable if missing values are MCAR. Otherwise, the robustness and proper estimation of the missing values can not be guaranteed (<xref ref-type="bibr" rid="bib73">Schafer and Graham, 2002</xref>; <xref ref-type="bibr" rid="bib8">Buuren, 2018</xref>). In the example data, subjects have been pooled together from different experiments. We know that the observed pattern of missingness (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>) is due to a set of animals where some of the outcome measures were not studied, suggesting that missing values are MNAR. We confirmed that missing values are not MCAR using a previously described test of MCAR (<xref ref-type="bibr" rid="bib41">Jamshidian and Jalal, 2010</xref>) implemented in the <italic>MissMech</italic> package in R (<xref ref-type="bibr" rid="bib40">Jamshidian et al., 2014</xref>), which rejected the hypothesis of MCAR missingness in our data. Thus, excluding subjects from the analysis is not justified. Instead, we have used multiple imputation through the <italic>mice</italic> R package (<xref ref-type="bibr" rid="bib9">Buuren and Groothuis-Oudshoorn, 2011</xref>) to generate 50 imputed datasets and pooled them using the mean of each observation. We will illustrate on the component stability section how the <italic>syndRomics</italic> package can be used to determine the robustness of multiple imputation for disease pattern analysis. We extracted the PCA solution of the pooled imputed data using the <italic>prcomp()</italic> function in R after centering and scaling the data to unit variance (R code box 1). Other similar functions in R or other software can be used.<code xml:space="preserve">R Code Box 1</code><code xml:space="preserve">pca&lt;-prcomp (pca_data, center = TRUE, scale. = TRUE).</code></p></sec><sec id="s2-2"><title>Step 2: Component selection: how many components to keep?</title><p>The first question, after running PCA for extracting the disease components is usually to determine how many PCs are relevant. As a general consideration, the PCs with lower eigenvalues (i.e. explain less variance) have a higher chance of representing noise in the data (<xref ref-type="bibr" rid="bib44">Jolliffe and Cadima, 2016</xref>), questioning their generality and value. The goal is to determine the minimal set of components that can be used to describe the disease space. Importantly, there is not a single, specific rule for this determination. A common method in PCA and related methods is the Scree test by <xref ref-type="bibr" rid="bib12">Cattell, 1966</xref>, where all PCs are ordered in descending rank by their eigenvalues, and PCs above the ‘elbow’ are retained. Another criterion is the eigenvalue greater than one rule which is applied to standardized PCAs (from the correlation matrix) with the criteria of only keeping PCs with an eigenvalue (i.e. the variance of a component) above 1 (<xref ref-type="bibr" rid="bib30">Guttman, 1954</xref>; <xref ref-type="bibr" rid="bib45">Kaiser, 1960</xref>). A more thorough description of these and others methods can be found elsewhere (<xref ref-type="bibr" rid="bib27">Glorfeld, 1995</xref>; <xref ref-type="bibr" rid="bib36">Horn, 1965</xref>; <xref ref-type="bibr" rid="bib81">Vitale et al., 2017</xref>; <xref ref-type="bibr" rid="bib92">Zwick and Velicer, 1986</xref>). Simulations have shown these methods (specially the eigenvalue greater than one rule) to be less robust than a re-sampling approach for selecting the number of relevant components (<xref ref-type="bibr" rid="bib92">Zwick and Velicer, 1986</xref>). The <italic>syndRomics</italic> package incorporates a nonparametric permutation test approximated through Monte Carlo re-sampling of the total ‘variance accounted for’ (VAF) of each PC to aid in the selection of relevant PCs (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib27">Glorfeld, 1995</xref>; <xref ref-type="bibr" rid="bib36">Horn, 1965</xref>; <xref ref-type="bibr" rid="bib50">Landgrebe et al., 2002</xref>). The permutation test can also assist in component interpretation by studying the contribution of each variable to the PCA solution (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>) as we will see in the next section.</p><p>The goal of the permutation test is to determine whether the extracted PCs can be considered to be generated not-at-random. This method has been shown to outperform parametric tests for PCA in situations similar to biomedical data where sample sizes are relatively small and the data rarely comply with the assumptions of the models (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib36">Horn, 1965</xref>; <xref ref-type="bibr" rid="bib92">Zwick and Velicer, 1986</xref>). In that regard, a hypothesis test is defined as:</p><list list-type="simple"><list-item><p>H<sub>(null)</sub>:PC VAF is indistinguishable from a random generation</p></list-item><list-item><p>H<sub>(alternative)</sub>:PC VAF is different from random</p></list-item></list><p>The p values are calculated by:<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mi>p</mml:mi><mml:mo>=</mml:mo> <mml:mi/><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>q</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:math></disp-formula>where <inline-formula><mml:math id="inf1"><mml:mi>q</mml:mi></mml:math></inline-formula> is the number of times the chosen metric is higher in the permuted distribution than in the original PCA solution and <inline-formula><mml:math id="inf2"><mml:mi>P</mml:mi></mml:math></inline-formula> is the number of permutations (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>). Rejecting the null hypothesis is interpreted as evidence of the tested PC being generated from true signal and not by random noise. This sets a lower bound for which PCs to consider 'important' above noise, but does not indicate the magnitude of the 'importance', which is represented by VAF. Importantly, for datasets with several directions of variance and high signal-to-noise ratio, PCs with low VAF can still be statistically significant. The value of interpreting such PCs must be judged by the researcher in the context analysis in question. It is also important to consider how big <inline-formula><mml:math id="inf3"><mml:mi>P</mml:mi></mml:math></inline-formula> needs to be when performing re-sampling, such as with the permutation test incorporated in the package. The reader should note that the lowest <inline-formula><mml:math id="inf4"><mml:mi>p</mml:mi></mml:math></inline-formula> value that can be calculated is dependent on <inline-formula><mml:math id="inf5"><mml:mi>P</mml:mi></mml:math></inline-formula>. For example, if <inline-formula><mml:math id="inf6"><mml:mi>P</mml:mi></mml:math></inline-formula> is set to a value of 10 (a relatively low value), the smallest p value that can be detected is 0.09, which occurs when <inline-formula><mml:math id="inf7"><mml:mi>q</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>. Accordingly, <inline-formula><mml:math id="inf8"><mml:mi>P</mml:mi></mml:math></inline-formula> should be set high enough to reach the desired minimum p value. Moreover, simulation studies have shown that <inline-formula><mml:math id="inf9"><mml:mi>P</mml:mi></mml:math></inline-formula> under 99 have low power and a minimum of 499 permutations is recommended (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib1">Abdi and Williams, 2010</xref>; <xref ref-type="bibr" rid="bib53">Linting, 2007</xref>). By default, we have set the number of permutations to 1000 (smallest p value approximately equal to 0.001) as this has been shown to produce good results (<xref ref-type="bibr" rid="bib50">Landgrebe et al., 2002</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). Users of the package should keep in mind that higher numbers of permutations will increase computation time with potentially only a small gain on the approximation. Our simulations indicate that between 500 and 1000 permutations provide a good compromise between computing time and precision in estimating confidence intervals, depending on the data volume (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). The package implements a single permutation strategy for VAF, the so-called <italic>permD</italic> (permutation of the entire data set) (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>) where variables are permuted independently and concomitantly (<xref ref-type="fig" rid="fig2">Figure 2A</xref>) opposed to <italic>permV</italic> (permutation of a single variable) (<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>) where variables are permuted one at the time (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). These methods are further discussed on the component interpretation section.</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Implementation of permutation algorithms.</title><p>(<bold>A</bold>) Shows a schematic example of the permutation procedure <italic>permD</italic> where all the variables are permuted concomitantly but independently. (<bold>B</bold>) Shows a schematic example of the permutation procedure <italic>permV</italic> where variables are permuted one at the time for each permutation samples (<italic>P</italic>), keeping the other variables as in the original dataset. (<bold>C</bold>) The implemented algorithm for the permutation test algorithm using <italic>permD</italic>: each one to <italic>n</italic> permutation sample (<italic>P</italic>) consist on a random reorganization of observations inside each variable independently and concomitantly for each variable. For each <italic>P</italic> sample, a PCA is run and either the loadings, communalities or VAF are calculated. All <italic>P</italic> PCA solutions form the null distribution for non-parametric hypothesis testing of loadings or VAF. (<bold>D</bold>) The permutation test algorithm for loadings under <italic>permV</italic> is performed with and extra step of Procrustes rotation between each of the <italic>P</italic> samples to the parent component loadings. The <italic>P</italic> rotated loadings will then form the null distribution for each variable.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig2-v2.tif"/></fig><p><code xml:space="preserve">R Code Box 2.</code> </p><p><code xml:space="preserve">permut_pc_test (pca, pca_data, p=10000, ndim = 5, statistic = 'VAF', perm.method = 'permD').</code></p><p>Case study: After performing a PCA, we first determined the number of components that can be regarded as informative. Several criteria can be used as mentioned earlier. Here, we opted for the permutation test of VAF, computed using the <italic>permut_pc_test()</italic> function (R Code Box 2). We have applied this test to the data using 10,000 permutations. The results show that the three first PCs (PC1, PC2, and PC3) are significantly different from random at an alpha of 0.05 adjusting the p value (<xref ref-type="fig" rid="fig3">Figure 3</xref>), and therefore we will keep these three PCs for subsequent analysis. PC1 accounts for 32.9% of the variance, PC2 18.3% and PC3 9.8%.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Permutation test of case study.</title><p>(<bold>A</bold>) The graph shows the original VAF for the first five PCs and the average and 95% confidence interval VAF of the permuted PCA distribution (p=10000) using the <italic>permD</italic> method. * Statistical difference for the non-parametric test at alpha = 0.05 and adjusted p value by BH. The three first PCs were selected for the subsequent analysis. (<bold>B</bold>) Barmap of the original communalities (bars) and the permuted distribution (<italic>permV</italic>, p=3000) for each variable calculated over the first three PCs. (<bold>C</bold>) Barmap of the original loadings (bars) and the permuted distribution (<italic>permV</italic>, p=3000) for each variable and each of the first three PCs. Solid dotes represent the mean of the permuted distribution and error bars represent the 95% CI.</p><p><supplementary-material id="fig3sdata1"><label>Figure 3—source data 1.</label><caption><title>csv file containing the source data for panel A in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig3-data1-v2.csv"/></supplementary-material></p><p><supplementary-material id="fig3sdata2"><label>Figure 3—source data 2.</label><caption><title>csv file containing the source data for panel B in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig3-data2-v2.csv"/></supplementary-material></p><p><supplementary-material id="fig3sdata3"><label>Figure 3—source data 3.</label><caption><title>csv file containing the source data for panel C in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig3-data3-v2.csv"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Missing data analysis of the first case study.</title><p>(<bold>A</bold>) Shadow plot of missing data for the variables selected for the case study. Approximately 17% are missing values. Four patterns of missingness are observed as shown by the upset plot (<bold>B</bold>), three patterns with involving more than one variable and a pattern with a single variable. The fact that most missing values are across variables for the same subject (two biggest missing pattern sets) suggest data is missing at random (MAR), meaning there is an external reason to the observed values for that missing. In order to assess the stability of the PCA analysis by performing multiple imputation, we calculated the distribution of loadings generated by 50 multiple imputed datasets (<bold>C</bold>). The small variation around a pooled loading (average, bar) suggest a very small variation introduced by imputing the data, further corroborated by the component similarity measures for the first three PCs (<xref ref-type="table" rid="table4">Tables 4</xref>–<xref ref-type="table" rid="table6">6</xref>). Solid dots represent the mean of the multiple imputed loading distribution and error bars represent the 95% CI.</p><p><supplementary-material id="fig3s1sdata1"><label>Figure 3—figure supplement 1—source data 1.</label><caption><title>csv file for the <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig3-figsupp1-data1-v2.csv"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig3-figsupp1-v2.tif"/></fig></fig-group></sec><sec id="s2-3"><title>Step 3: Component interpretation: what do these components mean?</title><p>A key part of the analytical workflow is the interpretation of the main components, where the most relevant PCs can be used to represent the correlation between the original variables as a proxy for multivariate disease patterns. Each component is composed of a weighted combination of all the variables. Some components might be explained by only a few variables with high importance, whereas others might have several variables with important contributions to them. There are a few metrics that can be used for interpreting the relation between the original variables and the PCs (<xref ref-type="bibr" rid="bib1">Abdi and Williams, 2010</xref>). In the <italic>syndRomics</italic> package, we use the standardized loadings or correlation vector coefficients (<xref ref-type="bibr" rid="bib39">Jackson and Hearne, 1973</xref>), and the communalities, which are the sum of squared loadings for each variable across selected PCs representing how much of the variance of each variable can be explained by the total number of kept components. Loadings can be interpreted as the Pearson’s r correlation coefficient between a PC and a variable, and it is used to assess the contribution of individual variables on each PC and the direction on which the variable moves along the PC (i.e. opposite or same direction as in the interpretation of a correlation). Communalities can be interpreted as the global impact of a variable in the chosen PCA solution.</p><p>In general, the strategy consists of determining a threshold for the absolute value of loadings or the communalities above which variables are considered to have important contribution in the definition of a component or across the chosen PCs. For example, if a threshold of |loading| &gt; 0.2 is chosen, all variables for a given PC with a loading &gt; 0.2 or a loading &lt; −0.2 will be considered to contribute on the PC (aka salient variable). The matter then turns to determining an appropriate threshold. Some somewhat arbitrary rules of thumb for the loadings have been established. However, those have a strong determination in psychological studies and whether they are appropriate in biomedical research has yet to be verified. An alternative ‘quasi-inferential’ method is to use permutation test as discussed above for PC VAF but testing for metrics of variable contribution such as loadings (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib66">Peres-Neto et al., 2003</xref>) or communalities (<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). Using resampling strategies, these permutation methods offer data-driven determination of variable importance and contribution, which might reduce subjective biases. Thus, rejecting the null hypothesis for a given metric, variable and PC, suggest that such variable has a contribution onto the construction of the component that is above what is expected by random noise. As in the case of VAF, this establishes a lower bound for |loadings| or communalities below which they should be considered noise. In situations with stable solutions and high signal-to-noise ratio, low |loadings| or communalities might still be statistically significant, but the contribution of the variable should be gauged respect to other variables. In the package, we have incorporated permutation test of the loadings as in <xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib66">Peres-Neto et al., 2003</xref> that can serve to determine the loading threshold, where the variables are permuted independently and concomitantly (<xref ref-type="fig" rid="fig2">Figure 2A and C</xref>). Linting et al., designed and tested an strategy for the communalities where only one variable is permuted at the time, showing great results in determining the contribution of variables using communalities (<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). This method has resulted in better determination of the significant contribution of variables on the PCA solution with higher statistical power and proper type I error, and therefore has been incorporated in the package as the default method for both the communalities and the loadings (<xref ref-type="fig" rid="fig2">Figure 2B and D</xref>). Following Linting et al., terminology, users can specify the permutation strategy for the loadings as one variable at the time (<italic>permV,</italic> as in [<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>]) or as all the variable together (<italic>permD,</italic> as in [<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>; <xref ref-type="bibr" rid="bib66">Peres-Neto et al., 2003</xref>]). See Materials and methods for details on the permutation algorithms. In addition to permutation strategies, the package implements bootstrapping methods for constructing confidence intervals of component loadings and communalities that can also facilitate PCs interpretation (see component stability).</p><p>The selection of number of permutations in this case follows similar rationale as described above for the VAF. It is important to note that the minimal number of permutation needed to have enough statistical power and precision will depend on the size of the dataset, both on the number of variables and samples (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). There is also the understanding that while the <italic>permD</italic> strategy is less robust than <italic>permV</italic> as suggested by Linting et al., the computational time increases considerably since variables are permuted one at the time. Moreover, adjusting p values for multiple testing might be recommended depending on the sample size. Linting et al., suggested controlling for false discovery rate (FDR) using the Benjamini and Hochberg (BH) (<xref ref-type="bibr" rid="bib4">Benjamini and Hochberg, 1995</xref>) method. As a rule of thumb, these researchers advised to only use multiple testing correction (for FDR) when the data contains at least 20 variables and 100 observations or subjects, and to use the uncorrected p-values otherwise (<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). p-Value adjustment has been incorporated in the permutation function on the package, with controlling for FDR by BH as default.</p><p>The reader should be cautioned against overinterpreting or misinterpreting the meaning of a PC. The interpretation can be subjective, and unconscious biases can be reflected on the interpretation of PCs. The tools offered by the package help mitigate potential subjective biases, although data biases will affect the results. Another consideration is that it is possible that some of these metrics seem to ‘contradict’ each other. For example, there is the possibility that a component has an important contribution to the variance of the data (high VAF) and yet all the loadings be small. Contrary, a component with a small set of high loadings could be considered to be insignificant by permuting its VAF (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>). As in any analytical approach, domain knowledge is critical for the interpretation of disease components.</p><p>Case study: After deciding to keep three components, we studied the communalities and loadings to determine their identity. Here, we applied the <italic>permut_pc_test()</italic> function (R Code Box 3) setting the argument <italic>statistic</italic> = ‘commun’ or ‘s.loadings’ and the <italic>perm.method</italic> = ‘permV’ and using the BH method for controlling for FDR. The results of the permutation test on the communalities can be seen in <xref ref-type="fig" rid="fig3">Figure 3B</xref> and in <xref ref-type="table" rid="table3">Table 3</xref>. We can appreciate that all variables are significantly represented by the three chosen PCs, although there are five variables with communality less than 0.5, indicating that the retained PCs only explain 50% of the variance on these variables. In PCA, communalities can suggest which variables do or do not contribute to the extracted components altogether. Considering the loadings, the results for PC1, PC2, and PC3 are shown in <xref ref-type="fig" rid="fig3">Figure 3C</xref> and in <xref ref-type="table" rid="table4">Tables 4</xref>, <xref ref-type="table" rid="table5">5</xref> and <xref ref-type="table" rid="table6">6</xref>, respectively. One can appreciate that the cutoff loading for significance at alpha 0.05 using the adjusted p value is approximately |0.21| for PC1, |0.25| for PC2 and |0.4| for PC3. This behavior of different thresholds for significance has been previously described (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>) and reflects the fact that PCs accounting for less variance might contain more random noise, thus needing a higher loading for a variable to be considered as an important contributor. Loadings are indicative of both strength of association between a variable and a PC and the direction in which they interact. For reading on the interpretation of the loadings and components in this case study, see <xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>.</p><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>List of variables included in the second case study.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="top">Variable</th><th valign="top">Description</th><th valign="top">Values</th></tr></thead><tbody><tr><td valign="top">CT_Marshall</td><td valign="top">Marshall CT Score</td><td valign="top">Range from 1 to 6</td></tr><tr><td valign="top">CT_Rotterdam</td><td valign="top">Rotterdam CT Score</td><td valign="top">Range from 1 to 6</td></tr><tr><td valign="top">CT_brain_pathology</td><td valign="top">CT Brain Pathology</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_skull_FX</td><td valign="top">CT Skull Fracture</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_skullbase_FX</td><td valign="top">CT Skull Base Fracture</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_facial_FX</td><td valign="top">CT Facial Fracture</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_EDH</td><td valign="top">CT Epidural Hematoma</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_SDH</td><td valign="top">CT Subdural Hematoma</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_SAH</td><td valign="top">CT Subarachnoid Hemorrhage</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_contusion</td><td valign="top">CT Contusion</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_midlineshift</td><td valign="top">CT Midline Shift</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">CT_cisterncomp</td><td valign="top">CT Cisternal Compression</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">PTSD_diagnosis_6mo</td><td valign="top">PTSD DSM-IV Diagnosis (6 months)</td><td valign="top">0 = ‘No’, 1 = ‘Yes’</td></tr><tr><td valign="top">GOSE_3mo</td><td valign="top">GOSE Score (3 months)</td><td valign="top">Range from 1 to 8</td></tr><tr><td valign="top">GOSE_6mo</td><td valign="top">GOSE Score (6 months)</td><td valign="top">Range from 1 to 8</td></tr><tr><td valign="top">WAIS_PSI_6mo</td><td valign="top">WAIS PSI Composite Score (6 months)</td><td valign="top">Range from 50 to 150</td></tr><tr><td valign="top">CVLT_short_6mo</td><td valign="top">CVLT Short Delay Cued Recall Standard Score (6 months)</td><td valign="top">Range from −4.0–2.5</td></tr><tr><td valign="top">CVLT_long_6mo</td><td valign="top">CVLT Long Delay Cued Recall Standard Score (6 months)</td><td valign="top">Range from −3.5–2.5</td></tr><tr><td valign="top">SNP_COMT</td><td valign="top">COMT SNP Genotype</td><td valign="top">1 = ‘Met/Met’, 2 = ‘Met/Val’, 3 = ‘Val/Val’</td></tr><tr><td valign="top">SNP_DRD2</td><td valign="top">DRD2 SNP Genotype</td><td valign="top">1 = ‘C/C’, 2 = ‘C/T’, 3 = ‘T/T’</td></tr><tr><td valign="top">SNP_PARP1</td><td valign="top">PARP1 SNP Genotype</td><td valign="top">1 = ‘A/A’, 2 = ‘A/T’, 3 = ‘T/T’</td></tr><tr><td valign="top">SNP_ANKK1_Gly318Arg</td><td valign="top">ANKK1 SNP Gly318Arg</td><td valign="top">1 = ‘A/A’, 2 = ‘A/G’, 3 = ‘G/G’</td></tr><tr><td valign="top">SNP_ANKK1_Gly442Arg</td><td valign="top">ANKK1 SNP Gly442Arg</td><td valign="top">1 = ‘C/C’, 2 = ‘C/G’, 3 = ‘G/G’</td></tr><tr><td valign="top">SNP_ANKK1_Glu713Lys</td><td valign="top">ANKK1 SNP Glu713Lys</td><td valign="top">1 = ‘C/C’, 2 = ‘C/T’, 3 = ‘T/T’</td></tr></tbody></table></table-wrap><table-wrap id="table3" position="float"><label>Table 3.</label><caption><title>Communalities of first three PCs on permutation test with 3000 random permutations using <italic>permV</italic> and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th>Variable</th><th>Original communalities</th><th>Permuted average</th><th>Lower 95% CI</th><th>Upper 95% CI</th><th>p value</th><th>Adjusted p value</th></tr></thead><tbody><tr><td>wtChng</td><td valign="top">0.46</td><td valign="top">0.06</td><td valign="top">0.01</td><td valign="top">0.20</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>RFSL</td><td valign="top">0.59</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.20</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>RFPA</td><td valign="top">0.85</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>StepDistRF</td><td valign="top">0.54</td><td valign="top">0.05</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>LFSL</td><td valign="top">0.57</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.18</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>LFPA</td><td valign="top">0.61</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.20</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>StepDistLF</td><td valign="top">0.71</td><td valign="top">0.05</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>RHSL</td><td valign="top">0.88</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.18</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>RHPA</td><td valign="top">0.81</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.19</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>StepDistRH</td><td valign="top">0.27</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.18</td><td valign="top">0.0020</td><td valign="top">0.0020</td></tr><tr><td>LHSL</td><td valign="top">0.79</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.20</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>LHPA</td><td valign="top">0.79</td><td valign="top">0.07</td><td valign="top">0.00</td><td valign="top">0.23</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>StepDistLH</td><td valign="top">0.46</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.20</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>Groom</td><td valign="top">0.53</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>PawPL</td><td valign="top">0.70</td><td valign="top">0.06</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>BBB_FergTrans</td><td valign="top">0.66</td><td valign="top">0.05</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>TotalSubscore</td><td valign="top">0.40</td><td valign="top">0.05</td><td valign="top">0.00</td><td valign="top">0.17</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr><tr><td>ForelimbOpenField</td><td valign="top">0.37</td><td valign="top">0.05</td><td valign="top">0.00</td><td valign="top">0.18</td><td valign="top">0.0003</td><td valign="top">0.0004</td></tr></tbody></table></table-wrap><table-wrap id="table4" position="float"><label>Table 4.</label><caption><title>PC1 loading results of permutation test for the first case study with 3000 random permutations using <italic>permV</italic> and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">wtChng</td><td valign="bottom">−0.34</td><td valign="bottom">0.00</td><td valign="bottom">−0.19</td><td valign="bottom">0.18</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">TotalSubscore</td><td valign="bottom">−0.56</td><td valign="bottom">0.00</td><td valign="bottom">−0.21</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">StepDistRH</td><td valign="bottom">0.89</td><td valign="bottom">0.01</td><td valign="bottom">−0.20</td><td valign="bottom">0.21</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">StepDistRF</td><td valign="bottom">−0.65</td><td valign="bottom">0.00</td><td valign="bottom">−0.19</td><td valign="bottom">0.17</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">StepDistLH</td><td valign="bottom">−0.28</td><td valign="bottom">0.00</td><td valign="bottom">−0.18</td><td valign="bottom">0.18</td><td valign="bottom">0.0043</td><td valign="bottom">0.0084</td></tr><tr><td valign="bottom">StepDistLF</td><td valign="bottom">0.54</td><td valign="bottom">0.01</td><td valign="bottom">−0.18</td><td valign="bottom">0.18</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RHSL</td><td valign="bottom">−0.76</td><td valign="bottom">0.00</td><td valign="bottom">−0.19</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RHPA</td><td valign="bottom">−0.85</td><td valign="bottom">0.00</td><td valign="bottom">−0.17</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RFSL</td><td valign="bottom">0.74</td><td valign="bottom">0.01</td><td valign="bottom">−0.18</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RFPA</td><td valign="bottom">−0.25</td><td valign="bottom">0.00</td><td valign="bottom">−0.18</td><td valign="bottom">0.18</td><td valign="bottom">0.0063</td><td valign="bottom">0.0114</td></tr><tr><td valign="bottom">PawPL</td><td valign="bottom">−0.76</td><td valign="bottom">0.00</td><td valign="bottom">−0.17</td><td valign="bottom">0.17</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LHSL</td><td valign="bottom">0.62</td><td valign="bottom">0.00</td><td valign="bottom">−0.18</td><td valign="bottom">0.18</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LHPA</td><td valign="bottom">−0.24</td><td valign="bottom">0.00</td><td valign="bottom">−0.19</td><td valign="bottom">0.17</td><td valign="bottom">0.0163</td><td valign="bottom">0.0259</td></tr><tr><td valign="bottom">LFSL</td><td valign="bottom">0.38</td><td valign="bottom">0.01</td><td valign="bottom">−0.17</td><td valign="bottom">0.20</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LFPA</td><td valign="bottom">−0.54</td><td valign="bottom">−0.01</td><td valign="bottom">−0.21</td><td valign="bottom">0.20</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">Groom</td><td valign="bottom">0.49</td><td valign="bottom">0.00</td><td valign="bottom">−0.20</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">ForelimbOpenField</td><td valign="bottom">0.20</td><td valign="bottom">−0.01</td><td valign="bottom">−0.19</td><td valign="bottom">0.18</td><td valign="bottom">0.0323</td><td valign="bottom">0.0459</td></tr><tr><td valign="bottom">BBB_FergTrans</td><td valign="bottom">0.51</td><td valign="bottom">0.01</td><td valign="bottom">−0.18</td><td valign="bottom">0.19</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr></tbody></table></table-wrap><table-wrap id="table5" position="float"><label>Table 5.</label><caption><title>PC2 loading results of permutation test for the first case study with 3000 random permutations using <italic>permV</italic> and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">wtChng</td><td valign="bottom">−0.37</td><td valign="bottom">0.00</td><td valign="bottom">−0.23</td><td valign="bottom">0.22</td><td valign="bottom">0.0023</td><td valign="bottom">0.0047</td></tr><tr><td valign="bottom">TotalSubscore</td><td valign="bottom">−0.48</td><td valign="bottom">−0.01</td><td valign="bottom">−0.23</td><td valign="bottom">0.21</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">StepDistRH</td><td valign="bottom">−0.07</td><td valign="bottom">−0.01</td><td valign="bottom">−0.23</td><td valign="bottom">0.23</td><td valign="bottom">0.5122</td><td valign="bottom">0.5644</td></tr><tr><td valign="bottom">StepDistRF</td><td valign="bottom">0.28</td><td valign="bottom">0.00</td><td valign="bottom">−0.23</td><td valign="bottom">0.21</td><td valign="bottom">0.0143</td><td valign="bottom">0.0234</td></tr><tr><td valign="bottom">StepDistLH</td><td valign="bottom">−0.66</td><td valign="bottom">0.01</td><td valign="bottom">−0.23</td><td valign="bottom">0.24</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">StepDistLF</td><td valign="bottom">0.27</td><td valign="bottom">0.00</td><td valign="bottom">−0.22</td><td valign="bottom">0.23</td><td valign="bottom">0.0203</td><td valign="bottom">0.0305</td></tr><tr><td valign="bottom">RHSL</td><td valign="bottom">0.34</td><td valign="bottom">0.00</td><td valign="bottom">−0.24</td><td valign="bottom">0.23</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RHPA</td><td valign="bottom">−0.30</td><td valign="bottom">0.00</td><td valign="bottom">−0.21</td><td valign="bottom">0.21</td><td valign="bottom">0.0083</td><td valign="bottom">0.0145</td></tr><tr><td valign="bottom">RFSL</td><td valign="bottom">0.40</td><td valign="bottom">0.00</td><td valign="bottom">−0.22</td><td valign="bottom">0.25</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">RFPA</td><td valign="bottom">0.21</td><td valign="bottom">0.00</td><td valign="bottom">−0.22</td><td valign="bottom">0.22</td><td valign="bottom">0.0643</td><td valign="bottom">0.0868</td></tr><tr><td valign="bottom">PawPL</td><td valign="bottom">−0.30</td><td valign="bottom">0.00</td><td valign="bottom">−0.19</td><td valign="bottom">0.22</td><td valign="bottom">0.0023</td><td valign="bottom">0.0047</td></tr><tr><td valign="bottom">LHSL</td><td valign="bottom">0.42</td><td valign="bottom">0.01</td><td valign="bottom">−0.21</td><td valign="bottom">0.23</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LHPA</td><td valign="bottom">0.12</td><td valign="bottom">0.00</td><td valign="bottom">−0.24</td><td valign="bottom">0.22</td><td valign="bottom">0.3182</td><td valign="bottom">0.3656</td></tr><tr><td valign="bottom">LFSL</td><td valign="bottom">−0.62</td><td valign="bottom">0.00</td><td valign="bottom">−0.25</td><td valign="bottom">0.25</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LFPA</td><td valign="bottom">0.63</td><td valign="bottom">0.00</td><td valign="bottom">−0.24</td><td valign="bottom">0.26</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">Groom</td><td valign="bottom">−0.65</td><td valign="bottom">0.00</td><td valign="bottom">−0.22</td><td valign="bottom">0.23</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">ForelimbOpenField</td><td valign="bottom">−0.59</td><td valign="bottom">−0.01</td><td valign="bottom">−0.25</td><td valign="bottom">0.23</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">BBB_FergTrans</td><td valign="bottom">−0.32</td><td valign="bottom">−0.01</td><td valign="bottom">−0.22</td><td valign="bottom">0.22</td><td valign="bottom">0.0023</td><td valign="bottom">0.0047</td></tr></tbody></table></table-wrap><table-wrap id="table6" position="float"><label>Table 6.</label><caption><title>PC3 loading results of permutation test for the first case study with 3000 random permutations using <italic>permV</italic> and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">wtChng</td><td valign="bottom">0.46</td><td valign="bottom">0.00</td><td valign="bottom">−0.43</td><td valign="bottom">0.41</td><td valign="bottom">0.0183</td><td valign="bottom">0.0283</td></tr><tr><td valign="bottom">TotalSubscore</td><td valign="bottom">0.22</td><td valign="bottom">0.00</td><td valign="bottom">−0.32</td><td valign="bottom">0.34</td><td valign="bottom">0.2463</td><td valign="bottom">0.2955</td></tr><tr><td valign="bottom">StepDistRH</td><td valign="bottom">0.23</td><td valign="bottom">0.01</td><td valign="bottom">−0.32</td><td valign="bottom">0.35</td><td valign="bottom">0.2303</td><td valign="bottom">0.2826</td></tr><tr><td valign="bottom">StepDistRF</td><td valign="bottom">−0.19</td><td valign="bottom">0.01</td><td valign="bottom">−0.32</td><td valign="bottom">0.34</td><td valign="bottom">0.3102</td><td valign="bottom">0.3642</td></tr><tr><td valign="bottom">StepDistLH</td><td valign="bottom">0.23</td><td valign="bottom">0.00</td><td valign="bottom">−0.34</td><td valign="bottom">0.36</td><td valign="bottom">0.2083</td><td valign="bottom">0.2615</td></tr><tr><td valign="bottom">StepDistLF</td><td valign="bottom">0.50</td><td valign="bottom">−0.01</td><td valign="bottom">−0.36</td><td valign="bottom">0.40</td><td valign="bottom">0.0063</td><td valign="bottom">0.0114</td></tr><tr><td valign="bottom">RHSL</td><td valign="bottom">0.12</td><td valign="bottom">0.00</td><td valign="bottom">−0.35</td><td valign="bottom">0.33</td><td valign="bottom">0.5382</td><td valign="bottom">0.5698</td></tr><tr><td valign="bottom">RHPA</td><td valign="bottom">0.26</td><td valign="bottom">0.00</td><td valign="bottom">−0.35</td><td valign="bottom">0.35</td><td valign="bottom">0.1903</td><td valign="bottom">0.2446</td></tr><tr><td valign="bottom">RFSL</td><td valign="bottom">0.32</td><td valign="bottom">0.00</td><td valign="bottom">−0.36</td><td valign="bottom">0.41</td><td valign="bottom">0.1043</td><td valign="bottom">0.1374</td></tr><tr><td valign="bottom">RFPA</td><td valign="bottom">0.41</td><td valign="bottom">0.00</td><td valign="bottom">−0.34</td><td valign="bottom">0.35</td><td valign="bottom">0.0223</td><td valign="bottom">0.0326</td></tr><tr><td valign="bottom">PawPL</td><td valign="bottom">0.35</td><td valign="bottom">−0.01</td><td valign="bottom">−0.37</td><td valign="bottom">0.35</td><td valign="bottom">0.0583</td><td valign="bottom">0.0807</td></tr><tr><td valign="bottom">LHSL</td><td valign="bottom">0.48</td><td valign="bottom">0.01</td><td valign="bottom">−0.39</td><td valign="bottom">0.44</td><td valign="bottom">0.0123</td><td valign="bottom">0.0208</td></tr><tr><td valign="bottom">LHPA</td><td valign="bottom">0.62</td><td valign="bottom">0.00</td><td valign="bottom">−0.38</td><td valign="bottom">0.35</td><td valign="bottom">0.0003</td><td valign="bottom">0.0007</td></tr><tr><td valign="bottom">LFSL</td><td valign="bottom">−0.05</td><td valign="bottom">0.00</td><td valign="bottom">−0.35</td><td valign="bottom">0.33</td><td valign="bottom">0.8001</td><td valign="bottom">0.8308</td></tr><tr><td valign="bottom">LFPA</td><td valign="bottom">−0.12</td><td valign="bottom">−0.01</td><td valign="bottom">−0.32</td><td valign="bottom">0.34</td><td valign="bottom">0.5302</td><td valign="bottom">0.5698</td></tr><tr><td valign="bottom">Groom</td><td valign="bottom">0.03</td><td valign="bottom">0.01</td><td valign="bottom">−0.34</td><td valign="bottom">0.34</td><td valign="bottom">0.8680</td><td valign="bottom">0.8844</td></tr><tr><td valign="bottom">ForelimbOpenField</td><td valign="bottom">−0.14</td><td valign="bottom">0.01</td><td valign="bottom">−0.32</td><td valign="bottom">0.34</td><td valign="bottom">0.4802</td><td valign="bottom">0.5402</td></tr><tr><td valign="bottom">BBB_FergTrans</td><td valign="bottom">0.00</td><td valign="bottom">0.02</td><td valign="bottom">−0.33</td><td valign="bottom">0.33</td><td valign="bottom">0.9900</td><td valign="bottom">0.9900</td></tr></tbody></table></table-wrap><p>R Code Box 3 <code xml:space="preserve">permut_pc_test (pca, pca_data, p=1000, ndim = 3, statistic = 's.loadings', perm.method = 'permV').</code><code xml:space="preserve">permut_pc_test (pca, pca_data, p=1000, ndim = 3, statistic = 'communa', perm.method = 'permV').</code></p></sec><sec id="s2-4"><title>Step 4: Component stability: how robust are the components?</title><p>The presence of a syndrome or disease pattern, represented by a component, should hold true regardless of variations in experiments or metrics meant to measure that same pattern. For example, two experiments with different subjects but the same collected variables should result in inferentially equivalent components if they are true features of the disease and not experimental artifacts. The sensitivity of PCs to experimental, metric, or other forms of variation is termed ‘component stability’. Components from different PCAs (from different experiments as an example) that are extremely similar are considered to be a stable, and characterizing component stability is important to determine the robustness of the initial PCA (<xref ref-type="bibr" rid="bib28">Guadagnoli and Velicer, 1988</xref>; <xref ref-type="bibr" rid="bib53">Linting, 2007</xref>). A robust PC would be largely unaffected by data variations (i.e. low sensitivity). The goal of the stability analysis is to determine such sensitivity.</p><p>Given that performing multiple replication experiments in biomedical research is not always possible, component stability can be approximated by resampling techniques such as bootstrapping (<xref ref-type="bibr" rid="bib3">Babamoradi et al., 2013</xref>; <xref ref-type="bibr" rid="bib54">Linting et al., 2007a</xref>; <xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>). Bootstrap methods for component stability have been extensively studied, but users should be aware of the limitations and advantages of these methods and their performance for component stability depending on the use case (<xref ref-type="bibr" rid="bib3">Babamoradi et al., 2013</xref>; <xref ref-type="bibr" rid="bib28">Guadagnoli and Velicer, 1988</xref>; <xref ref-type="bibr" rid="bib55">Linting et al., 2007b</xref>; <xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>).</p><p>The package implements functionalities to help study the component stability affected by data selection variability by implementing bootstrapping methods (<xref ref-type="bibr" rid="bib3">Babamoradi et al., 2013</xref>; <xref ref-type="bibr" rid="bib55">Linting et al., 2007b</xref>; <xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>) and stability metrics. The default method used in the package is the simple or ordinary bootstrap consisting of generating a new sample that has the same size (i.e. same number of subjects or observations) and same variables as the original data, but where the subjects have been randomly selected from the data with replacement (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). This process is repeated several times (here referred as b times) to generate a sample of bootstrapped data. In the first case example, each of the b bootstrapped samples contain 159 subjects and 18 variables, but one subject might appear more than once and another subject might not show up in a specific sample.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Implementation of bootstrapping algorithm.</title><p>(<bold>A</bold>) shows a schematic of the bootstrapping procedure where a bootstrap sample is generated by resampling the original samples as many times as there are samples in the original dataset but allowing for replacement. The bootstrapping algorithm for loadings is (<bold>B</bold>): for each of 1 to n bootstrap sample (<bold>b</bold>), run a PCA with the same specifications than the parent PCA on the original sample. The bootstrapping method (e.g. balanced bootstrap) can be specified with the <italic>sim</italic> argument passed to the <italic>boot()</italic> function of the boot R package. Then, the sample component loading is obtained from the PCA of the bootstrapped sample and a Procrustes rotation of the loading matrix is applied over the parent loading matrix to correct for PCA indeterminacies (<bold>C</bold>; see text). All <italic>b</italic> rotated loadings form the bootstrapped distribution of loadings. The component similarity of each <italic>b</italic> loading with the parent loading solution can be calculated to generate the bootstrapped distribution of component similarity. From these distributions, the average and confidence interval are estimated.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Computation time for resampling.</title><p>The <italic>pc_stability()</italic> and <italic>permut_pca_test()</italic> functions were run 10 times for different number of bootstrapped (<bold>A–C</bold>) and permuted (<bold>D–F</bold>) samples (10, 25, 50, 100, 250, 500, 750, 1000, 1500, or 2000) for two datasets with different sizes (<italic>n</italic>:rows x <italic>p</italic>:columns; 159 × 18 or 1590 × 54). The computation time (in seconds) increased linearly with the increase of samples, being the rate of increasement higher for the bigger dataset (<bold>A and D</bold>). The small margin of error for each condition (standard deviation) reflects the little effect of different runs (with different random generated numbers) on the computation time. The computed loading for a variable with high loading (~|0.75|) and another for low loading (~|0.25|) of the PC1 for each condition is shown in (<bold>B and E</bold>). As the sample size increases, the variability around the loading average decreases. The width of the 95% CI (based on t-distribution with 9 degrees of freedom) for each condition is shown (<bold>C and F</bold>) as measure of precision around the loading average estimate. The precision is smaller with the smaller size of the data, indicating that the uncertainty of the estimated averaged loading is affected by the data volume. The standard 1000 samples are a good compromise between computation time and precision of the estimated loadings for the big dataset, but smaller dataset might require bigger resamples.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig4-figsupp1-v2.tif"/></fig></fig-group><p>Component stability can be studied at the whole component level or at the level of the individual variables through the loadings and communalities. The package implements component similarity indexes (aka factor matching indexes)(<xref ref-type="bibr" rid="bib14">Cattell and Baggaley, 1960</xref>; <xref ref-type="bibr" rid="bib13">Cattell et al., 1969</xref>; <xref ref-type="bibr" rid="bib29">Guadagnoli and Velicer, 1991</xref>) as metrics to study the stability of PCs. These metrics can be used to determine the similarity between the different bootstrapped samples, to test the validity of the extracted component under two or more experimental conditions, to assess the multidimensional equivalence of two or more replication experiments, or to determine the impact of imputing missing values.</p><p>Case study: To understand the sensitivity of variables and components to experimental variations, we used the <italic>pc_stability()</italic> function with b = 1000 bootstrapped samples (R Code Box 4), setting the <italic>sim</italic> argument to ‘balanced’ to perform balanced bootstrapping. The function will return the average of the loadings and the specified similarity metric across all the b samples as well as the specified confidence interval. For this example, the 95% CI (accelerated and bias-corrected, see Materials and methods) and the bootstrapped average can be seen in <xref ref-type="fig" rid="fig5">Figure 5</xref>. In general, the original loadings are close to the bootstrapped average which indicates that the results are unbiased. Moreover, the confidence regions for most higher value loadings are reasonably small, suggesting that these loadings are stable to experimental variation. In addition, the similarity metrics for the three PCs suggest component stability, meaning that the composition of the components is also stable. The accepted values for these metrics indicating stability might vary by field and the metric of interest. Some indicative values are mentioned in the respective method section, but the user should be aware of subjective biases when determining a threshold for considering stability (<xref ref-type="bibr" rid="bib57">Lorenzo-Seva and ten Berge, 2006</xref>). Finally, the function will return the average and percentile CI of the communalities, which can be used to assess which variables are more stable in the selected PCA solution.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Principal component (PC) stability results of case study.</title><p>Barmap plot of the bootstrap distribution of loadings (<bold>A</bold>) and communalities (<bold>B</bold>) representing the average and the 95% confidence interval of 3000 bootstrapped samples for the first three PCs. Assessing the confidence region offers an indicator of the uncertainty of the estimated loadings for each variable on each PC. Solid dots represent the mean of the bootstrap distribution and error bars represent the 95% CI.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig5-v2.tif"/></fig><p><code xml:space="preserve">R Code Box 4.</code> </p><p><code xml:space="preserve">pc_stability (pca, pca_data, B = 1000, ndim = 3, s_cut_off = 0.1, test_similarity = T, similarity_metric = 'all', sim = 'balanced', barmap_plot = T).</code></p><p>Assessing the impact of imputation methods on introducing noise when dealing with missing data should be considered. As described earlier, we used multiple imputation to generate the dataset for analysis. Multiple imputation generates <italic>m</italic> complete datasets where the imputed values might vary, but the observed values are the same (in the case study <italic>m</italic> = 50). We used the stability analysis described above to determine the sensitivity of the PCA solution to variations introduced by imputing missing values. We calculated the similarity metrics between all the 50 imputed datasets for the first three PCs, as well as the loadings. We observed high similarities between the PCs obtained from the imputed datasets (<xref ref-type="table" rid="table7">Table 7</xref>) and the loadings showed narrow CIs (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). We concluded that multiple imputation has produced stable solutions with acceptable impact on both the component and variables. A future version of the package might include more robust methods for pooling and testing multiple imputation in PCA context (<xref ref-type="bibr" rid="bib80">van Ginkel and Kroonenberg, 2014</xref>). Altogether, the results suggest reliable and robust PCs extracted from the original data.</p><table-wrap id="table7" position="float"><label>Table 7.</label><caption><title>Similarity metrics of the first three PCs between 50 multiple imputed datasets for the first case study.</title><p>Silent cutoff for S index was set at |0.2|.</p></caption><table frame="hsides" rules="groups"><thead><tr><th valign="top"/><th colspan="2" valign="top">CC index</th><th colspan="2" valign="top">r index</th><th colspan="2" valign="top">RMSE</th><th colspan="2" valign="top">S index</th></tr><tr><th valign="top">PC</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th></tr></thead><tbody><tr><td valign="top">PC1</td><td valign="top">0.999</td><td valign="top">0.0003</td><td valign="top">0.999</td><td valign="top">0.0003</td><td valign="top">0.021</td><td valign="top">0.004</td><td valign="top">0.991</td><td valign="top">0.015</td></tr><tr><td valign="top">PC2</td><td valign="top">0.998</td><td valign="top">0.0005</td><td valign="top">0.998</td><td valign="top">0.0005</td><td valign="top">0.021</td><td valign="top">0.004</td><td valign="top">0.93</td><td valign="top">0.042</td></tr><tr><td valign="top">PC3</td><td valign="top">0.997</td><td valign="top">0.001</td><td valign="top">0.996</td><td valign="top">0.002</td><td valign="top">0.022</td><td valign="top">0.005</td><td valign="top">0.965</td><td valign="top">0.03</td></tr></tbody></table></table-wrap></sec><sec id="s2-5"><title>Step 5: Component visualization</title><p>Communicating the analysis is a necessary part of the workflow. Although we have included this at the end of the use case, visualization can be also used for aiding in component selection, component interpretation and component stability analysis. There are several ways a PCA solution can be visualized. Here, we describe the plots implemented in the <italic>syndRomics</italic> package.</p><p>We have coded three types of plots (syndromics plot, heatmap, and barmap) using the grammar of the graphics framework (<xref ref-type="bibr" rid="bib86">Wilkinson, 2005</xref>) implemented in R by the <italic>ggplot2</italic> package. This allow users to customize the plots using the rich landscape of the <italic>ggplot2</italic> universe. The syndromic plot was first published by <xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref> and represents PCs as the center of a Venn diagram (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), consisting of (1) a middle convex triangle displaying the ‘variance accounted for’ (VAF) for a given PC and (2) radial arrows pointing to the center of the triangle for each variable with a standardized loading above a certain threshold (<xref ref-type="fig" rid="fig6">Figure 6A–C</xref>). The width of each arrow and the color saturation are proportional to the magnitude of the standardized loading they represent. The color of each arrow additionally differentiates between positive or negative loadings (e.g. blue represents a loading of +1, red represents a loading of −1, and white represents a loading of 0). Syndromic plots are especially useful for conveying PC identity in an easy to understand, concise way for publication. Heatmap and barmap plots are alternative visualizations of the loadings beyond the syndromic plot. The major difference between these two plots and the syndromic plot is that both the barmap (<xref ref-type="fig" rid="fig5">Figure 5A</xref>) and heatmap (<xref ref-type="fig" rid="fig6">Figure 6D</xref>) plots display all variables (or a manually selected subset) instead of only the ones with loadings above a given threshold. The absolute loadings that exceed a cutoff threshold can be noted (e.g. by a star *). Moreover, in the case of barmap plots, the cutoff is represented in the graph by vertical lines. This is particularly useful when there are too many above-threshold variables, which would crowd the syndromic plot visualization, or when comparing loadings between PCs more easily. In addition, barmaps are useful for documenting the results of the resampling procedures since error bars can be used to represent the variation of the metrics over the resampling. The <italic>permut_pc_test</italic>() and <italic>pc_stability</italic>() functions return such plots.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Visualization of PCA solutions for syndromic analysis.</title><p>(<bold>A–C</bold>) show the layout of the PC1, PC2, and PC3 syndromic plot of variables |loadings| &gt; 0.45, respectively: arrows pointing the center of the plot representing the magnitude (arrow thickness and color saturation) and direction (color) of the loadings of selected variables. (<bold>D</bold>) illustrate an example of the same loading solution plotted by a heatmap. * Indicates variables with |loadings| &gt; 0.21, 0.25 or 0.4 for PC1, PC2, and PC3, respectively.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig6-v2.tif"/></fig><p>Case study (R Code Box 5): We can visualize the three selected PCs using the plotting functions in <italic>syndRomics</italic> (<xref ref-type="fig" rid="fig6">Figure 6</xref>). In this case, we chose to represent PC1, PC2 and PC3 using the syndromics plots (<xref ref-type="fig" rid="fig6">Figure 6A, B and C</xref>, respectably) using a cutoff threshold of |0.45|. Notice that this is higher than the threshold for significance found by the permutation analysis given the high number of variables. The full loading pattern of the three first PCs can be visualized by a heatmap (<xref ref-type="fig" rid="fig6">Figure 6D</xref>), where we have chosen a different cutoff for each PC (0.21, 0.25, and 0.4 for PC1, PC2, and PC3 respectively), or a Barmap (<xref ref-type="fig" rid="fig3">Figures 3B</xref> and <xref ref-type="fig" rid="fig5">5A</xref>). Barmaps can be obtained for the loadings (<italic>barmap_loadings()</italic>) or for the communalities (<italic>barmap_commun()</italic>).<code xml:space="preserve">R Code Box 5</code><code xml:space="preserve">syndromic_plot (pca, pca_data, cutoff = 0.45).</code><code xml:space="preserve">heatmap_loading (pca, pca_data, ndim = 3, cutoff = c(0.21,0.25,0.4), star_values = T, text_values = F).</code></p></sec><sec id="s2-6"><title>Case study 2</title><p>In the second case study, we used selected variables from the Transforming Research And Clinical Knowledge in Traumatic Brain Injury (TRACK-TBI) pilot study (<xref ref-type="bibr" rid="bib87">Yue et al., 2013</xref>) that were analyzed previously and made publicly available (<xref ref-type="bibr" rid="bib63">Nielson et al., 2017</xref>). The released dataset version contains 586 de-identified human subjects who were enrolled in the TRACK-TBI pilot study and the 26 selected variables previously analyzed (<xref ref-type="bibr" rid="bib63">Nielson et al., 2017</xref>). These variables are a subset of brain imaging results, outcome metrics and genetic polymorphism (<xref ref-type="table" rid="table2">Table 2</xref>). The goal is to describe patterns of association between these three categories of variables. A noticeable difference between this dataset and the one used in the first case study is that here we are dealing with a mixed type dataset, where some variables are continuous, some nominal and some ordinal. Therefore, we performed a version of nonlinear PCA that allows for the extraction of patterns in these kinds of data. The syndRomics package has been programmed to work with the results of the <italic>princals()</italic> function from the <italic>Gifi</italic> R package. The code for this analysis is found in the supplementary material.</p><p>Missing data analysis showed an overall 21.2% missingness distributed between the outcomes and genetic polymorphism variables (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>). With the exception of ‘MRI results’ that has high missingness (61.7% of the observations), all imaging variables are complete. ‘MRI results’ variable was excluded from the analysis. The subsequent test for MCAR suggest that there are 17 different patterns of missingness and that the hypothesis of MCAR can be rejected overall (p-value&lt;0.001). Thus, excluding subjects from the analysis is not justified (<xref ref-type="bibr" rid="bib73">Schafer and Graham, 2002</xref>; <xref ref-type="bibr" rid="bib8">Buuren, 2018</xref>). We instead performed 50 multiple imputations using the <italic>mice</italic> R package as in the first case study. The 50 imputed datasets where then aggregated to perform nonlinear PCA using <italic>princals()</italic> (see Materials and methods for details).</p><p>Permutation test of PC VAF suggests that the first 6 PCs contain information that can be regarded as significant above random chance. Although a deep analysis of these six PCs might be of interest, the first three PCs explain the major variance (25.6%, 10.6%, and 9.8%, respectively). Therefore, we focused on interpreting these for illustration purposes (<xref ref-type="fig" rid="fig7">Figure 7</xref> and <xref ref-type="table" rid="table8">Tables 8</xref>–<xref ref-type="table" rid="table10">10</xref>). The first PC significantly loaded highly on two genetic variants in opposite directions (SNP_DRD2 loading = −0.677, SNP_ANKK1_Gly318AR loading = 0.661) as well as outcomes of neuropsychological function at 6 months after TBI (CVLT_long loading = −0.614, CVLT_short loading = −0.542)(<xref ref-type="fig" rid="fig7">Figure 7A–C</xref>). All other variables also significantly loaded on to PC1, but with |loadings| ~ 0.3 (<xref ref-type="table" rid="table8">Tables 8</xref>–<xref ref-type="table" rid="table10">10</xref>) suggesting that their contribution in PC1 identity is less important. Lower values in CVLT (California Verbal Learning Test) suggest learning and memory impairments, which are well known after TBI. Given the negative loadings for the included CVLT measures (short and long recall), negative values in PC1 might reflect better CVLT outcomes at 6 months after TBI. The stability of the PC1 pattern to multiple imputation is relatively low, with higher loadings showing high variation (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>, <xref ref-type="table" rid="table11">Table 11</xref>), emphasizing the importance of studying stability of components to multiple imputation. Nonetheless, the bootstrapped loadings were stable (<xref ref-type="fig" rid="fig7">Figure 7B</xref>), and decay in CVLT performance after TBI has been previously associated to polymorphisms in DRD2 and ANKK1 genes (<xref ref-type="bibr" rid="bib22">Failla et al., 2015</xref>; <xref ref-type="bibr" rid="bib59">McAllister et al., 2008</xref>; <xref ref-type="bibr" rid="bib63">Nielson et al., 2017</xref>; <xref ref-type="bibr" rid="bib88">Yue et al., 2017</xref>), providing literature validation of PC1. The variables with higher positive loadings in PC2 were related to the imaging findings and negative loadings with global function outcomes at 3 and 6 months after TBI (GOSE score)(<xref ref-type="fig" rid="fig7">Figure 7</xref>. A, D). Lower scores in GOSE are indicative of lower global function and positive values in imaging findings are suggestive of a bigger or more noticeable brain damage. PC2 presented the higher stability to both resampling and to multiple imputation (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>, <xref ref-type="table" rid="table11">Table 11</xref>). Altogether, PC2 might be interpreted as a surrogate for ‘TBI severity’, where higher positive values would indicate higher brain damage with less function at 3 and 6 months after injury, a signature described in the previous analysis of this data (<xref ref-type="bibr" rid="bib63">Nielson et al., 2017</xref>). Finally, given the instability of PC3, with most loadings being considered non-significant by the permutation test and the high variance to multiple imputation (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>, <xref ref-type="table" rid="table11">Table 11</xref>), PC3 can not be interpreted with certainty, and we should not attempt its explanation.</p><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Analysis of case study 2 using non-linear PCA and the syndRomics package.</title><p>(<bold>A-B</bold>) show thebarmap plots for the loadings for the first three PCs with the 95% CI generated from 500 permutationand 1000 bootstrap resamples. (<bold>C-D</bold>) show the syndromic plots for the PC1 (VAF=25.8%) and PC2(VAF=10.6%) for |loading|&gt;0.4. Error bars represent the 95%CI of the resampling method.</p><p><supplementary-material id="fig7sdata1"><label>Figure 7—source data 1.</label><caption><title>csv file containing the source data for <xref ref-type="fig" rid="fig7">Figure 7</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig7-data1-v2.csv"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig7-v2.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>Missing data analysis of the second case study.</title><p>(<bold>A</bold>) Shadow plot of missing data for the variables selected for the case study. Approximately 22% are missing values. The fact that most missing values are across variables for the same subject (two biggest missing pattern sets) suggest data is missing at random (MAR), meaning there is an external reason to the observed values for that missing. In order to assess the stability of the PCA analysis by performing multiple imputation, we calculated the distribution of loadings generated by 50 multiple imputed datasets (<bold>B</bold>). Solid dots represent the mean of the multiple imputed loading distribution and error bars represent the 95% CI.</p><p><supplementary-material id="fig7s1sdata1"><label>Figure 7—figure supplement 1—source data 1.</label><caption><title>csv file containing the source data for <xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-61812-fig7-figsupp1-data1-v2.csv"/></supplementary-material></p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-61812-fig7-figsupp1-v2.tif"/></fig></fig-group><table-wrap id="table8" position="float"><label>Table 8.</label><caption><title>PC1 loading results of permutation test for the second case study with 500 random permutations using permV and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">CT_brain_pathology</td><td valign="bottom">0.440589</td><td valign="bottom">0.010978</td><td valign="bottom">−0.17606</td><td valign="bottom">0.202095</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_cisterncomp</td><td valign="bottom">0.1844</td><td valign="bottom">0.003484</td><td valign="bottom">−0.16947</td><td valign="bottom">0.209801</td><td valign="bottom">0.053892</td><td valign="bottom">0.061591</td></tr><tr><td valign="bottom">CT_contusion</td><td valign="bottom">0.382612</td><td valign="bottom">0.008019</td><td valign="bottom">−0.17053</td><td valign="bottom">0.187363</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_EDH</td><td valign="bottom">0.256368</td><td valign="bottom">0.015695</td><td valign="bottom">−0.16239</td><td valign="bottom">0.190167</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_facial_FX</td><td valign="bottom">0.267322</td><td valign="bottom">0.029918</td><td valign="bottom">−0.16465</td><td valign="bottom">0.202063</td><td valign="bottom">0.011976</td><td valign="bottom">0.015128</td></tr><tr><td valign="bottom">CT_Marshall</td><td valign="bottom">0.235242</td><td valign="bottom">0.003685</td><td valign="bottom">−0.16912</td><td valign="bottom">0.191229</td><td valign="bottom">0.00998</td><td valign="bottom">0.013307</td></tr><tr><td valign="bottom">CT_midlineshift</td><td valign="bottom">0.002539</td><td valign="bottom">−0.01082</td><td valign="bottom">−0.2005</td><td valign="bottom">0.190956</td><td valign="bottom">0.988024</td><td valign="bottom">0.988024</td></tr><tr><td valign="bottom">CT_Rotterdam</td><td valign="bottom">0.274699</td><td valign="bottom">0.010723</td><td valign="bottom">−0.17176</td><td valign="bottom">0.195646</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_SAH</td><td valign="bottom">0.376596</td><td valign="bottom">0.009848</td><td valign="bottom">−0.16185</td><td valign="bottom">0.191671</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_SDH</td><td valign="bottom">0.377542</td><td valign="bottom">0.018196</td><td valign="bottom">−0.15488</td><td valign="bottom">0.196538</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_skull_FX</td><td valign="bottom">0.404447</td><td valign="bottom">0.020049</td><td valign="bottom">−0.15398</td><td valign="bottom">0.198067</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CT_skullbase_FX</td><td valign="bottom">0.318179</td><td valign="bottom">0.019653</td><td valign="bottom">−0.18691</td><td valign="bottom">0.195547</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CVLT_long_6mo</td><td valign="bottom">−0.70622</td><td valign="bottom">−0.05218</td><td valign="bottom">−0.37582</td><td valign="bottom">0.295769</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">CVLT_short_6mo</td><td valign="bottom">−0.63345</td><td valign="bottom">−0.04306</td><td valign="bottom">−0.378</td><td valign="bottom">0.258602</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">GOSE_3mo</td><td valign="bottom">−0.32848</td><td valign="bottom">−0.00894</td><td valign="bottom">−0.17465</td><td valign="bottom">0.177456</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">GOSE_6mo</td><td valign="bottom">−0.25329</td><td valign="bottom">−0.00613</td><td valign="bottom">−0.19298</td><td valign="bottom">0.176343</td><td valign="bottom">0.003992</td><td valign="bottom">0.005988</td></tr><tr><td valign="bottom">PTSD_diagnosis_6mo</td><td valign="bottom">0.188743</td><td valign="bottom">0.013079</td><td valign="bottom">−0.15885</td><td valign="bottom">0.190364</td><td valign="bottom">0.041916</td><td valign="bottom">0.050299</td></tr><tr><td valign="bottom">SNP_ANKK1_Glu713Lys</td><td valign="bottom">0.358071</td><td valign="bottom">0.028077</td><td valign="bottom">−0.14647</td><td valign="bottom">0.190965</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly318Arg</td><td valign="bottom">0.613624</td><td valign="bottom">0.043104</td><td valign="bottom">−0.17747</td><td valign="bottom">0.244082</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly442Arg</td><td valign="bottom">−0.25194</td><td valign="bottom">−0.02327</td><td valign="bottom">−0.19756</td><td valign="bottom">0.16577</td><td valign="bottom">0.005988</td><td valign="bottom">0.008454</td></tr><tr><td valign="bottom">SNP_COMT</td><td valign="bottom">0.036687</td><td valign="bottom">0.001474</td><td valign="bottom">−0.17511</td><td valign="bottom">0.172543</td><td valign="bottom">0.696607</td><td valign="bottom">0.726894</td></tr><tr><td valign="bottom">SNP_DRD2</td><td valign="bottom">−0.62858</td><td valign="bottom">−0.05017</td><td valign="bottom">−0.24105</td><td valign="bottom">0.168359</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr><tr><td valign="bottom">SNP_PARP1</td><td valign="bottom">−0.05912</td><td valign="bottom">−0.00576</td><td valign="bottom">−0.18101</td><td valign="bottom">0.166475</td><td valign="bottom">0.512974</td><td valign="bottom">0.559608</td></tr><tr><td valign="bottom">WAIS_PSI_6mo</td><td valign="bottom">−0.36179</td><td valign="bottom">−0.0144</td><td valign="bottom">−0.19646</td><td valign="bottom">0.168075</td><td valign="bottom">0.001996</td><td valign="bottom">0.003194</td></tr></tbody></table></table-wrap><table-wrap id="table9" position="float"><label>Table 9.</label><caption><title>PC2 loading results of permutation test for the second case study with 500 random permutations using permV and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">CT_brain_pathology</td><td valign="bottom">0.662378</td><td valign="bottom">0.021655</td><td valign="bottom">−0.12084</td><td valign="bottom">0.158839</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_cisterncomp</td><td valign="bottom">0.709989</td><td valign="bottom">0.024823</td><td valign="bottom">−0.14181</td><td valign="bottom">0.178126</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_contusion</td><td valign="bottom">0.596311</td><td valign="bottom">0.01766</td><td valign="bottom">−0.12222</td><td valign="bottom">0.157785</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_EDH</td><td valign="bottom">0.253079</td><td valign="bottom">0.002735</td><td valign="bottom">−0.13662</td><td valign="bottom">0.135348</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_facial_FX</td><td valign="bottom">0.142602</td><td valign="bottom">0.000971</td><td valign="bottom">−0.12466</td><td valign="bottom">0.133574</td><td valign="bottom">0.041916</td><td valign="bottom">0.055888</td></tr><tr><td valign="bottom">CT_Marshall</td><td valign="bottom">0.809847</td><td valign="bottom">0.026415</td><td valign="bottom">−0.13438</td><td valign="bottom">0.173771</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_midlineshift</td><td valign="bottom">0.69605</td><td valign="bottom">0.034813</td><td valign="bottom">−0.12917</td><td valign="bottom">0.189917</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_Rotterdam</td><td valign="bottom">0.753498</td><td valign="bottom">0.01539</td><td valign="bottom">−0.13205</td><td valign="bottom">0.178638</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_SAH</td><td valign="bottom">0.689084</td><td valign="bottom">0.017617</td><td valign="bottom">−0.12425</td><td valign="bottom">0.155246</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_SDH</td><td valign="bottom">0.698728</td><td valign="bottom">0.017798</td><td valign="bottom">−0.12057</td><td valign="bottom">0.161901</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_skull_FX</td><td valign="bottom">0.493199</td><td valign="bottom">0.007764</td><td valign="bottom">−0.13921</td><td valign="bottom">0.161888</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CT_skullbase_FX</td><td valign="bottom">0.294691</td><td valign="bottom">0.003769</td><td valign="bottom">−0.12863</td><td valign="bottom">0.141375</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">CVLT_long_6mo</td><td valign="bottom">0.056095</td><td valign="bottom">0.019229</td><td valign="bottom">−0.21544</td><td valign="bottom">0.23499</td><td valign="bottom">0.674651</td><td valign="bottom">0.703983</td></tr><tr><td valign="bottom">CVLT_short_6mo</td><td valign="bottom">0.115663</td><td valign="bottom">0.020014</td><td valign="bottom">−0.20152</td><td valign="bottom">0.215121</td><td valign="bottom">0.353293</td><td valign="bottom">0.423952</td></tr><tr><td valign="bottom">GOSE_3mo</td><td valign="bottom">−0.43692</td><td valign="bottom">−0.00787</td><td valign="bottom">−0.14301</td><td valign="bottom">0.139508</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">GOSE_6mo</td><td valign="bottom">−0.40155</td><td valign="bottom">−0.00849</td><td valign="bottom">−0.14484</td><td valign="bottom">0.118386</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">PTSD_diagnosis_6mo</td><td valign="bottom">0.004807</td><td valign="bottom">−0.00863</td><td valign="bottom">−0.14863</td><td valign="bottom">0.140418</td><td valign="bottom">0.94012</td><td valign="bottom">0.94012</td></tr><tr><td valign="bottom">SNP_ANKK1_Glu713Lys</td><td valign="bottom">−0.26204</td><td valign="bottom">−0.01604</td><td valign="bottom">−0.15757</td><td valign="bottom">0.132164</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly318Arg</td><td valign="bottom">−0.28622</td><td valign="bottom">−0.01954</td><td valign="bottom">−0.15986</td><td valign="bottom">0.133637</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly442Arg</td><td valign="bottom">0.033308</td><td valign="bottom">−0.00308</td><td valign="bottom">−0.12662</td><td valign="bottom">0.129371</td><td valign="bottom">0.630739</td><td valign="bottom">0.688078</td></tr><tr><td valign="bottom">SNP_COMT</td><td valign="bottom">−0.0406</td><td valign="bottom">−0.00094</td><td valign="bottom">−0.14047</td><td valign="bottom">0.140783</td><td valign="bottom">0.588822</td><td valign="bottom">0.67294</td></tr><tr><td valign="bottom">SNP_DRD2</td><td valign="bottom">0.307457</td><td valign="bottom">0.023806</td><td valign="bottom">−0.11549</td><td valign="bottom">0.178777</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">SNP_PARP1</td><td valign="bottom">0.244246</td><td valign="bottom">0.000876</td><td valign="bottom">−0.14801</td><td valign="bottom">0.133131</td><td valign="bottom">0.001996</td><td valign="bottom">0.002818</td></tr><tr><td valign="bottom">WAIS_PSI_6mo</td><td valign="bottom">−0.13252</td><td valign="bottom">0.001279</td><td valign="bottom">−0.13167</td><td valign="bottom">0.134985</td><td valign="bottom">0.055888</td><td valign="bottom">0.070596</td></tr></tbody></table></table-wrap><table-wrap id="table10" position="float"><label>Table 10.</label><caption><title>PC3 loading results of permutation test for the second case study with 500 random permutations using permV and adjusting p values with BH.</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">Original loading</th><th valign="bottom">Permuted average</th><th valign="bottom">Lower 95% CI</th><th valign="bottom">Upper 95% CI</th><th valign="bottom">p value</th><th valign="top">Adjusted p value</th></tr></thead><tbody><tr><td valign="bottom">CT_brain_pathology</td><td valign="bottom">0.110149</td><td valign="bottom">0.002819</td><td valign="bottom">−0.30242</td><td valign="bottom">0.309376</td><td valign="bottom">0.528942</td><td valign="bottom">0.641916</td></tr><tr><td valign="bottom">CT_cisterncomp</td><td valign="bottom">−0.32449</td><td valign="bottom">−0.00334</td><td valign="bottom">−0.33379</td><td valign="bottom">0.305672</td><td valign="bottom">0.047904</td><td valign="bottom">0.13839</td></tr><tr><td valign="bottom">CT_contusion</td><td valign="bottom">0.060972</td><td valign="bottom">−0.00689</td><td valign="bottom">−0.31444</td><td valign="bottom">0.270176</td><td valign="bottom">0.698603</td><td valign="bottom">0.728977</td></tr><tr><td valign="bottom">CT_EDH</td><td valign="bottom">0.104859</td><td valign="bottom">−0.00125</td><td valign="bottom">−0.27352</td><td valign="bottom">0.321535</td><td valign="bottom">0.518962</td><td valign="bottom">0.641916</td></tr><tr><td valign="bottom">CT_facial_FX</td><td valign="bottom">0.371649</td><td valign="bottom">0.062687</td><td valign="bottom">−0.31692</td><td valign="bottom">0.355691</td><td valign="bottom">0.01996</td><td valign="bottom">0.07984</td></tr><tr><td valign="bottom">CT_Marshall</td><td valign="bottom">−0.24284</td><td valign="bottom">−0.01023</td><td valign="bottom">−0.3038</td><td valign="bottom">0.288171</td><td valign="bottom">0.129741</td><td valign="bottom">0.259481</td></tr><tr><td valign="bottom">CT_midlineshift</td><td valign="bottom">−0.30562</td><td valign="bottom">−0.01988</td><td valign="bottom">−0.32888</td><td valign="bottom">0.308785</td><td valign="bottom">0.063872</td><td valign="bottom">0.153293</td></tr><tr><td valign="bottom">CT_Rotterdam</td><td valign="bottom">−0.24002</td><td valign="bottom">−0.01377</td><td valign="bottom">−0.32884</td><td valign="bottom">0.292959</td><td valign="bottom">0.161677</td><td valign="bottom">0.284003</td></tr><tr><td valign="bottom">CT_SAH</td><td valign="bottom">0.16969</td><td valign="bottom">−0.0017</td><td valign="bottom">−0.30469</td><td valign="bottom">0.323896</td><td valign="bottom">0.347305</td><td valign="bottom">0.520958</td></tr><tr><td valign="bottom">CT_SDH</td><td valign="bottom">0.164146</td><td valign="bottom">0.011718</td><td valign="bottom">−0.33537</td><td valign="bottom">0.308091</td><td valign="bottom">0.339321</td><td valign="bottom">0.520958</td></tr><tr><td valign="bottom">CT_skull_FX</td><td valign="bottom">0.30911</td><td valign="bottom">0.013422</td><td valign="bottom">−0.28309</td><td valign="bottom">0.317483</td><td valign="bottom">0.047904</td><td valign="bottom">0.13839</td></tr><tr><td valign="bottom">CT_skullbase_FX</td><td valign="bottom">0.412507</td><td valign="bottom">0.027909</td><td valign="bottom">−0.29748</td><td valign="bottom">0.350075</td><td valign="bottom">0.005988</td><td valign="bottom">0.047904</td></tr><tr><td valign="bottom">CVLT_long_6mo</td><td valign="bottom">0.071561</td><td valign="bottom">0.007602</td><td valign="bottom">−0.26795</td><td valign="bottom">0.330045</td><td valign="bottom">0.662675</td><td valign="bottom">0.722918</td></tr><tr><td valign="bottom">CVLT_short_6mo</td><td valign="bottom">0.01478</td><td valign="bottom">0.003403</td><td valign="bottom">−0.29415</td><td valign="bottom">0.341327</td><td valign="bottom">0.922156</td><td valign="bottom">0.922156</td></tr><tr><td valign="bottom">GOSE_3mo</td><td valign="bottom">0.512173</td><td valign="bottom">0.027225</td><td valign="bottom">−0.3035</td><td valign="bottom">0.347452</td><td valign="bottom">0.001996</td><td valign="bottom">0.023952</td></tr><tr><td valign="bottom">GOSE_6mo</td><td valign="bottom">0.519654</td><td valign="bottom">0.030422</td><td valign="bottom">−0.28179</td><td valign="bottom">0.361045</td><td valign="bottom">0.001996</td><td valign="bottom">0.023952</td></tr><tr><td valign="bottom">PTSD_diagnosis_6mo</td><td valign="bottom">−0.29067</td><td valign="bottom">−0.02728</td><td valign="bottom">−0.31023</td><td valign="bottom">0.227907</td><td valign="bottom">0.051896</td><td valign="bottom">0.13839</td></tr><tr><td valign="bottom">SNP_ANKK1_Glu713Lys</td><td valign="bottom">−0.47272</td><td valign="bottom">−0.02347</td><td valign="bottom">−0.40038</td><td valign="bottom">0.368874</td><td valign="bottom">0.007984</td><td valign="bottom">0.047904</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly318Arg</td><td valign="bottom">−0.14092</td><td valign="bottom">−0.02104</td><td valign="bottom">−0.34164</td><td valign="bottom">0.27799</td><td valign="bottom">0.379242</td><td valign="bottom">0.5354</td></tr><tr><td valign="bottom">SNP_ANKK1_Gly442Arg</td><td valign="bottom">−0.34766</td><td valign="bottom">−0.02656</td><td valign="bottom">−0.38332</td><td valign="bottom">0.353987</td><td valign="bottom">0.083832</td><td valign="bottom">0.182907</td></tr><tr><td valign="bottom">SNP_COMT</td><td valign="bottom">−0.10171</td><td valign="bottom">0.001249</td><td valign="bottom">−0.26635</td><td valign="bottom">0.284647</td><td valign="bottom">0.53493</td><td valign="bottom">0.641916</td></tr><tr><td valign="bottom">SNP_DRD2</td><td valign="bottom">0.081296</td><td valign="bottom">0.013862</td><td valign="bottom">−0.2912</td><td valign="bottom">0.323114</td><td valign="bottom">0.61477</td><td valign="bottom">0.702595</td></tr><tr><td valign="bottom">SNP_PARP1</td><td valign="bottom">0.233007</td><td valign="bottom">0.010172</td><td valign="bottom">−0.29054</td><td valign="bottom">0.318385</td><td valign="bottom">0.165669</td><td valign="bottom">0.284003</td></tr><tr><td valign="bottom">WAIS_PSI_6mo</td><td valign="bottom">0.39706</td><td valign="bottom">0.017391</td><td valign="bottom">−0.27022</td><td valign="bottom">0.337713</td><td valign="bottom">0.011976</td><td valign="bottom">0.057485</td></tr></tbody></table></table-wrap><table-wrap id="table11" position="float"><label>Table 11.</label><caption><title>Similarity metrics of the first 3PCs between 50 multiple imputed datasets for the second case study.</title><p>Silent cutoff for S index was set at |0.2|.</p></caption><table frame="hsides" rules="groups"><thead><tr><th valign="top"/><th colspan="2" valign="top">CC index</th><th colspan="2" valign="top">r index</th><th colspan="2" valign="top">RMSE</th><th colspan="2" valign="top">S index</th></tr><tr><th valign="top">PC</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th><th valign="top">Mean</th><th valign="top">SD</th></tr></thead><tbody><tr><td valign="top">PC1</td><td valign="top">0.955</td><td valign="top">0.035</td><td valign="top">0.958</td><td valign="top">0.033</td><td valign="top">0.094</td><td valign="top">0.06</td><td valign="top">0.88</td><td valign="top">0.037</td></tr><tr><td valign="top">PC2</td><td valign="top">0.992</td><td valign="top">0.004</td><td valign="top">0.991</td><td valign="top">0.0054</td><td valign="top">0.056</td><td valign="top">0.039</td><td valign="top">0.93</td><td valign="top">0.04</td></tr><tr><td valign="top">PC3</td><td valign="top">0.87</td><td valign="top">0.097</td><td valign="top">0.874</td><td valign="top">0.097</td><td valign="top">0.133</td><td valign="top">0.127</td><td valign="top">0.71</td><td valign="top">0.06</td></tr></tbody></table></table-wrap></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Biomedical research needs more multivariate analytics to help realize the potential of precision medicine. While multiple variables are collected in typical preclinical experiments and clinical trials, univariate statistics continue to be the major analytical and decision-making approaches across the different biomedical fields, narrowing our understanding of the complexity of any disease. With the advent of ‘omics’, analytical approaches for high-dimensional data have started to become more prevalent for the analysis of biological data. Yet, outside the realm of medical bioinformatics, biomedical research continues to be, for most part univariate. The lack of multivariate approaches in analyzing biomedical data can cause biases and constraints to the interpretation of the results and contribute to the lack of reproducibility and bench-to-bedside translation (<xref ref-type="bibr" rid="bib23">Ferguson et al., 2011</xref>; <xref ref-type="bibr" rid="bib38">Huie et al., 2018</xref>).</p><p>The extraction of disease space, through the use of multivariate methods, can increase our understanding of complex relationships commonly present in biomedical data while preventing some of the issues associated with an excessive use of univariate analytics such as multiple comparison testing and associated false discoveries by chance (<xref ref-type="bibr" rid="bib4">Benjamini and Hochberg, 1995</xref>; <xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>; <xref ref-type="bibr" rid="bib48">Krzywinski and Altman, 2014</xref>). For example, it is common in biomedicine to measure several behavioral and histopathological outcomes that are analyzed independently at the univariate level. This approach increases the chance of false-positive results due to the accumulation of type I testing errors (<xref ref-type="bibr" rid="bib4">Benjamini and Hochberg, 1995</xref>; <xref ref-type="bibr" rid="bib18">Dunn, 1961</xref>; <xref ref-type="bibr" rid="bib48">Krzywinski and Altman, 2014</xref>). Although there are methods to correct for errors when running numerous tests such as multiple-testing correction, their use in biomedicine outside of bioinformatic analysis is scarce. Even when correcting for multiple testing, performing several univariate analyses limits our understanding since univariate analysis does not allow us to study and infer the relationship between measures that might capture different aspects of the matter of study. In our first example case study, several functional tests can be used to study the recovery of forelimb motor function after cervical spinal cord injury in animal models. Each test further contains multiple measures about particular aspects of recovery. Knowing the relationship between these measures through multivariate approaches can increase our understanding of the matter of study while reducing the burden of multiple testing (<xref ref-type="bibr" rid="bib24">Ferguson et al., 2013</xref>). Importantly, it is also possible that a single univariate test that does not produce significant results misses true biological effects, while a multivariate analysis including the same variables can find patterns and relationship between variables that are significant. Syndromic analysis is, therefore, a framework that uses multivariate analysis of biomedical data in a holistic way, aiming to reveal interactions within complex (patho)-physiological niches, that would be otherwise challenging to discern. Applying syndromic analysis to biomedical data will help uncover the complex relationships of variables and features that constitute different disease and biological states and ultimately accelerate research toward precision medicine.</p><p>The <italic>syndRomics</italic> package implements several functionalities for the visualization, the interpretation, and the analysis of the stability of principal components to facilitate the extraction and analysis of disease patterns. We have demonstrated its usage, showing the potential of the package to support PCA-based analysis in understanding disease complexity. Although the core functionalities of the package are included, future versions might also implement outputs from other PCA functions as inputs, such as those from the PCA functions in the <italic>FactoMineR</italic> package (<xref ref-type="bibr" rid="bib51">Lê et al., 2008</xref>) or the <italic>psych</italic> package (<xref ref-type="bibr" rid="bib68">Revelle, 2017</xref>), allowing for better integration to the PCA landscape in R. In addition, other algorithms of bootstrapping and permutation methods for PCA solutions could be incorporated to increase the options and better adapt to the specifics needs of the user (<xref ref-type="bibr" rid="bib35">Hong et al., 2006</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>; <xref ref-type="bibr" rid="bib81">Vitale et al., 2017</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>).</p><p>Here, we emphasize guidance and tools for robust determination of PCA-based disease patterns. We have incorporated resampling methods aiming to reduce subjective biases and to study the stability and generality of the analysis. Although we have shown the use of these functions in different contexts along the process, much more work can be done to extend <italic>syndRomics</italic>. For example, we demonstrated the stability of our analysis under multiple imputation, and future research could investigate number of multiple imputations or missing conditions necessary for stable disease pattern detection. In addition, visualization features of syndRomics may be extended to help interpret disease patterns resolved by other multivariate or machine learning tools involving structure coefficients or feature impact scores. The <italic>syndRomics</italic> resampling methods could also be used to estimate the sample size required for stable PCs in the context of syndromic analysis, allowing for sample planning. The implementations in the package are thus positioned to empower both biological and statistical research toward understanding complex biology and diseases.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Availability and requirements</title><p>The code to reproduce this analysis can be found in the supplementary material. The data for the first use case comes from the ODC-SCI (Open Data Commons for Spinal Cord Injury, RRID:<ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_016673">SCR_016673</ext-link>, <ext-link ext-link-type="uri" xlink:href="http://odc-sci.org">http://odc-sci.org</ext-link>), ODC-SCI:26 dataset (<ext-link ext-link-type="uri" xlink:href="https://scicrunch.org/odc-sci/about/odc-sci_26">https://scicrunch.org/odc-sci/about/odc-sci_26</ext-link>). The data for the second use case comes from TRACK-SCI and can be downloaded from <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0169490">10.1371/journal.pone.0169490</ext-link>. The package can be installed from GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/ucsf-ferguson-lab/syndRomics">https://github.com/ucsf-ferguson-lab/syndRomics</ext-link>) where installation instructions, package manual and examples of usage are provided. Descriptions of the arguments and function usage can be found in the internal package documentation once installed or in the package manual. The package has been implemented in R (<xref ref-type="bibr" rid="bib67">R Development Core Team, 2019</xref>) through RStudio (<xref ref-type="bibr" rid="bib75">Team RS, 2018</xref>) using a few other packages beyond the ones bundled in R as dependencies: <italic>dplyr</italic> (<xref ref-type="bibr" rid="bib83">Wickham et al., 2018</xref>), <italic>ggplot2</italic> (<xref ref-type="bibr" rid="bib82">Wickham, 2016</xref>), <italic>stringr</italic> (<xref ref-type="bibr" rid="bib84">Wickham, 2019</xref>), <italic>tidyr</italic> (<xref ref-type="bibr" rid="bib85">Wickham and Henry, 2020</xref>), <italic>ggrepel</italic> (<xref ref-type="bibr" rid="bib74">Slowikowski, 2019</xref>), <italic>ggnewscale</italic> (<xref ref-type="bibr" rid="bib20">Elio Campitelli, 2020</xref>), <italic>pracma</italic> (<xref ref-type="bibr" rid="bib5">Borchers, 2019</xref>), <italic>png</italic> (<xref ref-type="bibr" rid="bib79">Urbanek, 2013</xref>), <italic>boot</italic> (<xref ref-type="bibr" rid="bib11">Canty and Ripley, 2019</xref>; <xref ref-type="bibr" rid="bib16">Davison and Hinkley, 1997</xref>), <italic>rlang</italic> (<xref ref-type="bibr" rid="bib33">Henry and Wickham, 2020</xref>), and <italic>Gifi</italic> (<xref ref-type="bibr" rid="bib58">Mair and Leeuw, 2019</xref>).</p></sec><sec id="s4-2"><title>Package implementation</title><p>The <italic>syndRomics</italic> package offers two major functionalities for the purpose of aiding in the process of syndromics analysis: (1) visualization functions and (2) functions incorporating resampling methods to determine stability and inference of PCs.</p></sec><sec id="s4-3"><title>Visualization functions</title><p>The visualization functions are: syndromic_plot(), heatmap_loadings(), barmap_loadings(), barmap_commun() and VAF_plot(). For the visualization functions, the user can pass an R <italic>data.frame</italic> object with the standardized loadings (or other metrics) obtained by running PCA and related multivariate methods in their preferred software. We opted for this approach to avoid requiring specific implementations of PCA. Loadings obtained from any PCA solution can be easily formatted for usage with the <italic>syndRomics</italic> visualization functions. All functions in the package that takes a data frame as argument use the same format (<xref ref-type="table" rid="table12">Table 12</xref>): variables are organized as rows, and the first column is called ‘Variables’ and contains the names of the respective variables. The other columns contain the PC loadings and are named ‘PC1’, ‘PC2’, etc. Alternatively, the visualizations can also be obtained from the output of the <italic>prcomp()</italic> function in the <italic>stats</italic> package in R (linear PCA) or from the output of the <italic>princals()</italic> function in the <italic>Gifi</italic> package in R (non-linear PCA by categorical PCA). Finally, the results from <italic>pc_stability()</italic> and <italic>permut_pc_test()</italic> can be passed to the <italic>plot()</italic> generic function in R as the package incorporate the corresponding S3 method for ‘syndromics’ class object.</p><table-wrap id="table12" position="float"><label>Table 12.</label><caption><title>Template/example of data.frame containing loadings that can be passed to the visualization functions (only the loadings for the first three PCs are shown).</title></caption><table frame="hsides" rules="groups"><thead><tr><th valign="bottom">Variable</th><th valign="bottom">PC1</th><th valign="bottom">PC2</th><th valign="bottom">PC3</th></tr></thead><tbody><tr><td valign="bottom"> wtChng</td><td valign="bottom">−0.34</td><td valign="bottom">−0.37</td><td valign="bottom">0.46</td></tr><tr><td valign="bottom"> TotalSubscore</td><td valign="bottom">−0.56</td><td valign="bottom">−0.48</td><td valign="bottom">0.22</td></tr><tr><td valign="bottom"> StepDistRH</td><td valign="bottom">0.89</td><td valign="bottom">−0.07</td><td valign="bottom">0.23</td></tr><tr><td valign="bottom"> StepDistRF</td><td valign="bottom">−0.65</td><td valign="bottom">0.28</td><td valign="bottom">−0.19</td></tr><tr><td valign="bottom"> StepDistLH</td><td valign="bottom">−0.28</td><td valign="bottom">−0.66</td><td valign="bottom">0.23</td></tr><tr><td valign="bottom"> StepDistLF</td><td valign="bottom">0.54</td><td valign="bottom">0.27</td><td valign="bottom">0.50</td></tr><tr><td valign="bottom"> RHSL</td><td valign="bottom">−0.76</td><td valign="bottom">0.34</td><td valign="bottom">0.12</td></tr><tr><td valign="bottom"> RHPA</td><td valign="bottom">−0.85</td><td valign="bottom">−0.30</td><td valign="bottom">0.26</td></tr><tr><td valign="bottom"> RFSL</td><td valign="bottom">0.74</td><td valign="bottom">0.40</td><td valign="bottom">0.32</td></tr><tr><td valign="bottom"> RFPA</td><td valign="bottom">−0.25</td><td valign="bottom">0.21</td><td valign="bottom">0.41</td></tr><tr><td valign="bottom"> PawPL</td><td valign="bottom">−0.76</td><td valign="bottom">−0.30</td><td valign="bottom">0.35</td></tr><tr><td valign="bottom"> LHSL</td><td valign="bottom">0.62</td><td valign="bottom">0.42</td><td valign="bottom">0.48</td></tr><tr><td valign="bottom"> LHPA</td><td valign="bottom">−0.24</td><td valign="bottom">0.12</td><td valign="bottom">0.62</td></tr><tr><td valign="bottom"> LFSL</td><td valign="bottom">0.38</td><td valign="bottom">−0.62</td><td valign="bottom">−0.05</td></tr><tr><td valign="bottom"> LFPA</td><td valign="bottom">−0.54</td><td valign="bottom">0.63</td><td valign="bottom">−0.12</td></tr><tr><td valign="bottom"> Groom</td><td valign="bottom">0.49</td><td valign="bottom">−0.65</td><td valign="bottom">0.03</td></tr><tr><td valign="bottom"> ForelimbOpenField</td><td valign="bottom">0.20</td><td valign="bottom">−0.59</td><td valign="bottom">−0.14</td></tr><tr><td valign="bottom"> BBB_FergTrans</td><td valign="bottom">0.51</td><td valign="bottom">−0.32</td><td valign="bottom">−0.03</td></tr></tbody></table></table-wrap><sec id="s4-3-1"><title>syndromic_plot ()</title><p>The list of arguments for the <italic>syndromic_plot()</italic> function are presented in the package manual. The <italic>syndromic_plot()</italic> function will internally call <italic>extract_syndromic_plot()</italic> function (see utility functions) and return a list of <italic>ggplot2</italic> objects containing the syndromic plot for the first <italic>ndim</italic> PCs. For example, if <italic>ndim</italic> = 5, a syndromic plot for PCs 1 to 5 will be generated. Another important argument is the <italic>cut_off</italic>, which determines the threshold of absolute standardized loadings to consider for plotting. This argument is chosen by the user and is required (with no default). Another required argument is <italic>VAF</italic> in case the <italic>syndromic_plot()</italic> function is called using a <italic>data.frame</italic> input. If the output of the <italic>prcomp()</italic> or <italic>princals()</italic> functions is used, the <italic>syndromic_plot()</italic> function extracts <italic>VAF</italic> internally and the user-defined <italic>VAF</italic> will be ignored. When required, <italic>VAF</italic> is a character vector of the form ‘XX%”,”XX%”, etc., where XX is the VAF for each PC to plot, starting with the first PC, followed by the second, etc. (e.g. <italic>c(‘60.1%”,”25.3%”)</italic> for PC1 and PC2, respectively). An issue we found during the implementation is that the arrow visualization does not display correctly in the R graphical device on Windows machines. Rendering the plot into *.pdf format, for instance using the <italic>ggsave()</italic> function from the <italic>ggplot2</italic> package, solves the problem.</p></sec><sec id="s4-3-2"><title><italic>heatmap_loadings()</italic>, <italic>barmap_loadings()</italic> and <italic>barmap_commun()</italic></title><p>Most of the functionalities described for the <italic>syndromic_plot()</italic> function also apply for the <italic>heatmap_loading()</italic>, the <italic>barmap_loading()</italic>, and the barmap_commun() functions. A noticeable difference in <italic>barmap_loading()</italic> is that the function will plot the PCs specified in <italic>ndim</italic> instead of the first <italic>ndim</italic> components. For example, if <italic>ndim</italic> = <italic>c(3,4,5)</italic>, components 3, 4, and 5 will be plotted. This allows for more flexibility on which components to plot, such as isolating a single component (e.g. <italic>ndim</italic> = 3 will only plot component 3).</p></sec><sec id="s4-3-3"><title>VAF_plot()</title><p>This function can be used to plot a VAF plot from a <italic>prcomp()</italic> or <italic>princals()</italic> output. There are two <italic>style</italic> options, ‘line’ or ‘reduced’.</p></sec></sec><sec id="s4-4"><title>Resampling functions</title><p>There are two major functions using resampling methods, the <italic>permut_pc_test()</italic> function that implements nonparametric permutation test for either PC VAF for aiding in component selection or PC loadings and communalities for aiding in component interpretation, and the <italic>pc_stability()</italic> function that implements bootstrapping of PC loadings for stability analysis. These functions take as input the output of the <italic>prcomp()</italic> or the <italic>princals()</italic> functions in R as well as the original dataset used on these functions as inputs. The specific call of <italic>prcomp()</italic> or the <italic>princals()</italic> used to obtain the original PCA solution is passed down to the resampling functions in the <italic>syndRomics</italic> package, ensuring that the same arguments are used for resampling (with the exception of the <italic>data</italic> argument on the original <italic>prcomp()</italic> or the <italic>princals()</italic> call, that will be internally changed for each resampling iteration).</p><sec id="s4-4-1"><title>permut_pc_test()</title><p>In the <italic>syndRomics</italic> package, the null distribution for the permutation test is generated by permuting the values of each variable independently and concomitantly several times (<italic>permD</italic>) or permuting one variable at the time (<italic>permV</italic>) and re-running the PCA on each permuted sample (<xref ref-type="fig" rid="fig2">Figure 2</xref>; <xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>; <xref ref-type="bibr" rid="bib27">Glorfeld, 1995</xref>; <xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). When <italic>permV</italic> method is selected to measure the impact of permuting on loadings, a step of Procrustes rotation of each loading matrix toward the original loading matrix is added to resolve sign reflection, rotation indeterminacy and component translocation (<xref ref-type="fig" rid="fig2">Figure 2D</xref>, see pc_stability for detailed explanation). This step is not performed when the analysis is performed on the communalities since are invariant to such PCA resampling issues (<xref ref-type="bibr" rid="bib56">Linting et al., 2011</xref>). Confidence intervals of the permuted distribution (null distribution) are calculated using the (1-α)x100% (percentile) of the distribution (<xref ref-type="bibr" rid="bib6">Buja and Eyuboglu, 1992</xref>).</p><p>The function calls the <italic>permut_pca_D() or permut_pca_V()</italic> utility generic function internally to generate the permuted distribution of the selected metric (either “VAF”,“s.loadings” or “comuna”) using either the <italic>prcomp()</italic> function for linear PCA or the <italic>princals()</italic> function for nonlinear PCA implemented as S3 R method for the class “prcomp” or “princals”. If “VAF” is specified the <italic>permD</italic> permutation will be used, ignoring the input of the user on the <italic>perm.method</italic> argument, returning a matrix containing the VAF for the original PCs, as well as the average and the CI of the permuted VAF distribution. In case “s. loadings” or “communa” are specified, the specified permutation method will be considered (i.e. <italic>permD</italic> or <italic>permV</italic>) and the function will return a list of matrices, one for each selected PC, with the original loadings, and the average and CI of the permuted loadings distribution. In both cases, p values are calculated as described in the main text and returned. Adjusted p values using the specified method in the <italic>adjust.method</italic> argument are also returned.</p></sec><sec id="s4-4-2"><title>pc_stability()</title><p>Component stability can be studied at the whole component level, known as factor invariance, or at the level of the individual loadings. We have implemented both options in the package. By default, the <italic>pc_stability()</italic> function returns the average and the accelerated and bias-corrected 95% confidence intervals (CI) of the loadings of the bootstrap distribution (<xref ref-type="bibr" rid="bib19">Efron, 1987</xref>). Depending on the sample size and the number of chosen resamples, the bias-corrected CI will fail and the percentile (1-α)x100% CI will be returned (with corresponding notification). In addition, component similarity or factor matching metrics can be computed by setting the <italic>test_similarity</italic>=TRUE, which will call the <italic>component_similarity()</italic> function. For each of the specified similarity metrics, this function returns the average of the metric and its confidence interval (95% CI by default) by the percentiles of the bootstrap distribution. The confidence level and the CI method for the loadings can be changed by changing the <italic>conf</italic> and <italic>ci_type</italic> arguments. The function uses the <italic>boot()</italic> function for generating the bootstrapped samples and the <italic>boot.ci()</italic> function for extracting the confidence intervals of the loadings. Both <italic>boot()</italic> and <italic>boot.ci()</italic> are from the <italic>boot</italic> package in R. This allows the use of different bootstrapping strategies such as simple or ordinary bootstrapping (by default) or balanced bootstrapping. The reader is referred to the <italic>boot</italic> package documentation for more details on the different <italic>sim</italic> methods.</p><p>A major problem of performing resampling procedures in PCA is what is known as indeterminacies that can invalidate comparing between bootstrapped samples (<xref ref-type="bibr" rid="bib3">Babamoradi et al., 2013</xref>; <xref ref-type="bibr" rid="bib15">Chan et al., 1999</xref>; <xref ref-type="bibr" rid="bib53">Linting, 2007</xref>; <xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib89">Zabala and Pascual, 2016</xref>). Sign reflection refers to the change of sign on the component loadings in a PC given slight variation of the data. In addition, slight data variation can also cause component/factor translocation, the change in the position of a component in the PCA solution (e.g. PC1 shifts to the position of PC2), especially when two components have similar VAF. Another problem on performing PCAs with variations in the data is the possibility of rotation indeterminacy when the PCA solution of a resampled data presents with a different rotation of the original PCA solution. These issues generate artificially biased bootstrapped distributions, potentially invalidating the procedure (<xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>). We have implemented a step of procrustes rotation between the original loadings (target) and the bootstrapped sample, as has been previously demonstrated to be a reasonable method to deal with such issues (<xref ref-type="bibr" rid="bib77">Timmerman et al., 2007</xref>; <xref ref-type="bibr" rid="bib91">Zientek and Thompson, 2007</xref>). The Procrustes rotation is obtained by the <italic>procrustes()</italic> function from the <italic>pracma</italic> package. The algorithm for bootstrapping the PCA solutions is represented in <xref ref-type="fig" rid="fig4">Figure 4</xref> and implemented in the utility function <italic>boot_pca_sample()</italic>. The number of bootstrap samples is set to 1000 by default. The user must be careful on setting the number too low, reducing the performance of the approximation (<xref ref-type="bibr" rid="bib19">Efron, 1987</xref>). However, setting the number of bootstrap samples too high might unnecessarily increase computing time with little gain (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>).</p></sec></sec><sec id="s4-5"><title>Indexes of component similarity</title><p>We have included several component similarity indexes for determining component/factor invariance in <italic>syndRomics</italic>. The function <italic>component_similiarity ()</italic> returns the specified similarity metrics as well as their summary statistics (average and standard deviation, if applicable) from a list of loading matrices (<italic>load.list</italic>). The argument <italic>s_cut_off</italic> is used in the calculation of the Cattell’s <italic>s</italic> index (see below) and <italic>ndim</italic> is used to limit the number of components from which to compute the indexes from. Each index has been programmed in a separate utility function for convenience. Although they are not meant to be manually called, users can call them to calculate any of these metrics for a given set of two component loadings. The <italic>similarity_metric</italic> argument takes a single character or a vector of characters to specify which metrics to compute. These can be: ‘cc_index’, ‘r_correlation’, ‘rmse’ and/or ‘s_index’. The user can also specify ‘all’ to get all metrics. Their definitions are documented below.</p><sec id="s4-5-1"><title>Congruence coefficient (CC, ‘cc_index’)</title><p>First suggested by <xref ref-type="bibr" rid="bib7">Burt, 1948</xref>, it was popularized by <xref ref-type="bibr" rid="bib78">Tucker, 1951</xref> and therefore is also known as Tucker’s congruence coefficient. It is calculated as (2):<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ϕ</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mrow><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:msqrt><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mi>y</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></disp-formula>where <inline-formula><mml:math id="inf10"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf11"><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the loadings of the variable <inline-formula><mml:math id="inf12"> <mml:mi/><mml:mi>i</mml:mi></mml:math></inline-formula> on the component or factor <inline-formula><mml:math id="inf13"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf14"><mml:mi>y</mml:mi></mml:math></inline-formula> respectively. <inline-formula><mml:math id="inf15"><mml:mi>∅</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mfenced></mml:math></inline-formula> is equivalent to the cosine of the angle between two vectors and is also referred to as the cosine similarity metric. CC is a measure of proportional similarity between two components, and technically the index has a range from -1 (perfect negative congruence) to 1 (perfect positive congruence). In practice, because the all the loadings of a PC can be multiplied by -1 without changing the meaning of the PC, the absolute value of CC is considered, which correspondingly ranges from 0 to 1. The closer to 1, the more similar the two components are. Chan et al. discussed the 0.9 rule of thumb as an indicator of good matching between PCs (<xref ref-type="bibr" rid="bib15">Chan et al., 1999</xref>). The application of CC as a similarity metric for factor invariance has been extensively studied (<xref ref-type="bibr" rid="bib15">Chan et al., 1999</xref>; <xref ref-type="bibr" rid="bib57">Lorenzo-Seva and ten Berge, 2006</xref>).</p></sec><sec id="s4-5-2"><title>Pearson’s correlation coefficient (<italic>r,</italic> ‘r_correlation’)</title><p>The calculation of <italic>r</italic> between two vectors of component loadings has also been used as a pattern matching metric (<xref ref-type="bibr" rid="bib29">Guadagnoli and Velicer, 1991</xref>). It is computed as (3):<disp-formula id="equ3"><label>(3)</label><mml:math id="m3"><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>-</mml:mo></mml:mover></mml:mrow></mml:mfenced><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo>-</mml:mo></mml:mover></mml:mrow></mml:mfenced></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>-</mml:mo></mml:mover></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:msqrt><mml:msqrt><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo>-</mml:mo></mml:mover></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>In the <italic>syndRomics</italic> package, the Pearson’s correlation coefficient is calculated using the <italic>cor()</italic> function of the <italic>stats</italic> package.</p></sec><sec id="s4-5-3"><title>Root mean square error (RMSE, ‘rmse’)</title><p>RMSE has also been used as a metric for factor matching (<xref ref-type="bibr" rid="bib29">Guadagnoli and Velicer, 1991</xref>). It is calculated as the square root of the average squared difference of the loadings of the variables as (4):<disp-formula id="equ4"><label>(4)</label><mml:math id="m4"><mml:msub><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mfrac><mml:mrow><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:msqrt></mml:math></disp-formula></p><p><table-wrap id="inlinetable1" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th valign="top"/><th colspan="4">Component 2</th></tr><tr><th valign="bottom">Component 1</th><th>PS</th><th>H</th><th>NS</th></tr></thead><tbody><tr><td> PS</td><td><inline-formula><mml:math id="inf16"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf17"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf18"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td></tr><tr><td> H</td><td><inline-formula><mml:math id="inf19"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>21</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf20"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>22</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf21"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>23</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td></tr><tr><td> NS</td><td><inline-formula><mml:math id="inf22"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>31</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf23"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>32</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td><td><inline-formula><mml:math id="inf24"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>33</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></td></tr></tbody></table></table-wrap></p><p>where <inline-formula><mml:math id="inf25"><mml:mi>n</mml:mi></mml:math></inline-formula> is the number of variables in both components <inline-formula><mml:math id="inf26"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf27"><mml:mi>y</mml:mi></mml:math></inline-formula>. A RMSE of 0 determines a perfect matching, and therefore the smaller the RMSE is, the more equivalent the two components <inline-formula><mml:math id="inf28"><mml:mi>x</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf29"><mml:mi>y</mml:mi></mml:math></inline-formula> are.</p></sec><sec id="s4-5-4"><title>Cattell’s <italic>s</italic> index (‘s_index’)</title><p>The <italic>s</italic> index was first suggested by <xref ref-type="bibr" rid="bib14">Cattell and Baggaley, 1960</xref>; <xref ref-type="bibr" rid="bib13">Cattell et al., 1969</xref>. It is based on the <italic>factor mandate matrix</italic> (<xref ref-type="bibr" rid="bib14">Cattell and Baggaley, 1960</xref>) where loadings are either one if a component is considered to act on a variable, called a <italic>salient variable</italic>, or 0 if not (forming the <italic>hyperplane</italic> space). Cattell’s suggested an arbitrary ± 0.1 cutoff where variables with loadings outside the cutoff range are removed from the <italic>hyperplane</italic> and considered to be <italic>salient variables.</italic> In practice, one might want to alter the threshold depending on the experimental conditions. Any loading inside the cutoff range is then interpreted as having been produced by chance. The s index is calculated from the cross-classification of the common variables of two components/factors:</p><p>where PS = positive salient variable; H = hyperplane variable; NS = negative salient variable; <inline-formula><mml:math id="inf30"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the joint frequency. Positive and negative salient variables are variables outside the cutoff range with positive or negative loadings respectively.</p><p>Pattern matching is determined by comparing the cell frequencies in the cross-classification table. Here we implement the simplified form of calculating <italic>s</italic> (5):<disp-formula id="equ5"><label>(5)</label><mml:math id="m5"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>33</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>31</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>33</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>31</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mfenced separators="|"><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>21</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>23</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mn>32</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac></mml:math></disp-formula></p><p>The reader is referred to <xref ref-type="bibr" rid="bib14">Cattell and Baggaley, 1960</xref>; <xref ref-type="bibr" rid="bib13">Cattell et al., 1969</xref>; <xref ref-type="bibr" rid="bib29">Guadagnoli and Velicer, 1991</xref> for details on reasoning and calculations. <italic>s</italic> ranges from 1 (perfect similarity) to −1 (perfect dissimilarity) centered at 0 (pattern due to chance). Similar to CC, the absolute value of <italic>s</italic> is considered.</p></sec></sec><sec id="s4-6"><title>Internal functions</title><p>There are internal functions used by the package that the user might never have to call directly, although they are accessible in case the user needs them. Here, we provided a general description of those, leaving the details to the package documentation. All the internal functions to extract similarity metrics are: <italic>extract_cc(), extract_s()</italic> and <italic>extract_rmse().</italic> They all take two numeric vectors and return the corresponding similarity metric between them.</p><sec id="s4-6-1"><title>new_syndromics()</title><p>Helper function to construct the ‘syndromics’ class object that will be use in the S3 generic and method functions. It returns an object of class ‘syndromics’ of the type list.</p></sec><sec id="s4-6-2"><title>stand_loadings()</title><p>This function extracts the standardized loadings from the output of the <italic>prcomp()</italic> or the <italic>princals()</italic> functions. In the case of the <italic>prcomp()</italic> solution, the standardized loadings are calculated as: <inline-formula><mml:math id="inf31"><mml:mi>s</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>s</mml:mi><mml:mo>×</mml:mo><mml:msqrt><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:msqrt></mml:math></inline-formula> if the PCA was performed on the standardized (scaled to unit variance) data or <inline-formula><mml:math id="inf32"><mml:mi>s</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mo>×</mml:mo><mml:msqrt><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:msqrt><mml:mo>)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> where <inline-formula><mml:math id="inf33"><mml:mi>S</mml:mi></mml:math></inline-formula> is the vector of the variables standard deviation. In the case of <italic>princals(),</italic> standardized loadings are returned directly in its output and therefore <italic>stand_loadings()</italic> returns those. The function returns a data frame with the standardized loadings in the form of variables as rows and PCs as columns.</p></sec><sec id="s4-6-3"><title>extract_loadings()</title><p>This is a wrapper function for <italic>stand_loadings()</italic> with added functionalities such as error breakers that is used by most functions in the package.</p></sec><sec id="s4-6-4"><title>extract_syndromic_plot()</title><p>This function is internally called by the <italic>syndromic_plot()</italic> function and returns a <italic>ggplot2</italic> object with the syndromic plot for the specified PC. The only argument that is not present in the <italic>syndromic_plot()</italic> function is the <italic>pc</italic> argument that specifies which PC to plot. Users should always use <italic>syndromic_plot()</italic> function instead of <italic>extract_syndromic_plot()</italic> since <italic>syndromic_plot()</italic> automatically incorporates other functionalities.</p></sec><sec id="s4-6-5"><title>component_similarity()</title><p>This function is called by the <italic>pc_stability()</italic> function to calculate the specified similarity metric (see above) between the given list of data frames of loadings. While pc_similarity() uses this function to calculate similarity between the original (parent) loadings and a B sample loadings, the passed list of loadings can be n &gt; 2. Then, the similarity metrics will be calculated between all combinations of n. It returns a list of objects containing a list of the comparisons, a data frame with the averaged metric and the bounds of confidence interval for each specified metric and PC.</p></sec><sec id="s4-6-6"><title>boot_pca_sample()</title><p>This generic function is passed to the <italic>statistic</italic> argument of the <italic>boot()</italic> function internally called by the <italic>pc_stability()</italic> function. It implements the bootstrapping algorithm described above (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). Then the <italic>boot()</italic> function will call <italic>boot_pca_sample()</italic> B times from the specified data and the pca output of the <italic>prcomp()</italic> (through the method <italic>boot_pca_sample.prcomp()</italic>) or <italic>princals()</italic> (through the method <italic>boot_pca_sample.princals()</italic>) function, returning a list of B data frames of loadings. The bootstrapping method can be specified using the <italic>sim</italic> argument.</p></sec><sec id="s4-6-7"><title>permut_pca_D() or permut_pca_V()</title><p>This is a generic function internally called by <italic>permut_pca_test()</italic> to produce <italic>P</italic> permutations of the given output of the <italic>prcomp()</italic> or the <italic>princals()</italic> functions using <italic>permD</italic> or <italic>permV</italic> method. Four S3 R function methods are implemented: <italic>permut_pca_D.prcomp()</italic>, <italic>permut_pca_D.princals()</italic>, <italic>permut_pca_V.prcomp()</italic>, <italic>permut_pca_V.princals()</italic>. It returns a list of the results of permuting the data, conducting a PCA and extracting either the VAF or the standardized loadings for each P as in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p></sec><sec id="s4-6-8"><title>Plot.syndromics()</title><p>This function implement the S3 method for plotting ‘syndromics’ class objects generated by <italic>pc_stability()</italic> and <italic>permut_pc_test()</italic> functions using the R generic <italic>plot()</italic>. It returns specific plots calling the visualization functions implemented in the package.</p></sec></sec><sec id="s4-7"><title>Nonlinear PCA</title><p>Nonlinear PCA by optimal scaling and alternating least square was obtained using the <italic>princals()</italic> function from the {Gifi} package in R. We specified to analyze all variables with nominal restriction scaling, allowing for non-monotonic transformations, and set a restriction of 3 degrees in polynomial transformations for nonlinearity. The corresponding instruction was: <italic>princals(nlpca_data, ndim = ncol(nlpca_data), ordinal = FALSE, degrees = 3, knots = knotsGifi(nlpca_data, type=‘E’)),</italic> where <italic>nlpca_data</italic> is the imputed dataset for case study 2 (see supplementary code for more details).</p></sec><sec id="s4-8"><title>Missing data analysis</title><p>Details on the code are available as supplementary material. Data wrangling for the two case studies was performed using R packages included in the <italic>Tidyverse</italic> package. Missing pattern visualization were obtained using the <italic>naniar</italic> (<xref ref-type="bibr" rid="bib76">Tierney et al., 2020</xref>) R packages. Test for MCAR was performed using the <italic>TestMCARNormality()</italic> function from the <italic>MissMech</italic> package (<xref ref-type="bibr" rid="bib40">Jamshidian et al., 2014</xref>). Multiple imputation was performed using predicting mean matching method available in the <italic>mice</italic> (<xref ref-type="bibr" rid="bib9">Buuren and Groothuis-Oudshoorn, 2011</xref>) R package, setting the number of imputations to <italic>m</italic> = 50. A list of 50 complete datasets were then obtained and processed by PCA as specified in the main text. For each <italic>m</italic> dataset, the loadings where extracted and rotated using Procrustes rotation (<italic>pracma</italic> package) toward the average of the imputed datasets. The distributions of loadings and component similarities for the first three PCs where calculated using the <italic>syndRomics</italic> package as described above.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>This work is supported by NIH grants NS106899 (ARF), NS088475 (ARF); VA Grants 1I01R × 002245 (ARF) and I01R × 002787 (ARF)</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Software, Formal analysis, Supervision, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Software, Visualization, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Visualization, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Software, Validation, Writing - review and editing</p></fn><fn fn-type="con" id="con5"><p>Conceptualization, Software, Writing - review and editing</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Supervision, Funding acquisition, Visualization, Methodology, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="scode1"><label>Source code 1.</label><caption><title>The R script reproducing the analysis of this manuscript in a Rmarkdown file.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-61812-code1-v2.zip"/></supplementary-material><supplementary-material id="scode2"><label>Source code 2.</label><caption><title>A rendered script with code and outputs of running the code in a html file.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-61812-code2-v2.zip"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-61812-transrepform-v2.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>This work used already available data at the Open Data Commons for Spinal Cord Injury (<ext-link ext-link-type="uri" xlink:href="http://odc-sci.org/">http://odc-sci.org/</ext-link>) and Plos One (<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0169490">https://doi.org/10.1371/journal.pone.0169490</ext-link>).</p><p>The following previously published datasets were used:</p><p><element-citation id="dataset1" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Ferguson</surname><given-names>AR</given-names></name><name><surname>Irvine</surname><given-names>K-A</given-names></name><name><surname>Gensel</surname><given-names>JC</given-names></name><name><surname>Nielson</surname><given-names>JL</given-names></name><name><surname>Lin</surname><given-names>A</given-names></name><name><surname>Ly</surname><given-names>J</given-names></name><name><surname>Segal</surname><given-names>MR</given-names></name><name><surname>Ratan</surname><given-names>RR</given-names></name><name><surname>Bresnahan</surname><given-names>JC</given-names></name><name><surname>Beattie</surname><given-names>MS</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>Cervical (C5), unilateral spinal cord injury with diverse 740 injury modalities, multiple behavioral outcomes, and histopathology</data-title><source>Open Data Common for Spinal Cord Injury</source><pub-id assigning-authority="other" pub-id-type="accession" xlink:href="https://scicrunch.org/odc-sci/about/odc-sci_26">(ODC-SCI:26)</pub-id></element-citation></p><p><element-citation id="dataset5" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Nielson</surname><given-names>JL</given-names></name><name><surname>Cooper</surname><given-names>SR</given-names></name><name><surname>Yue</surname><given-names>JK</given-names></name><name><surname>Sorani</surname><given-names>MD</given-names></name><name><surname>Inoue</surname><given-names>T</given-names></name><name><surname>Yuh</surname><given-names>EL</given-names></name><name><surname>Mukherjee</surname><given-names>P</given-names></name><name><surname>Petrossian</surname><given-names>TC</given-names></name><name><surname>Paquette</surname><given-names>J</given-names></name><name><surname>Lum</surname><given-names>PY</given-names></name><name><surname>Carlsson</surname><given-names>GE</given-names></name><name><surname>Vassar</surname><given-names>MJ</given-names></name><name><surname>Lingsma</surname><given-names>HF</given-names></name><name><surname>Gordon</surname><given-names>WA</given-names></name><name><surname>Valadka</surname><given-names>AB</given-names></name><name><surname>Okonkwo</surname><given-names>DO</given-names></name><name><surname>Manley</surname><given-names>GT</given-names></name><name><surname>Ferguson</surname><given-names>AR</given-names></name><name><surname>TRACK-TBI</surname><given-names>Investigators</given-names></name></person-group><year iso-8601-date="2017">2017</year><data-title>Uncovering precision phenotype-biomarker associations in traumatic brain injury using topological data analysis</data-title><source>journal</source><pub-id assigning-authority="other" pub-id-type="doi">10.1371/journal.pone.0169490</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abdi</surname> <given-names>H</given-names></name><name><surname>Williams</surname> <given-names>LJ</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Principal component analysis</article-title><source>Wiley Interdisciplinary Reviews: Computational Statistics</source><volume>2</volume><fpage>433</fpage><lpage>459</lpage><pub-id pub-id-type="doi">10.1002/wics.101</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Altman</surname> <given-names>N</given-names></name><name><surname>Krzywinski</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The curse(s) of dimensionality</article-title><source>Nature Methods</source><volume>15</volume><fpage>399</fpage><lpage>400</lpage><pub-id pub-id-type="doi">10.1038/s41592-018-0019-x</pub-id><pub-id pub-id-type="pmid">29855577</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Babamoradi</surname> <given-names>H</given-names></name><name><surname>van den Berg</surname> <given-names>F</given-names></name><name><surname>Rinnan</surname> <given-names>Å</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Bootstrap based confidence limits in principal component analysis — A case study</article-title><source>Chemometrics and Intelligent Laboratory Systems</source><volume>120</volume><fpage>97</fpage><lpage>105</lpage><pub-id pub-id-type="doi">10.1016/j.chemolab.2012.10.007</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benjamini</surname> <given-names>Y</given-names></name><name><surname>Hochberg</surname> <given-names>Y</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title><source>Journal of the Royal Statistical Society</source><volume>57</volume><fpage>289</fpage><lpage>300</lpage><pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Borchers</surname> <given-names>HW</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>Pracma: Practical Numerical Math Functions</source><version designator="2.2.9">2.2.9</version><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=pracma">https://CRAN.R-project.org/package=pracma</ext-link></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buja</surname> <given-names>A</given-names></name><name><surname>Eyuboglu</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Remarks on parallel analysis</article-title><source>Multivariate Behavioral Research</source><volume>27</volume><fpage>509</fpage><lpage>540</lpage><pub-id pub-id-type="doi">10.1207/s15327906mbr2704_2</pub-id><pub-id pub-id-type="pmid">26811132</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burt</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="1948">1948</year><article-title>The factorial study of temperamental traits</article-title><source>British Journal of Statistical Psychology</source><volume>1</volume><fpage>178</fpage><lpage>203</lpage><pub-id pub-id-type="doi">10.1111/j.2044-8317.1948.tb00236.x</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Buuren</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2018">2018</year><source>Flexible Imputation of Missing Data</source><publisher-loc>Ohio, United States</publisher-loc><publisher-name>CRC Press</publisher-name></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buuren</surname> <given-names>S</given-names></name><name><surname>Groothuis-Oudshoorn</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Mice : Multivariate Imputation by Chained Equations in R</article-title><source>Journal of Statistical Software</source><volume>45</volume><fpage>1</fpage><lpage>67</lpage><pub-id pub-id-type="doi">10.18637/jss.v045.i03</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Callahan</surname> <given-names>A</given-names></name><name><surname>Anderson</surname> <given-names>KD</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Bixby</surname> <given-names>JL</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Fouad</surname> <given-names>K</given-names></name><name><surname>Jakeman</surname> <given-names>LB</given-names></name><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Popovich</surname> <given-names>PG</given-names></name><name><surname>Schwab</surname> <given-names>JM</given-names></name><name><surname>Lemmon</surname> <given-names>VP</given-names></name><collab>FAIR Share Workshop Participants</collab></person-group><year iso-8601-date="2017">2017</year><article-title>Developing a data sharing community for spinal cord injury research</article-title><source>Experimental Neurology</source><volume>295</volume><fpage>135</fpage><lpage>143</lpage><pub-id pub-id-type="doi">10.1016/j.expneurol.2017.05.012</pub-id><pub-id pub-id-type="pmid">28576567</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Canty</surname> <given-names>A</given-names></name><name><surname>Ripley</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>Boot: Bootstrap R (S-Plus) Functions</source><version designator="1.3-23">1.3-23</version><ext-link ext-link-type="uri" xlink:href="https://astrostatistics.psu.edu/su07/R/html/boot/html/00Index.html">https://astrostatistics.psu.edu/su07/R/html/boot/html/00Index.html</ext-link></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cattell</surname> <given-names>RB</given-names></name></person-group><year iso-8601-date="1966">1966</year><article-title>The scree test for the number of factors</article-title><source>Multivariate Behavioral Research</source><volume>1</volume><fpage>245</fpage><lpage>276</lpage><pub-id pub-id-type="doi">10.1207/s15327906mbr0102_10</pub-id><pub-id pub-id-type="pmid">26828106</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cattell</surname> <given-names>RB</given-names></name><name><surname>Balcar</surname> <given-names>KR</given-names></name><name><surname>Horn</surname> <given-names>JL</given-names></name><name><surname>Nesselroade</surname> <given-names>JR</given-names></name></person-group><year iso-8601-date="1969">1969</year><article-title>Factor matching procedures: an improvement of the s index; with tables</article-title><source>Educational and Psychological Measurement</source><volume>29</volume><fpage>781</fpage><lpage>792</lpage><pub-id pub-id-type="doi">10.1177/001316446902900405</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cattell</surname> <given-names>RB</given-names></name><name><surname>Baggaley</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="1960">1960</year><article-title>The salient variable similarity index for factor matching</article-title><source>British Journal of Statistical Psychology</source><volume>13</volume><fpage>33</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1111/j.2044-8317.1960.tb00037.x</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>W</given-names></name><name><surname>Ho</surname> <given-names>RM</given-names></name><name><surname>Leung</surname> <given-names>K</given-names></name><name><surname>Chan</surname> <given-names>DK-S</given-names></name><name><surname>Yung</surname> <given-names>Y-F</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>An alternative method for evaluating congruence coefficients with procrustes rotation: a bootstrap procedure</article-title><source>Psychological Methods</source><volume>4</volume><fpage>378</fpage><lpage>402</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.4.4.378</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Davison</surname> <given-names>AC</given-names></name><name><surname>Hinkley</surname> <given-names>DV</given-names></name></person-group><year iso-8601-date="1997">1997</year><source>Bootstrap Methods and Their Application</source><publisher-loc>Cambridge, United KIngdom</publisher-loc><publisher-name>Cambridge University Press</publisher-name></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dray</surname> <given-names>S</given-names></name><name><surname>Josse</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Principal component analysis with missing values: a comparative survey of methods</article-title><source>Plant Ecology</source><volume>216</volume><fpage>657</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1007/s11258-014-0406-z</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dunn</surname> <given-names>OJ</given-names></name></person-group><year iso-8601-date="1961">1961</year><article-title>Multiple comparisons among means</article-title><source>Journal of the American Statistical Association</source><volume>56</volume><fpage>52</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1080/01621459.1961.10482090</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Efron</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="1987">1987</year><article-title>Better bootstrap confidence intervals</article-title><source>Journal of the American Statistical Association</source><volume>82</volume><fpage>171</fpage><lpage>185</lpage><pub-id pub-id-type="doi">10.1080/01621459.1987.10478410</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="software"><person-group person-group-type="author"><collab>Elio Campitelli</collab></person-group><year iso-8601-date="2020">2020</year><source>Ggnewscale: Multiple Fill and Colour Scales in “Ggplot2”</source><version designator="0.4.1">0.4.1</version></element-citation></ref><ref id="bib21"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Everitt</surname> <given-names>B</given-names></name><name><surname>Hothorn</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2011">2011</year><source>An Introduction to Applied Multivariate Analysis with R</source><publisher-loc>Berlin, Germany</publisher-loc><publisher-name>Springer-Verlag</publisher-name></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Failla</surname> <given-names>MD</given-names></name><name><surname>Myrga</surname> <given-names>JM</given-names></name><name><surname>Ricker</surname> <given-names>JH</given-names></name><name><surname>Dixon</surname> <given-names>CE</given-names></name><name><surname>Conley</surname> <given-names>YP</given-names></name><name><surname>Wagner</surname> <given-names>AK</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Posttraumatic brain injury cognitive performance is moderated by variation within ANKK1 and DRD2 genes</article-title><source>Journal of Head Trauma Rehabilitation</source><volume>30</volume><fpage>E54</fpage><lpage>E66</lpage><pub-id pub-id-type="doi">10.1097/HTR.0000000000000118</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Stück</surname> <given-names>ED</given-names></name><name><surname>Nielson</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Syndromics: a bioinformatics approach for neurotrauma research</article-title><source>Translational Stroke Research</source><volume>2</volume><fpage>438</fpage><lpage>454</lpage><pub-id pub-id-type="doi">10.1007/s12975-011-0121-1</pub-id><pub-id pub-id-type="pmid">22207883</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Irvine</surname> <given-names>KA</given-names></name><name><surname>Gensel</surname> <given-names>JC</given-names></name><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Lin</surname> <given-names>A</given-names></name><name><surname>Ly</surname> <given-names>J</given-names></name><name><surname>Segal</surname> <given-names>MR</given-names></name><name><surname>Ratan</surname> <given-names>RR</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Derivation of multivariate syndromic outcome metrics for consistent testing across multiple models of cervical spinal cord injury in rats</article-title><source>PLOS ONE</source><volume>8</volume><elocation-id>e59712</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0059712</pub-id><pub-id pub-id-type="pmid">23544088</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="data"><person-group person-group-type="author"><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Irvine</surname> <given-names>K-A</given-names></name><name><surname>Gensel</surname> <given-names>JC</given-names></name><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Lin</surname> <given-names>A</given-names></name><name><surname>Ly</surname> <given-names>J</given-names></name><name><surname>Segal</surname> <given-names>MR</given-names></name><name><surname>Ratan</surname> <given-names>RR</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>Cervical (C5), unilateral spinal cord injury with diverse injury modalities, multiple behavioral outcomes, and histopathology</data-title><source>Open Data Common for Spinal Cord Injury</source><pub-id pub-id-type="doi">10.7295/W9T72FMZ</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fouad</surname> <given-names>K</given-names></name><name><surname>Bixby</surname> <given-names>JL</given-names></name><name><surname>Callahan</surname> <given-names>A</given-names></name><name><surname>Grethe</surname> <given-names>JS</given-names></name><name><surname>Jakeman</surname> <given-names>LB</given-names></name><name><surname>Lemmon</surname> <given-names>VP</given-names></name><name><surname>Magnuson</surname> <given-names>DS</given-names></name><name><surname>Martone</surname> <given-names>ME</given-names></name><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Schwab</surname> <given-names>J</given-names></name><name><surname>Taylor-Burds</surname> <given-names>C</given-names></name><name><surname>Tetzlaff</surname> <given-names>W</given-names></name><name><surname>Torres-Espín</surname> <given-names>A</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>FAIR SCI ahead: the evolution of the open data commons for Pre-Clinical spinal cord injury research</article-title><source>Journal of Neurotrauma</source><volume>37</volume><elocation-id>6674</elocation-id><pub-id pub-id-type="doi">10.1089/neu.2019.6674</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Glorfeld</surname> <given-names>LW</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>An improvement on Horn's Parallel Analysis Methodology for Selecting the Correct Number of Factors to Retain</article-title><source>Educational and Psychological Measurement</source><volume>55</volume><fpage>377</fpage><lpage>393</lpage><pub-id pub-id-type="doi">10.1177/0013164495055003002</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guadagnoli</surname> <given-names>E</given-names></name><name><surname>Velicer</surname> <given-names>WF</given-names></name></person-group><year iso-8601-date="1988">1988</year><article-title>Relation of sample size to the stability of component patterns</article-title><source>Psychological Bulletin</source><volume>103</volume><fpage>265</fpage><lpage>275</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.103.2.265</pub-id><pub-id pub-id-type="pmid">3363047</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guadagnoli</surname> <given-names>E</given-names></name><name><surname>Velicer</surname> <given-names>W</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>A comparison of pattern matching indices</article-title><source>Multivariate Behavioral Research</source><volume>26</volume><fpage>323</fpage><lpage>343</lpage><pub-id pub-id-type="doi">10.1207/s15327906mbr2602_7</pub-id><pub-id pub-id-type="pmid">26828257</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guttman</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="1954">1954</year><article-title>Some necessary conditions for common-factor analysis</article-title><source>Psychometrika</source><volume>19</volume><fpage>149</fpage><lpage>161</lpage><pub-id pub-id-type="doi">10.1007/BF02289162</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haefeli</surname> <given-names>J</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Bingham</surname> <given-names>D</given-names></name><name><surname>Orr</surname> <given-names>A</given-names></name><name><surname>Won</surname> <given-names>SJ</given-names></name><name><surname>Lam</surname> <given-names>TI</given-names></name><name><surname>Shi</surname> <given-names>J</given-names></name><name><surname>Hawley</surname> <given-names>S</given-names></name><name><surname>Liu</surname> <given-names>J</given-names></name><name><surname>Swanson</surname> <given-names>RA</given-names></name><name><surname>Massa</surname> <given-names>SM</given-names></name></person-group><year iso-8601-date="2017">2017a</year><article-title>A data-driven approach for evaluating multi-modal therapy in traumatic brain injury</article-title><source>Scientific Reports</source><volume>7</volume><elocation-id>42474</elocation-id><pub-id pub-id-type="doi">10.1038/srep42474</pub-id><pub-id pub-id-type="pmid">28205533</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haefeli</surname> <given-names>J</given-names></name><name><surname>Mabray</surname> <given-names>MC</given-names></name><name><surname>Whetstone</surname> <given-names>WD</given-names></name><name><surname>Dhall</surname> <given-names>SS</given-names></name><name><surname>Pan</surname> <given-names>JZ</given-names></name><name><surname>Upadhyayula</surname> <given-names>P</given-names></name><name><surname>Manley</surname> <given-names>GT</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Talbott</surname> <given-names>JF</given-names></name></person-group><year iso-8601-date="2017">2017b</year><article-title>Multivariate analysis of MRI biomarkers for predicting neurologic impairment in cervical spinal cord injury</article-title><source>American Journal of Neuroradiology</source><volume>38</volume><fpage>648</fpage><lpage>655</lpage><pub-id pub-id-type="doi">10.3174/ajnr.A5021</pub-id><pub-id pub-id-type="pmid">28007771</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Henry</surname> <given-names>L</given-names></name><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Rlang: Functions for Base Types and Core R and “Tidyverse” Features</source><version designator="0.4.4">0.4.4</version><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=rlang">https://CRAN.R-project.org/package=rlang</ext-link></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hollestein</surname> <given-names>LM</given-names></name><name><surname>Carpenter</surname> <given-names>JR</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Missing data in clinical research: an integrated approach</article-title><source>British Journal of Dermatology</source><volume>177</volume><fpage>1463</fpage><lpage>1465</lpage><pub-id pub-id-type="doi">10.1111/bjd.16010</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname> <given-names>S</given-names></name><name><surname>Mitchell</surname> <given-names>SK</given-names></name><name><surname>Harshman</surname> <given-names>RA</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Bootstrap scree tests: a monte carlo simulation and applications to published data</article-title><source>British Journal of Mathematical and Statistical Psychology</source><volume>59</volume><fpage>35</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1348/000711005X66770</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Horn</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="1965">1965</year><article-title>A rationale and test for the number of factors in factor analysis</article-title><source>Psychometrika</source><volume>30</volume><fpage>179</fpage><lpage>185</lpage><pub-id pub-id-type="doi">10.1007/BF02289447</pub-id><pub-id pub-id-type="pmid">14306381</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hotelling</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="1933">1933</year><article-title>Analysis of a complex of statistical variables into principal components</article-title><source>Journal of Educational Psychology</source><volume>24</volume><fpage>417</fpage><lpage>441</lpage><pub-id pub-id-type="doi">10.1037/h0071325</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huie</surname> <given-names>JR</given-names></name><name><surname>Almeida</surname> <given-names>CA</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Neurotrauma as a big-data problem</article-title><source>Current Opinion in Neurology</source><volume>31</volume><fpage>702</fpage><lpage>708</lpage><pub-id pub-id-type="doi">10.1097/WCO.0000000000000614</pub-id><pub-id pub-id-type="pmid">30379703</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jackson</surname> <given-names>JE</given-names></name><name><surname>Hearne</surname> <given-names>FT</given-names></name></person-group><year iso-8601-date="1973">1973</year><article-title>Relationships among coefficients of vectors used in principal components</article-title><source>Technometrics</source><volume>15</volume><fpage>601</fpage><lpage>610</lpage><pub-id pub-id-type="doi">10.1080/00401706.1973.10489087</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jamshidian</surname> <given-names>M</given-names></name><name><surname>Jalal</surname> <given-names>S</given-names></name><name><surname>Jansen</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>MissMech : AnR package for testing homoscedasticity, multivariate Normality, and missing completely at random (MCAR)</article-title><source>Journal of Statistical Software</source><volume>56</volume><fpage>1</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.18637/jss.v056.i06</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jamshidian</surname> <given-names>M</given-names></name><name><surname>Jalal</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Tests of Homoscedasticity, normality, and missing completely at random for incomplete multivariate data</article-title><source>Psychometrika</source><volume>75</volume><fpage>649</fpage><lpage>674</lpage><pub-id pub-id-type="doi">10.1007/s11336-010-9175-3</pub-id><pub-id pub-id-type="pmid">21720450</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>H</given-names></name><name><surname>Eskridge</surname> <given-names>KM</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Bias in principal components analysis due to correlated observations</article-title><conf-name>Conference on Applied Statistics in Agriculture</conf-name><pub-id pub-id-type="doi">10.4148/2475-7772.1247</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johnson</surname> <given-names>SR</given-names></name><name><surname>Reimer</surname> <given-names>SC</given-names></name><name><surname>Rothrock</surname> <given-names>TP</given-names></name></person-group><year iso-8601-date="1973">1973</year><article-title>Principal components and the problem of multicollinearity(*)</article-title><source>Metroeconomica</source><volume>25</volume><fpage>306</fpage><lpage>317</lpage><pub-id pub-id-type="doi">10.1111/j.1467-999X.1973.tb00218.x</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jolliffe</surname> <given-names>IT</given-names></name><name><surname>Cadima</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Principal component analysis: a review and recent developments</article-title><source>Philosophical Transactions of the Royal Society A: Mathematical, Physical and Engineering Sciences</source><volume>374</volume><elocation-id>20150202</elocation-id><pub-id pub-id-type="doi">10.1098/rsta.2015.0202</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaiser</surname> <given-names>HF</given-names></name></person-group><year iso-8601-date="1960">1960</year><article-title>The Application of Electronic Computers to Factor Analysis</article-title><source>Educational and Psychological Measurement</source><volume>20</volume><fpage>141</fpage><lpage>151</lpage><pub-id pub-id-type="doi">10.1177/001316446002000116</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaushal</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Missing data in clinical trials: pitfalls and remedies</article-title><source>International Journal of Applied &amp; Basic Medical Research</source><volume>4</volume><fpage>S6</fpage><lpage>S7</lpage><pub-id pub-id-type="pmid">25298948</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Konishi</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Principal component analysis for designed experiments</article-title><source>BMC Bioinformatics</source><volume>16 Suppl 18</volume><elocation-id>S7</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-16-S18-S7</pub-id><pub-id pub-id-type="pmid">26678818</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krzywinski</surname> <given-names>M</given-names></name><name><surname>Altman</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Comparing samples—part II</article-title><source>Nature Methods</source><volume>11</volume><fpage>355</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2900</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kutcher</surname> <given-names>ME</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Cohen</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>A principal component analysis of coagulation after trauma</article-title><source>Journal of Trauma and Acute Care Surgery</source><volume>74</volume><fpage>1223</fpage><lpage>1230</lpage><pub-id pub-id-type="doi">10.1097/TA.0b013e31828b7fa1</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landgrebe</surname> <given-names>J</given-names></name><name><surname>Wurst</surname> <given-names>W</given-names></name><name><surname>Welzl</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Permutation-validated principal components analysis of microarray data</article-title><source>Genome Biology</source><volume>3</volume><elocation-id>research0019.1</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2002-3-4-research0019</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lê</surname> <given-names>S</given-names></name><name><surname>Josse</surname> <given-names>J</given-names></name><name><surname>Husson</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>FactoMineR : An R package for multivariate analysis</article-title><source>Journal of Statistical Software</source><volume>25</volume><fpage>1</fpage><lpage>18</lpage><pub-id pub-id-type="doi">10.18637/jss.v025.i01</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lever</surname> <given-names>J</given-names></name><name><surname>Krzywinski</surname> <given-names>M</given-names></name><name><surname>Altman</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Principal component analysis</article-title><source>Nature Methods</source><volume>14</volume><fpage>641</fpage><lpage>642</lpage><pub-id pub-id-type="doi">10.1038/nmeth.4346</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Linting</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2007">2007</year><source>Doctoral Thesis: Nonparametric Inference in Nonlinear Principal Components Analysis: Exploration and Beyond</source><ext-link ext-link-type="uri" xlink:href="https://openaccess.leidenuniv.nl/handle/1887/12386">https://openaccess.leidenuniv.nl/handle/1887/12386</ext-link></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Linting</surname> <given-names>M</given-names></name><name><surname>Meulman</surname> <given-names>JJ</given-names></name><name><surname>Groenen</surname> <given-names>PJ</given-names></name><name><surname>van der Kooij</surname> <given-names>AJ</given-names></name></person-group><year iso-8601-date="2007">2007a</year><article-title>Stability of nonlinear principal components analysis: an empirical study using the balanced bootstrap</article-title><source>Psychological Methods</source><volume>12</volume><fpage>359</fpage><lpage>379</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.12.3.359</pub-id><pub-id pub-id-type="pmid">17784799</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Linting</surname> <given-names>M</given-names></name><name><surname>Meulman</surname> <given-names>JJ</given-names></name><name><surname>Groenen</surname> <given-names>PJ</given-names></name><name><surname>van der Koojj</surname> <given-names>AJ</given-names></name></person-group><year iso-8601-date="2007">2007b</year><article-title>Nonlinear principal components analysis: introduction and application</article-title><source>Psychological Methods</source><volume>12</volume><fpage>336</fpage><lpage>358</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.12.3.336</pub-id><pub-id pub-id-type="pmid">17784798</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Linting</surname> <given-names>M</given-names></name><name><surname>van Os</surname> <given-names>BJ</given-names></name><name><surname>Meulman</surname> <given-names>JJ</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Statistical significance of the contribution of variables to the PCA solution: an alternative permutation strategy</article-title><source>Psychometrika</source><volume>76</volume><fpage>440</fpage><lpage>460</lpage><pub-id pub-id-type="doi">10.1007/s11336-011-9216-6</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lorenzo-Seva</surname> <given-names>U</given-names></name><name><surname>ten Berge</surname> <given-names>JMF</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Tucker's Congruence Coefficient as a Meaningful Index of Factor Similarity</article-title><source>Methodology</source><volume>2</volume><fpage>57</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1027/1614-2241.2.2.57</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Mair</surname> <given-names>P</given-names></name><name><surname>Leeuw</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>Gifi: Multivariate Analysis with Optimal Scaling</source><version designator="0.3-9">0.3-9</version><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=Gifi">https://CRAN.R-project.org/package=Gifi</ext-link></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McAllister</surname> <given-names>TW</given-names></name><name><surname>Flashman</surname> <given-names>LA</given-names></name><name><surname>Harker Rhodes</surname> <given-names>C</given-names></name><name><surname>Tyler</surname> <given-names>AL</given-names></name><name><surname>Moore</surname> <given-names>JH</given-names></name><name><surname>Saykin</surname> <given-names>AJ</given-names></name><name><surname>McDonald</surname> <given-names>BC</given-names></name><name><surname>Tosteson</surname> <given-names>TD</given-names></name><name><surname>Tsongalis</surname> <given-names>GJ</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Single nucleotide polymorphisms in ANKK1 and the dopamine D2 receptor gene affect cognitive outcome shortly after traumatic brain injury: a replication and extension study</article-title><source>Brain Injury</source><volume>22</volume><fpage>705</fpage><lpage>714</lpage><pub-id pub-id-type="doi">10.1080/02699050802263019</pub-id><pub-id pub-id-type="pmid">18698520</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>LH</given-names></name><name><surname>Holmes</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Ten quick tips for effective dimensionality reduction</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1006907</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006907</pub-id><pub-id pub-id-type="pmid">31220072</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Guandique</surname> <given-names>CF</given-names></name><name><surname>Liu</surname> <given-names>AW</given-names></name><name><surname>Burke</surname> <given-names>DA</given-names></name><name><surname>Lash</surname> <given-names>AT</given-names></name><name><surname>Moseanko</surname> <given-names>R</given-names></name><name><surname>Hawbecker</surname> <given-names>S</given-names></name><name><surname>Strand</surname> <given-names>SC</given-names></name><name><surname>Zdunowski</surname> <given-names>S</given-names></name><name><surname>Irvine</surname> <given-names>KA</given-names></name><name><surname>Brock</surname> <given-names>JH</given-names></name><name><surname>Nout-Lomas</surname> <given-names>YS</given-names></name><name><surname>Gensel</surname> <given-names>JC</given-names></name><name><surname>Anderson</surname> <given-names>KD</given-names></name><name><surname>Segal</surname> <given-names>MR</given-names></name><name><surname>Rosenzweig</surname> <given-names>ES</given-names></name><name><surname>Magnuson</surname> <given-names>DS</given-names></name><name><surname>Whittemore</surname> <given-names>SR</given-names></name><name><surname>McTigue</surname> <given-names>DM</given-names></name><name><surname>Popovich</surname> <given-names>PG</given-names></name><name><surname>Rabchevsky</surname> <given-names>AG</given-names></name><name><surname>Scheff</surname> <given-names>SW</given-names></name><name><surname>Steward</surname> <given-names>O</given-names></name><name><surname>Courtine</surname> <given-names>G</given-names></name><name><surname>Edgerton</surname> <given-names>VR</given-names></name><name><surname>Tuszynski</surname> <given-names>MH</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Development of a database for translational spinal cord injury research</article-title><source>Journal of Neurotrauma</source><volume>31</volume><fpage>1789</fpage><lpage>1799</lpage><pub-id pub-id-type="doi">10.1089/neu.2014.3399</pub-id><pub-id pub-id-type="pmid">25077610</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Haefeli</surname> <given-names>J</given-names></name><name><surname>Salegio</surname> <given-names>EA</given-names></name><name><surname>Liu</surname> <given-names>AW</given-names></name><name><surname>Guandique</surname> <given-names>CF</given-names></name><name><surname>Stück</surname> <given-names>ED</given-names></name><name><surname>Hawbecker</surname> <given-names>S</given-names></name><name><surname>Moseanko</surname> <given-names>R</given-names></name><name><surname>Strand</surname> <given-names>SC</given-names></name><name><surname>Zdunowski</surname> <given-names>S</given-names></name><name><surname>Brock</surname> <given-names>JH</given-names></name><name><surname>Roy</surname> <given-names>RR</given-names></name><name><surname>Rosenzweig</surname> <given-names>ES</given-names></name><name><surname>Nout-Lomas</surname> <given-names>YS</given-names></name><name><surname>Courtine</surname> <given-names>G</given-names></name><name><surname>Havton</surname> <given-names>LA</given-names></name><name><surname>Steward</surname> <given-names>O</given-names></name><name><surname>Reggie Edgerton</surname> <given-names>V</given-names></name><name><surname>Tuszynski</surname> <given-names>MH</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Leveraging biomedical informatics for assessing plasticity and repair in primate spinal cord injury</article-title><source>Brain Research</source><volume>1619</volume><fpage>124</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.1016/j.brainres.2014.10.048</pub-id><pub-id pub-id-type="pmid">25451131</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Cooper</surname> <given-names>SR</given-names></name><name><surname>Yue</surname> <given-names>JK</given-names></name><name><surname>Sorani</surname> <given-names>MD</given-names></name><name><surname>Inoue</surname> <given-names>T</given-names></name><name><surname>Yuh</surname> <given-names>EL</given-names></name><name><surname>Mukherjee</surname> <given-names>P</given-names></name><name><surname>Petrossian</surname> <given-names>TC</given-names></name><name><surname>Paquette</surname> <given-names>J</given-names></name><name><surname>Lum</surname> <given-names>PY</given-names></name><name><surname>Carlsson</surname> <given-names>GE</given-names></name><name><surname>Vassar</surname> <given-names>MJ</given-names></name><name><surname>Lingsma</surname> <given-names>HF</given-names></name><name><surname>Gordon</surname> <given-names>WA</given-names></name><name><surname>Valadka</surname> <given-names>AB</given-names></name><name><surname>Okonkwo</surname> <given-names>DO</given-names></name><name><surname>Manley</surname> <given-names>GT</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><collab>TRACK-TBI Investigators</collab></person-group><year iso-8601-date="2017">2017</year><article-title>Uncovering precision phenotype-biomarker associations in traumatic brain injury using topological data analysis</article-title><source>PLOS ONE</source><volume>12</volume><elocation-id>e0169490</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0169490</pub-id><pub-id pub-id-type="pmid">28257413</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nielson</surname> <given-names>JL</given-names></name><name><surname>Cooper</surname> <given-names>SR</given-names></name><name><surname>Seabury</surname> <given-names>SA</given-names></name><name><surname>Luciani</surname> <given-names>D</given-names></name><name><surname>Fabio</surname> <given-names>A</given-names></name><name><surname>Temkin</surname> <given-names>NR</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><collab>TRACK-TBI Investigators</collab></person-group><year iso-8601-date="2020">2020</year><article-title>StatisticalStatistical guidelines for handling missing data in traumatic brain injury clinical research</article-title><source>Journal of Neurotrauma</source><volume>10</volume><elocation-id>6702</elocation-id><pub-id pub-id-type="doi">10.1089/neu.2019.6702</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Panaretos</surname> <given-names>D</given-names></name><name><surname>Tzavelas</surname> <given-names>G</given-names></name><name><surname>Vamvakari</surname> <given-names>M</given-names></name><name><surname>Panagiotakos</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2017">2017</year><source>Factor Analysis as a Tool for Pattern Recognition in Biomedical Research; a Review with Application in R Software</source></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Peres-Neto</surname> <given-names>PR</given-names></name><name><surname>Jackson</surname> <given-names>DA</given-names></name><name><surname>Somers</surname> <given-names>KM</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Giving meaningful interpretation to ordination axes: assessing loading significance in principal component analysis</article-title><source>Ecology</source><volume>84</volume><fpage>2347</fpage><lpage>2363</lpage><pub-id pub-id-type="doi">10.1890/00-0634</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="software"><person-group person-group-type="author"><collab>R Development Core Team</collab></person-group><year iso-8601-date="2019">2019</year><data-title>R: A Language and Environment for Statistical Computing</data-title><publisher-loc>Vienna, Austria</publisher-loc><publisher-name>R Foundation for Statistical Computing</publisher-name><ext-link ext-link-type="uri" xlink:href="https://www.R-project.org/">https://www.R-project.org/</ext-link></element-citation></ref><ref id="bib68"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Revelle</surname> <given-names>WR</given-names></name></person-group><year iso-8601-date="2017">2017</year><source>Psych: Procedures for Personality and Psychological Research</source><ext-link ext-link-type="uri" xlink:href="https://www.scholars.northwestern.edu/en/publications/psych-procedures-for-personality-and-psychological-research">https://www.scholars.northwestern.edu/en/publications/psych-procedures-for-personality-and-psychological-research</ext-link></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rosenzweig</surname> <given-names>ES</given-names></name><name><surname>Courtine</surname> <given-names>G</given-names></name><name><surname>Jindrich</surname> <given-names>DL</given-names></name><name><surname>Brock</surname> <given-names>JH</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Strand</surname> <given-names>SC</given-names></name><name><surname>Nout</surname> <given-names>YS</given-names></name><name><surname>Roy</surname> <given-names>RR</given-names></name><name><surname>Miller</surname> <given-names>DM</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Havton</surname> <given-names>LA</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Edgerton</surname> <given-names>VR</given-names></name><name><surname>Tuszynski</surname> <given-names>MH</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Extensive spontaneous plasticity of corticospinal projections after primate spinal cord injury</article-title><source>Nature Neuroscience</source><volume>13</volume><fpage>1505</fpage><lpage>1510</lpage><pub-id pub-id-type="doi">10.1038/nn.2691</pub-id><pub-id pub-id-type="pmid">21076427</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rosenzweig</surname> <given-names>ES</given-names></name><name><surname>Brock</surname> <given-names>JH</given-names></name><name><surname>Lu</surname> <given-names>P</given-names></name><name><surname>Kumamaru</surname> <given-names>H</given-names></name><name><surname>Salegio</surname> <given-names>EA</given-names></name><name><surname>Kadoya</surname> <given-names>K</given-names></name><name><surname>Weber</surname> <given-names>JL</given-names></name><name><surname>Liang</surname> <given-names>JJ</given-names></name><name><surname>Moseanko</surname> <given-names>R</given-names></name><name><surname>Hawbecker</surname> <given-names>S</given-names></name><name><surname>Huie</surname> <given-names>JR</given-names></name><name><surname>Havton</surname> <given-names>LA</given-names></name><name><surname>Nout-Lomas</surname> <given-names>YS</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Tuszynski</surname> <given-names>MH</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Restorative effects of human neural stem cell grafts on the primate spinal cord</article-title><source>Nature Medicine</source><volume>24</volume><fpage>484</fpage><lpage>490</lpage><pub-id pub-id-type="doi">10.1038/nm.4502</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rosenzweig</surname> <given-names>ES</given-names></name><name><surname>Salegio</surname> <given-names>EA</given-names></name><name><surname>Liang</surname> <given-names>JJ</given-names></name><name><surname>Weber</surname> <given-names>JL</given-names></name><name><surname>Weinholtz</surname> <given-names>CA</given-names></name><name><surname>Brock</surname> <given-names>JH</given-names></name><name><surname>Moseanko</surname> <given-names>R</given-names></name><name><surname>Hawbecker</surname> <given-names>S</given-names></name><name><surname>Pender</surname> <given-names>R</given-names></name><name><surname>Cruzen</surname> <given-names>CL</given-names></name><name><surname>Iaci</surname> <given-names>JF</given-names></name><name><surname>Caggiano</surname> <given-names>AO</given-names></name><name><surname>Blight</surname> <given-names>AR</given-names></name><name><surname>Haenzi</surname> <given-names>B</given-names></name><name><surname>Huie</surname> <given-names>JR</given-names></name><name><surname>Havton</surname> <given-names>LA</given-names></name><name><surname>Nout-Lomas</surname> <given-names>YS</given-names></name><name><surname>Fawcett</surname> <given-names>JW</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Beattie</surname> <given-names>MS</given-names></name><name><surname>Bresnahan</surname> <given-names>JC</given-names></name><name><surname>Tuszynski</surname> <given-names>MH</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Chondroitinase improves anatomical and functional outcomes after primate spinal cord injury</article-title><source>Nature Neuroscience</source><volume>22</volume><fpage>1269</fpage><lpage>1275</lpage><pub-id pub-id-type="doi">10.1038/s41593-019-0424-1</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rubin</surname> <given-names>DB</given-names></name></person-group><year iso-8601-date="1976">1976</year><article-title>Inference and missing data</article-title><source>Biometrika</source><volume>63</volume><fpage>581</fpage><lpage>592</lpage><pub-id pub-id-type="doi">10.1093/biomet/63.3.581</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schafer</surname> <given-names>JL</given-names></name><name><surname>Graham</surname> <given-names>JW</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Missing data: Our view of the state of the art</article-title><source>Psychological Methods</source><volume>7</volume><fpage>147</fpage><lpage>177</lpage><pub-id pub-id-type="doi">10.1037/1082-989X.7.2.147</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Slowikowski</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>Ggrepel: Automatically Position Non-Overlapping Text Labels with Ggplot2</source><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=ggrepel">https://CRAN.R-project.org/package=ggrepel</ext-link></element-citation></ref><ref id="bib75"><element-citation publication-type="software"><person-group person-group-type="author"><collab>Team RS</collab></person-group><year iso-8601-date="2018">2018</year><source>RStudio: Integrated Development for R. RStudio, Inc</source><ext-link ext-link-type="uri" xlink:href="http://www.rstudio.com/">http://www.rstudio.com/</ext-link></element-citation></ref><ref id="bib76"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Tierney</surname> <given-names>N</given-names></name><name><surname>Cook</surname> <given-names>D</given-names></name><name><surname>McBain</surname> <given-names>M</given-names></name><name><surname>Fay</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Naniar: Data Structures, Summaries, and Visualisations for Missing Data (0.5.0)</source><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=naniar">https://CRAN.R-project.org/package=naniar</ext-link></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Timmerman</surname> <given-names>ME</given-names></name><name><surname>Kiers</surname> <given-names>HAL</given-names></name><name><surname>Smilde</surname> <given-names>AK</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Estimating confidence intervals for principal component loadings: A comparison between the bootstrap and asymptotic results</article-title><source>British Journal of Mathematical and Statistical Psychology</source><volume>60</volume><fpage>295</fpage><lpage>314</lpage><pub-id pub-id-type="doi">10.1348/000711006X109636</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Tucker</surname> <given-names>LR</given-names></name></person-group><year iso-8601-date="1951">1951</year><source>A Method for Synthesis of Factor Analysis Studies</source><publisher-name>Department of the Army</publisher-name></element-citation></ref><ref id="bib79"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Urbanek</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2013">2013</year><source>png: Read and write PNG images</source><version designator="0.1-7">R package version 0.1-7</version><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=png">https://CRAN.R-project.org/package=png</ext-link></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Ginkel</surname> <given-names>JR</given-names></name><name><surname>Kroonenberg</surname> <given-names>PM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Using Generalized Procrustes Analysis for Multiple Imputation in Principal Component Analysis</article-title><source>Journal of Classification</source><volume>31</volume><fpage>242</fpage><lpage>269</lpage><pub-id pub-id-type="doi">10.1007/s00357-014-9154-y</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vitale</surname> <given-names>R</given-names></name><name><surname>Westerhuis</surname> <given-names>JA</given-names></name><name><surname>Naes</surname> <given-names>T</given-names></name><name><surname>Smilde</surname> <given-names>AK</given-names></name><name><surname>de Noord</surname> <given-names>OE</given-names></name><name><surname>Ferrer</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Selecting the number of factors in principal component analysis by permutation testing-Numerical and practical aspects</article-title><source>Journal of Chemometrics</source><volume>31</volume><elocation-id>e2937</elocation-id><pub-id pub-id-type="doi">10.1002/cem.2937</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><source>ggplot2: Elegant Graphics for Data Analysis</source><publisher-name>Springer-Verlag</publisher-name></element-citation></ref><ref id="bib83"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name><name><surname>François</surname> <given-names>R</given-names></name><name><surname>Henry</surname> <given-names>L</given-names></name><name><surname>Müller</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2018">2018</year><source>dplyr: A Grammar of Data Manipulation</source><version designator="0.8.4">R package version 0.8.4</version></element-citation></ref><ref id="bib84"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>stringr: Simple, Consistent Wrappers for Common String Operations</source><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=stringr">https://CRAN.R-project.org/package=stringr</ext-link></element-citation></ref><ref id="bib85"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name><name><surname>Henry</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>tidyr: Tidy Messy Data</source><ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=tidyr">https://CRAN.R-project.org/package=tidyr</ext-link></element-citation></ref><ref id="bib86"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wilkinson</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2005">2005</year><source>The Grammar of Graphics</source><publisher-name>Springer-Verlag</publisher-name></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yue</surname> <given-names>JK</given-names></name><name><surname>Vassar</surname> <given-names>MJ</given-names></name><name><surname>Lingsma</surname> <given-names>HF</given-names></name><name><surname>Cooper</surname> <given-names>SR</given-names></name><name><surname>Okonkwo</surname> <given-names>DO</given-names></name><name><surname>Valadka</surname> <given-names>AB</given-names></name><name><surname>Gordon</surname> <given-names>WA</given-names></name><name><surname>Maas</surname> <given-names>AIR</given-names></name><name><surname>Mukherjee</surname> <given-names>P</given-names></name><name><surname>Yuh</surname> <given-names>EL</given-names></name><name><surname>Puccio</surname> <given-names>AM</given-names></name><name><surname>Schnyer</surname> <given-names>DM</given-names></name><name><surname>Casey</surname> <given-names>SS</given-names></name><name><surname>Cheong</surname> <given-names>M</given-names></name><name><surname>Dams-O'Connor</surname> <given-names>K</given-names></name><name><surname>Hricik</surname> <given-names>AJ</given-names></name><name><surname>Knight</surname> <given-names>EE</given-names></name><name><surname>Kulubya</surname> <given-names>ES</given-names></name><name><surname>Menon</surname> <given-names>DK</given-names></name><name><surname>Morabito</surname> <given-names>DJ</given-names></name><name><surname>Pacheco</surname> <given-names>JL</given-names></name><name><surname>Sinha</surname> <given-names>TK</given-names></name><name><surname>Manley</surname> <given-names>GT</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Transforming research and clinical Knowledge in traumatic brain injury pilot: multicenter implementation of the common data elements for traumatic brain injury</article-title><source>Journal of Neurotrauma</source><volume>30</volume><fpage>1831</fpage><lpage>1844</lpage><pub-id pub-id-type="doi">10.1089/neu.2013.2970</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yue</surname> <given-names>JK</given-names></name><name><surname>Winkler</surname> <given-names>EA</given-names></name><name><surname>Rick</surname> <given-names>JW</given-names></name><name><surname>Burke</surname> <given-names>JF</given-names></name><name><surname>McAllister</surname> <given-names>TW</given-names></name><name><surname>Oh</surname> <given-names>SS</given-names></name><name><surname>Burchard</surname> <given-names>EG</given-names></name><name><surname>Hu</surname> <given-names>D</given-names></name><name><surname>Rosand</surname> <given-names>J</given-names></name><name><surname>Temkin</surname> <given-names>NR</given-names></name><name><surname>Korley</surname> <given-names>FK</given-names></name><name><surname>Sorani</surname> <given-names>MD</given-names></name><name><surname>Ferguson</surname> <given-names>AR</given-names></name><name><surname>Lingsma</surname> <given-names>HF</given-names></name><name><surname>Sharma</surname> <given-names>S</given-names></name><name><surname>Robinson</surname> <given-names>CK</given-names></name><name><surname>Yuh</surname> <given-names>EL</given-names></name><name><surname>Tarapore</surname> <given-names>PE</given-names></name><name><surname>Wang</surname> <given-names>KKW</given-names></name><name><surname>Puccio</surname> <given-names>AM</given-names></name><name><surname>Mukherjee</surname> <given-names>P</given-names></name><name><surname>Diaz-Arrastia</surname> <given-names>R</given-names></name><name><surname>Gordon</surname> <given-names>WA</given-names></name><name><surname>Valadka</surname> <given-names>AB</given-names></name><name><surname>Okonkwo</surname> <given-names>DO</given-names></name><name><surname>Manley</surname> <given-names>GT</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>DRD2 C957T polymorphism is associated with improved 6-month verbal learning following traumatic brain injury</article-title><source>Neurogenetics</source><volume>18</volume><fpage>29</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1007/s10048-016-0500-6</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zabala</surname> <given-names>A</given-names></name><name><surname>Pascual</surname> <given-names>U</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Bootstrapping Q Methodology to Improve the Understanding of Human Perspectives</article-title><source>PLOS ONE</source><volume>11</volume><elocation-id>e0148087</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0148087</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z</given-names></name><name><surname>Castelló</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Principal components analysis in clinical studies</article-title><source>Annals of Translational Medicine</source><volume>5</volume><elocation-id>351</elocation-id><pub-id pub-id-type="doi">10.21037/atm.2017.07.12</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zientek</surname> <given-names>LR</given-names></name><name><surname>Thompson</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Applying the bootstrap to the multivariate case: Bootstrap component/factor analysis</article-title><source>Behavior Research Methods</source><volume>39</volume><fpage>318</fpage><lpage>325</lpage><pub-id pub-id-type="doi">10.3758/BF03193163</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zwick</surname> <given-names>WR</given-names></name><name><surname>Velicer</surname> <given-names>WF</given-names></name></person-group><year iso-8601-date="1986">1986</year><article-title>Comparison of five rules for determining the number of components to retain</article-title><source>Psychological Bulletin</source><volume>99</volume><fpage>432</fpage><lpage>442</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.99.3.432</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.61812.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Zaidi</surname><given-names>Mone</given-names></name><role>Reviewing Editor</role><aff><institution>Icahn School of Medicine at Mount Sinai</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>The paper documents the implementation of syndRomics, an analytical framework for measuring disease states using principal component analysis and multivariate statistics as primary tools for extracting underlying disease patterns in neurological trauma. The method is robust and will serve as an open-source R package for the visualization of disease component more broadly.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Reproducible analysis of disease space via principal components: a brief tutorial and R package (syndRomics)&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by two peer reviewers, and the evaluation has been overseen by a Reviewing Editor and a Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers have discussed the reviews with one another and the Reviewing Editor has drafted this decision to help you prepare a revised submission.</p><p>We would like to draw your attention to changes in our revision policy that we have made in response to COVID-19 (https://elifesciences.org/articles/57162). Specifically, when editors judge that a submitted work as a whole belongs in <italic>eLife</italic> but that some conclusions require a modest amount of additional new data, as they do with your paper, we are asking that the manuscript be revised to either limit claims to those supported by data in hand, or to explicitly state that the relevant conclusions require additional supporting data.</p><p>Our expectation is that the authors will eventually carry out the additional experiments and report on how they affect the relevant conclusions either in a preprint on bioRxiv or medRxiv, or if appropriate, as a Research Advance in <italic>eLife</italic>, either of which would be linked to the original paper.</p><p>Summary:</p><p>The manuscript is a tutorial and an open-source software to analyze disease patterns using principal components analysis. The software – “syndromics” – is available as a part of an R-package. The authors document the implementation of syndromics in the case studies of neurological trauma data and provide a practical guide to the application of PCA to extract disease patterns.</p><p>Essential revisions:</p><p>This is a well-written manuscript that could be a helpful manual to biomedical researchers in many different fields. The authors present a new software package called &quot;syndRomics&quot;, to extract disease features using PCA and describe the reproducible analysis workflow. Whereas there is general enthusiasm, the Editors request that specific issues need to be addressed.</p><p>1) If the authors wish to make syndromics as specialized “one-stop” package for neurotrauma data analysis then they should go over and above to show that PCA works on multiple and different data sets than just one that is presented in the manuscript.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.61812.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>This is a well-written manuscript that could be a helpful manual to biomedical researchers in many different fields. The authors present a new software package called &quot;syndRomics&quot;, to extract disease features using PCA and describe the reproducible analysis workflow. Whereas there is general enthusiasm, the Editors request that specific issues need to be addressed.</p><p>1) If the authors wish to make syndromics as specialized “one-stop” package for neurotrauma data analysis then they should go over and above to show that PCA works on multiple and different data sets than just one that is presented in the manuscript.</p></disp-quote><p>In response to the reviewers, we have incorporated the analysis on another publicly available dataset of clinical neurotrauma as a second case study to illustrate the utility of the proposed analytical workflow and the use of the package. This dataset is a subset of the variables of the Transforming Research and Clinical Knowledge in Traumatic Brain Injury (TRACK-TBI) Pilot Study. The mixed variable type nature of the dataset allows us to illustrate the use of the package in a nonlinear variant of PCA obtained from functions implemented in the Gifi R package. We demonstrate that the use of syndRomics analysis through nonlinear PCA in this dataset resolves in patterns of variable associations previously described in the literature, but in an unsupervised manner, illustrating the value of the analysis for extracting informative disease patterns. Moreover, this second case study serves as an example of the importance of studying the sensitivity of the analysis in different settings such as multiple imputation of missingness. Altogether, we believe the incorporation of this second use case helps to illustrate generalizability of the package and the analytical workflow in different biomedical settings. The analysis has been added at the end of the Results section, including a new figure in the main text (Figure 7), a new supplementary figure (Figure 7—figure supplement 1) and four new tables.</p></body></sub-article></article>