<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS ONE</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PONE-D-18-03097</article-id>
<article-id pub-id-type="doi">10.1371/journal.pone.0202344</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Cardiology</subject><subj-group><subject>Myocardial infarction</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Engineering and technology</subject><subj-group><subject>Management engineering</subject><subj-group><subject>Decision analysis</subject><subj-group><subject>Decision trees</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Research and analysis methods</subject><subj-group><subject>Decision analysis</subject><subj-group><subject>Decision trees</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Computer and information sciences</subject><subj-group><subject>Artificial intelligence</subject><subj-group><subject>Machine learning</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Vascular medicine</subject><subj-group><subject>Angina</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Ecology</subject><subj-group><subject>Ecosystems</subject><subj-group><subject>Forests</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Ecology and environmental sciences</subject><subj-group><subject>Ecology</subject><subj-group><subject>Ecosystems</subject><subj-group><subject>Forests</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Ecology and environmental sciences</subject><subj-group><subject>Terrestrial environments</subject><subj-group><subject>Forests</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Surgical and invasive medical procedures</subject><subj-group><subject>Cardiovascular procedures</subject><subj-group><subject>Coronary artery bypass grafting</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Biochemistry</subject><subj-group><subject>Lipids</subject><subj-group><subject>Cholesterol</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Vascular medicine</subject><subj-group><subject>Coronary heart disease</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Medicine and health sciences</subject><subj-group><subject>Cardiology</subject><subj-group><subject>Coronary heart disease</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Machine learning models in electronic health records can outperform conventional survival models for predicting patient mortality in coronary artery disease</article-title>
<alt-title alt-title-type="running-head">Machine learning can outperform conventional survival models in electronic health records</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-6836-5925</contrib-id>
<name name-style="western">
<surname>Steele</surname> <given-names>Andrew J.</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Data curation</role>
<role content-type="http://credit.casrai.org/">Formal analysis</role>
<role content-type="http://credit.casrai.org/">Investigation</role>
<role content-type="http://credit.casrai.org/">Methodology</role>
<role content-type="http://credit.casrai.org/">Software</role>
<role content-type="http://credit.casrai.org/">Validation</role>
<role content-type="http://credit.casrai.org/">Visualization</role>
<role content-type="http://credit.casrai.org/">Writing – original draft</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Denaxas</surname> <given-names>Spiros C.</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Data curation</role>
<role content-type="http://credit.casrai.org/">Investigation</role>
<role content-type="http://credit.casrai.org/">Project administration</role>
<role content-type="http://credit.casrai.org/">Supervision</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Shah</surname> <given-names>Anoop D.</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Data curation</role>
<role content-type="http://credit.casrai.org/">Investigation</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Hemingway</surname> <given-names>Harry</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Funding acquisition</role>
<role content-type="http://credit.casrai.org/">Project administration</role>
<role content-type="http://credit.casrai.org/">Resources</role>
<role content-type="http://credit.casrai.org/">Supervision</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Luscombe</surname> <given-names>Nicholas M.</given-names></name>
<role content-type="http://credit.casrai.org/">Conceptualization</role>
<role content-type="http://credit.casrai.org/">Funding acquisition</role>
<role content-type="http://credit.casrai.org/">Project administration</role>
<role content-type="http://credit.casrai.org/">Resources</role>
<role content-type="http://credit.casrai.org/">Supervision</role>
<role content-type="http://credit.casrai.org/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff004"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff005"><sup>5</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>The Francis Crick Institute, London, United Kingdom</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Farr Institute of Health Informatics Research, Institute of Health Informatics, University College London, London, United Kingdom</addr-line>
</aff>
<aff id="aff003">
<label>3</label>
<addr-line>University College London Hospitals NHS Foundation Trust, London, United Kingdom</addr-line>
</aff>
<aff id="aff004">
<label>4</label>
<addr-line>UCL Genetics Institute, Department of Genetics Evolution and Environment, University College London, London, United Kingdom</addr-line>
</aff>
<aff id="aff005">
<label>5</label>
<addr-line>Okinawa Institute of Science &amp; Technology Graduate University, Okinawa, Japan</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Singh</surname> <given-names>Tiratha Raj</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>Jaypee University of Information Technology, INDIA</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">statto+plosone@andrewsteele.co.uk</email></corresp>
</author-notes>
<pub-date pub-type="collection">
<year>2018</year>
</pub-date>
<pub-date pub-type="epub">
<day>31</day>
<month>8</month>
<year>2018</year>
</pub-date>
<volume>13</volume>
<issue>8</issue>
<elocation-id>e0202344</elocation-id>
<history>
<date date-type="received">
<day>5</day>
<month>2</month>
<year>2018</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>7</month>
<year>2018</year>
</date>
</history>
<permissions>
<copyright-year>2018</copyright-year>
<copyright-holder>Steele et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0202344"/>
<abstract>
<p>Prognostic modelling is important in clinical practice and epidemiology for patient management and research. Electronic health records (EHR) provide large quantities of data for such models, but conventional epidemiological approaches require significant researcher time to implement. Expert selection of variables, fine-tuning of variable transformations and interactions, and imputing missing values are time-consuming and could bias subsequent analysis, particularly given that missingness in EHR is both high, and may carry meaning. Using a cohort of 80,000 patients from the CALIBER programme, we compared traditional modelling and machine-learning approaches in EHR. First, we used Cox models and random survival forests with and without imputation on 27 expert-selected, preprocessed variables to predict all-cause mortality. We then used Cox models, random forests and elastic net regression on an extended dataset with 586 variables to build prognostic models and identify novel prognostic factors without prior expert input. We observed that data-driven models used on an extended dataset can outperform conventional models for prognosis, without data preprocessing or imputing missing values. An elastic net Cox regression based with 586 unimputed variables with continuous values discretised achieved a C-index of 0.801 (bootstrapped 95% CI 0.799 to 0.802), compared to 0.793 (0.791 to 0.794) for a traditional Cox model comprising 27 expert-selected variables with imputation for missing values. We also found that data-driven models allow identification of novel prognostic variables; that the absence of values for particular variables carries meaning, and can have significant implications for prognosis; and that variables often have a nonlinear association with mortality, which discretised Cox models and random forests can elucidate. This demonstrates that machine-learning approaches applied to raw EHR data can be used to build models for use in research and clinical practice, and identify novel predictive variables and their effects to inform future research.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000265</institution-id>
<institution>Medical Research Council</institution>
</institution-wrap>
</funding-source>
<award-id>MR/L016311/1</award-id>
</award-group>
<award-group id="award002">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000265</institution-id>
<institution>Medical Research Council</institution>
</institution-wrap>
</funding-source>
<award-id>FC001110</award-id>
</award-group>
<award-group id="award003">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000289</institution-id>
<institution>Cancer Research UK</institution>
</institution-wrap>
</funding-source>
<award-id>FC001110</award-id>
</award-group>
<award-group id="award004">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100004440</institution-id>
<institution>Wellcome Trust</institution>
</institution-wrap>
</funding-source>
<award-id>FC001110</award-id>
</award-group>
<award-group id="award005">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100004440</institution-id>
<institution>Wellcome Trust</institution>
</institution-wrap>
</funding-source>
<award-id>WT 086091/Z/08/Z</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Hemingway</surname> <given-names>Harry</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award006">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000272</institution-id>
<institution>National Institute for Health Research</institution>
</institution-wrap>
</funding-source>
<award-id>RP-PG-0407-10314</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Hemingway</surname> <given-names>Harry</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award007">
<funding-source>
<institution>Medical Research Prognosis Research Strategy Partnership</institution>
</funding-source>
<award-id>G0902393/99558</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Hemingway</surname> <given-names>Harry</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award008">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000265</institution-id>
<institution>Medical Research Council</institution>
</institution-wrap>
</funding-source>
<award-id>K006584/1</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Hemingway</surname> <given-names>Harry</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>This work was supported by the Francis Crick Institute which receives its core funding from Cancer Research UK (FC001110), the UK Medical Research Council (FC001110), and the Wellcome Trust (FC001110). NML and HH were supported by the Medical Research Council Medical Bioinformatics Award eMedLab (grant number MR/L016311/1). The CALIBER programme was supported by the National Institute for Health Research (RP-PG-0407-10314, PI HH); Wellcome Trust (WT 086091/Z/08/Z, PI HH); the Medical Research Prognosis Research Strategy Partnership (G0902393/99558, PI HH) and the Farr Institute of Health Informatics Research, funded by the Medical Research Council (K006584/1, PI HH), in partnership with Arthritis Research UK, the British Heart Foundation, Cancer Research UK, the Economic and Social Research Council, the Engineering and Physical Sciences Research Council, the National Institute of Health Research, the National Institute for Social Care and Health Research (Welsh Assembly Government), the Chief Scientist Office (Scottish Government Health Directorates) and the Wellcome Trust.</funding-statement>
</funding-group>
<counts>
<fig-count count="5"/>
<table-count count="2"/>
<page-count count="20"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>While our data does not contain any personal sensitive identifiers, it’s deemed as sensitive as it contains sufficient clinical information about patients such as dates of clinical events for there to be a potential risk of patient re-identification. This restriction has been imposed by the data owner (CPRD/MHRA) and the data sharing agreements between UCL and the CPRD/MHRA. Access to data may be requested via the Clinical Practice Research Datalink (CPRD) and applying to the CPRD’s Independent Scientific Advisory Committee (<ext-link ext-link-type="uri" xlink:href="https://www.cprd.com/researcher/" xlink:type="simple">https://www.cprd.com/researcher/</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Advances in precision medicine will require increasingly individualised prognostic assessments for patients in order to guide appropriate therapy. Conventional statistical methods for prognostic modelling require significant human involvement in selection of prognostic variables (based on an understanding of disease aetiology), variable transformation and imputation, and model optimisation on a per-condition basis.</p>
<p>Electronic health records (EHR) contain large amounts of information about patients’ medical history including symptoms, examination findings, test results, prescriptions and procedures. The increasing quantity of data in EHR means that they are becoming more valuable for research [<xref ref-type="bibr" rid="pone.0202344.ref001">1</xref>–<xref ref-type="bibr" rid="pone.0202344.ref005">5</xref>]. Although EHR are a rich data source, many of the data items are collected in a non-systematic manner according to clinical need, so missingness is often high [<xref ref-type="bibr" rid="pone.0202344.ref006">6</xref>, <xref ref-type="bibr" rid="pone.0202344.ref007">7</xref>]. The population of patients with missing data may be systematically different depending on the reason that the data are missing [<xref ref-type="bibr" rid="pone.0202344.ref008">8</xref>]: tests may be omitted if the clinician judges they are not necessary [<xref ref-type="bibr" rid="pone.0202344.ref009">9</xref>], the patient refuses [<xref ref-type="bibr" rid="pone.0202344.ref010">10</xref>], or the patient fails to attend. Multiple imputation to handle missing data can be computationally intensive, and needs to include sufficient information about the reason for missingness to avoid bias [<xref ref-type="bibr" rid="pone.0202344.ref011">11</xref>].</p>
<sec id="sec002">
<title>Conventional versus data-driven approaches to prognostic modelling</title>
<p>Conventional statistical models, with a priori expert selection of predictor variables [<xref ref-type="bibr" rid="pone.0202344.ref001">1</xref>, <xref ref-type="bibr" rid="pone.0202344.ref012">12</xref>], have a number of potential shortcomings. First, they may be time-consuming to fit and require expert knowledge of the aetiology of a given condition [<xref ref-type="bibr" rid="pone.0202344.ref013">13</xref>]. Second, such models are unable to utilise the richness of EHR data; a recent meta-analysis [<xref ref-type="bibr" rid="pone.0202344.ref001">1</xref>] found that EHR studies used a median of just 27 variables, despite thousands potentially being available [<xref ref-type="bibr" rid="pone.0202344.ref005">5</xref>]. Third, parametric models rely on assumptions which may not be borne out in practice. For example Cox proportional hazards models assume that a change in a predictor variable is associated with a multiplicative response in the baseline hazard that is constant over time. Nonlinearity and interactions need to be built into models explicitly based on prior clinical knowledge. Finally, missing data need to be handled using a separate process such as multiple imputation. Given that most imputation techniques assume that data are missing at random [<xref ref-type="bibr" rid="pone.0202344.ref011">11</xref>], this may mean that their results are unreliable in this context. Imputation can also be time-consuming both for researchers, and computationally given the large cohorts available in EHR data. Categorisation of continuous variables can accommodate nonlinear relationships and allow the inclusion of missing data as an additional category [<xref ref-type="bibr" rid="pone.0202344.ref014">14</xref>], but may substantially increase the number of parameters in the model.</p>
<p>Machine-learning approaches include automatic variable selection techniques and non-parametric regression methods which can handle large numbers of predictors and may not require as many assumptions about the relationship between particular variables and outcomes of interest [<xref ref-type="bibr" rid="pone.0202344.ref015">15</xref>]. These have the potential to reduce the amount of human intervention required in fitting prognostic models [<xref ref-type="bibr" rid="pone.0202344.ref016">16</xref>–<xref ref-type="bibr" rid="pone.0202344.ref018">18</xref>]. For example, random survival forests [<xref ref-type="bibr" rid="pone.0202344.ref019">19</xref>–<xref ref-type="bibr" rid="pone.0202344.ref021">21</xref>] can accommodate nonlinearities and interactions between variables, and are not restricted to a common baseline hazard for all patients, avoiding the assumptions inherent in Cox proportional hazards models.</p>
<p>Previous studies have used machine learning on EHR data for tasks such as patient classification and diagnosis [<xref ref-type="bibr" rid="pone.0202344.ref022">22</xref>–<xref ref-type="bibr" rid="pone.0202344.ref024">24</xref>] or predicting future hospitalisation [<xref ref-type="bibr" rid="pone.0202344.ref025">25</xref>], but we are unaware of any systematic comparison of machine learning methods for predicting all-cause mortality in a large, richly characterised EHR cohort of patients with stable coronary artery disease.</p>
<p>In this study we compared different analytic approaches for predicting mortality, using a Cox model with 27 expert-selected variables as the reference [<xref ref-type="bibr" rid="pone.0202344.ref013">13</xref>]. We aimed to:</p>
<list list-type="order">
<list-item>
<p>Compare continuous and discrete Cox models, random forests and elastic net regression for predicting all-cause mortality;</p>
</list-item>
<list-item>
<p>Compare methods for handling missing data, and data-driven versus expert variable selection;</p>
</list-item>
<list-item>
<p>Investigate whether data-driven approaches are able to identify novel variables and variable effects.</p>
</list-item>
</list>
</sec>
</sec>
<sec id="sec003" sec-type="materials|methods">
<title>Methods</title>
<sec id="sec004">
<title>Data sources</title>
<p>We used a cohort of over 80,000 patients from the CALIBER programme [<xref ref-type="bibr" rid="pone.0202344.ref026">26</xref>]. CALIBER links 4 sources of electronic health data in England: primary care health records (coded diagnoses, clinical measurements, and prescriptions) from 244 general practices contributing to the Clinical Practice Research Datalink (CPRD); coded hospital discharges (Hospital Episode Statistics, HES); the Myocardial Ischemia National Audit Project (MINAP); and death registrations (ONS). CALIBER includes about 4% of the population of England [<xref ref-type="bibr" rid="pone.0202344.ref027">27</xref>] and is representative in terms of age, sex, ethnicity, and mortality [<xref ref-type="bibr" rid="pone.0202344.ref026">26</xref>].</p>
</sec>
<sec id="sec005">
<title>Patient population</title>
<p>Patients were eligible to enter the cohort after being registered with the general practice for a year. If they had pre-existing coronary artery disease (myocardial infarction [MI], unstable angina or stable angina) they entered the cohort on their eligibility date, otherwise they entered on the date of the first stable angina diagnosis after eligibility, or six months after the first acute coronary syndrome diagnosis (MI or unstable angina) after eligibility. The six-month delay was chosen to differentiate long-term prognosis from the high-risk period that typically follows acute coronary syndromes. For patients whose index diagnosis was MI, we used information in CPRD and MINAP to attempt to classify the type of MI as ST-elevation myocardial infarction (STEMI) or non–ST-elevation myocardial infarction (NSTEMI).</p>
<p>Diagnoses were identified in CPRD, HES, or MINAP records according to definitions in the CALIBER data portal (<ext-link ext-link-type="uri" xlink:href="http://www.caliberresearch.org/portal" xlink:type="simple">www.caliberresearch.org/portal</ext-link>). Stable angina was defined by angina diagnoses in CPRD (Read codes) and HES (ICD-10 codes), repeat prescriptions for nitrates, coronary revascularisation (Read codes in CPRD or OPCS-4 codes in HES), or ischaemia test results (CPRD). Acute coronary syndromes were defined in MINAP or diagnoses in CPRD and HES.</p>
</sec>
<sec id="sec006">
<title>Follow-up and endpoints</title>
<p>The primary endpoint was death from any cause. Patients were followed up until they died (as identified in ONS or CPRD) or transferred out of practice, or until the last data collection date of the practice.</p>
</sec>
<sec id="sec007">
<title>Prognostic factors</title>
<p>Expert-selected predictors [<xref ref-type="bibr" rid="pone.0202344.ref028">28</xref>] were extracted from primary care (CPRD) and coded hospital discharges (HES). These included cardiovascular risk factors (e.g. hypertension, diabetes, smoking, lipid profile), laboratory values, pre-existing diagnoses and prescribed medication. Small-area index of multiple deprivation (IMD) score was derived from the patient’s postcode. <xref ref-type="table" rid="pone.0202344.t001">Table 1</xref> shows all the expert-selected predictors used.</p>
<table-wrap id="pone.0202344.t001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.t001</object-id>
<label>Table 1</label>
<caption>
<title>The 27 expert-selected predictors used.</title>
</caption>
<alternatives>
<graphic id="pone.0202344.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.t001" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left" style="border-bottom:thick">Category</th>
<th align="left" style="border-bottom:thick">Prognostic factors</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Sociodemographic characteristics</td>
<td align="left">Age, gender, most deprived quintile</td>
</tr>
<tr>
<td align="left">CVD diagnosis and severity</td>
<td align="left">SCAD subtype (stable angina, unstable angina, STEMI,<break/>NSTEMI, other CHD), PCI in last six months, CABG in<break/>last six months, previous/recurrent MI, use of nitrates</td>
</tr>
<tr>
<td align="left">CVD risk factors</td>
<td align="left">Smoking status (current, ex, never), hypertension, diabetes<break/>mellitus, total cholesterol, HDL</td>
</tr>
<tr>
<td align="left">CVD comorbidities</td>
<td align="left">Heart failure, peripheral arterial disease, atrial fibrillation,<break/>stroke</td>
</tr>
<tr>
<td align="left">Non-CVD comorbidities</td>
<td align="left">Chronic kidney disease, chronic obstructive pulmonary<break/>disease, cancer, chronic liver disease</td>
</tr>
<tr>
<td align="left">Psychosocial characteristics</td>
<td align="left">Depression at diagnosis, anxiety at diagnosis</td>
</tr>
<tr>
<td align="left">Biomarkers</td>
<td align="left">Heart rate, creatinine, white cell count, haemoglobin</td>
</tr>
</tbody>
</table>
</alternatives>
<table-wrap-foot>
<fn id="t001fn001">
<p>CABG = coronary artery bypass graft; CVD = cardiovascular disease; HDL = high-density lipoprotein; PCI = percutaneous coronary intervention; MI = myocardial infarction; NSTEMI = non-ST-elevated MI; SCAD = stable coronary artery disease; STEMI = ST-elevated MI.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>We also generated an extended set of predictor variables for the data-driven models, derived from rich data prior to the index date in CALIBER: primary care (CPRD) tables of coded diagnoses, clinical measurements and prescriptions; and hospital care (HES) tables of coded diagnoses and procedures. From each of these sources, we generated new binary variables for the presence or absence of a coded diagnosis or procedure, and numeric variables for laboratory or clinical measurements. We selected the 100 least missing variables from each source table, and the 100 least missing of both clinical history and measurements from the primary care data. After combining with the pre-selected predictors and removing duplicate and erroneous variables, there were 586 variables per patient in the extended dataset.</p>
</sec>
<sec id="sec008">
<title>Ethics</title>
<p>Approval was granted by the Independent Scientific Advisory Committee (ISAC) of the Medicines and Healthcare Products Regulatory Agency (protocol 14_107). Approval from ISAC is equivalent to approval from an institutional review board. All data were fully, irreversibly anonymised before access was provided to the authors.</p>
</sec>
<sec id="sec009">
<title>Statistical methods</title>
<sec id="sec010">
<title>Imputation of missing data</title>
<p>Multiple imputation was implemented using multivariate imputation by chained equations in the R package <monospace>mice</monospace> [<xref ref-type="bibr" rid="pone.0202344.ref029">29</xref>], using the full dataset of 115,305 patients. Imputation models included:</p>
<list list-type="bullet">
<list-item>
<p>Covariates at baseline, including all of the expert-selected predictors and additional variables: age, quadratic age, diabetes, smoking, systolic blood pressure, diastolic blood pressure, total cholesterol, HDL cholesterol, body mass index, serum creatinine, haemoglobin, total white blood cell count, CABG or PCI surgery in the six months prior to study entry, abdominal aortic aneurysm prior to study entry, index of multiple deprivation, ethnicity, hypertension diagnosis or medication prior to study entry, use of long acting nitrates prior to study entry, diabetes diagnosis prior to study entry, peripheral arterial disease prior to study entry, and history of myocardial infarction, depression, anxiety disorder, cancer, renal disease, liver disease, chronic obstructive pulmonary disease, atrial fibrillation, or stroke.</p>
</list-item>
<list-item>
<p>Prior (between 1 and 2 years before study entry) and post (between 0 and 2 years after study entry) averages of continuous expert-selected covariates and other measurements: haemoglobin A1c (HbA1c), eGFR score [<xref ref-type="bibr" rid="pone.0202344.ref030">30</xref>], lymphocyte counts, neutrophil counts, eosinophil counts, monocyte counts, basophil counts, platelet counts, pulse pressure.</p>
</list-item>
<list-item>
<p>The Nelson–Aalen hazard and the event status for all-cause mortality.</p>
</list-item>
</list>
<p>Since many of the continuous variables were non-normally distributed, all continuous variables were log-transformed for imputation and exponentiated back to their original scale for analysis. Imputed values were estimated separately for men and women, and five multiply imputed datasets were generated. We verified that the distributions of observed and imputed values of all variables were similarly distributed.</p>
</sec>
<sec id="sec011">
<title>Training, validation and test sets</title>
<p>Before building prognostic models for all-cause mortality, we randomly split the cohort into a training set (<inline-formula id="pone.0202344.e001"><alternatives><graphic id="pone.0202344.e001g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0202344.e001" xlink:type="simple"/><mml:math display="inline" id="M1"><mml:mfrac><mml:mn>2</mml:mn> <mml:mn>3</mml:mn></mml:mfrac></mml:math></alternatives></inline-formula> of the patients), and a test set (the remaining <inline-formula id="pone.0202344.e002"><alternatives><graphic id="pone.0202344.e002g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0202344.e002" xlink:type="simple"/><mml:math display="inline" id="M2"><mml:mfrac><mml:mn>1</mml:mn> <mml:mn>3</mml:mn></mml:mfrac></mml:math></alternatives></inline-formula>) which was used to evaluate performance of the final models. Performance statistics were calculated on repeated bootstrap samples from this held-out test set in order to estimate variability. Where cross-validation was performed other than elastic net regression, the training set was randomly split into three folds, and the model trained on two of the folds and performance validated against the third to establish optimal parameters, before those parameters were used to build a model on the full training set.</p>
</sec>
<sec id="sec012">
<title>Continuous Cox proportional hazards models</title>
<p>Cox proportional hazards models are survival models which assume all patients share a common baseline hazard function which is multiplied by a factor based on the values of various predictor variables for an individual.</p>
<p>Missing values in continuous variables were accommodated both by imputation, and, separately, by explicit inclusion in the model. The latter was done by first setting the value to 0, such that it did not contribute to the patient’s risk, and then employing a missingness indicator with an independent coefficient, meaning an additional binary dummy variable to account for whether a value is missing or not [<xref ref-type="bibr" rid="pone.0202344.ref008">8</xref>, <xref ref-type="bibr" rid="pone.0202344.ref014">14</xref>]. For example, risk as a function of time λ(<italic>t</italic>) in a Cox model with a single continuous variable with some missingness would be expressed as
<disp-formula id="pone.0202344.e003"><alternatives><graphic id="pone.0202344.e003g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0202344.e003" xlink:type="simple"/><mml:math display="block" id="M3"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mo>λ</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>t</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mn>0</mml:mn></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>t</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo form="prefix">exp</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>β</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msub><mml:mi>x</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mo>+</mml:mo> <mml:msub><mml:mi>β</mml:mi> <mml:mtext>missing</mml:mtext></mml:msub> <mml:msub><mml:mi>x</mml:mi> <mml:mtext>missing</mml:mtext></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(1)</label></disp-formula>
where λ<sub>0</sub>(<italic>t</italic>) is the baseline hazard function, <italic>β</italic><sub>1</sub> is the coefficient relating to the continuous variable; <italic>x</italic><sub>1</sub> is the variable’s value (standardised if necessary), set to 0 if the value is missing; <italic>β</italic><sub>missing</sub> is the coefficient for that value being missing; and <italic>x</italic><sub>missing</sub> is 0 if the value is present, and 1 if it is missing.</p>
</sec>
<sec id="sec013">
<title>Missing categorical data</title>
<p>Missingness in categorical data can be dealt with very simply, by adding an extra category for ‘missing’. This is compatible with both Cox models (in which variables with <italic>n</italic> categories are parameterised as <italic>n</italic> − 1 dummy variables) and random forests (which can include categorical variables as-is).</p>
</sec>
<sec id="sec014">
<title>Discrete Cox models for missing values</title>
<p>Another method to incorporate missing data is to discretise all continuous predictor variables, and include missing data as an additional category [<xref ref-type="bibr" rid="pone.0202344.ref008">8</xref>, <xref ref-type="bibr" rid="pone.0202344.ref014">14</xref>]. Choice of category boundaries then becomes an additional hyperparameter.</p>
<p>Discretisation schemes were chosen by cross-validation to avoid overfitting. Discretising a value too finely may allow a model to fit the training data with spurious precision and generalise poorly to new data, whilst doing so too coarsely risks losing information carried by that variable. Every continuous variable <italic>i</italic> was split into 10 ≤ <italic>n</italic><sub><italic>i</italic></sub> ≤ 20 bins, which were assigned in quantiles; for example, a 10-bin discretisation would split patients into deciles. An additional category was then assigned for missing values, resulting in <italic>n</italic><sub><italic>i</italic></sub> + 1 coefficients per variable when fitting the Cox model.</p>
</sec>
<sec id="sec015">
<title>Random survival forests</title>
<p>Random survival forests [<xref ref-type="bibr" rid="pone.0202344.ref019">19</xref>–<xref ref-type="bibr" rid="pone.0202344.ref021">21</xref>] is an alternative method for survival analysis which has previously been used to model deaths in the context of cardiovascular disease [<xref ref-type="bibr" rid="pone.0202344.ref031">31</xref>]. It is a machine-learning technique which builds a ‘forest’ of decision trees, each of which calculates patient outcomes by splitting them into groups with similar characteristics. These hundreds to thousands of decision trees each have random imperfections, meaning that whilst individual trees are then relatively poor predictors, the averaged result from the forest is more accurate and less prone to overfitting than an individual ‘perfect’ decision tree [<xref ref-type="bibr" rid="pone.0202344.ref032">32</xref>].</p>
<p>At each node in a decision tree, starting at its root, patients are split into two branches by looking at an individual covariate. The algorithm selects a split point which maximises the difference between the survival curves of patients in the two branches defined by that split. For a continuous parameter, this is a value where patients above are taken down one branch, and patients below are taken down the other; for a categorical variable, a subset of values is associated with each branch. Split points are defined so as to maximise the homogeneity within each branch, and the inhomogeneity between them.</p>
<p>Decision trees for regression or classification often use variance or Gini impurity, respectively. Survival forests, by contrast, maximise the difference between survival curves as measured by the logrank test [<xref ref-type="bibr" rid="pone.0202344.ref019">19</xref>], or optimise the C-index using which branch a patient falls into as a predictor [<xref ref-type="bibr" rid="pone.0202344.ref021">21</xref>]. After testing, we found splitting on C-index to be too computationally intensive for negligible performance benefit, so logrank tests were used.</p>
<p>This process is repeated until the leaves of the tree are reached. These are nodes where either there are no further criteria remains by which the patients at that leaf can be distinguished, including the possibility of only a single patient remaining, or splitting may be stopped early with a minimum node size or maximum tree depth.</p>
<p>In a random forest, each tree is created using only a random sample with replacement of the training data, and only a subset of parameters is considered when splitting at each node. A ‘forest’ comprising such trees can then use use the average, majority vote or combined survival curve, for regression, classification and survival forests, respectively, to aggregate the results of the individual trees and predict results for new data.</p>
<p>Random survival forests are thus, in principle, able to model arbitrarily shaped survival curves without assumptions such as proportional hazards, and can fit nonlinear responses to covariates, and arbitrary interactions between them.</p>
<p>One issue with random forests is that variables with a large number of different values offer many possible split points, which gives them more chances to provide the optimal split. One approach to combat this is to use multiple-comparisons corrections to account for the resulting increased likelihood of finding a split point in variables where there are many options [<xref ref-type="bibr" rid="pone.0202344.ref033">33</xref>, <xref ref-type="bibr" rid="pone.0202344.ref034">34</xref>], but this is complicated by split statistics for adjacent split points being correlated. To avoid this issue, we instead used a value <italic>n</italic><sub>split</sub> to fix the number of split points tested in continuous variables which also has the advantage of reducing computational time for model building [<xref ref-type="bibr" rid="pone.0202344.ref035">35</xref>].</p>
<p>Many methods have been proposed for dealing with missing data in random forests, and these primarily divide into two categories. Some methods effectively perform on-the-fly imputation: for example the surrogate variable method, proposed in the classic classification and regression trees (CART) decision tree algorithm, attempts to use nonmissing values to find the variable which splits the data most similarly to the variable initially selected for the split [<xref ref-type="bibr" rid="pone.0202344.ref036">36</xref>]. Other methods allow missing values to continue through the network, for example by assigning them at random to a branch weighted by the ratio of nonmissing values between branches, known as adaptive tree imputation [<xref ref-type="bibr" rid="pone.0202344.ref019">19</xref>], or sending missing cases down both branches, but with their weights diminished by the ratio of nonmissing values between branches, as used in the C4.5 algorithm [<xref ref-type="bibr" rid="pone.0202344.ref037">37</xref>]. We used adaptive tree imputation, with multiple iterations of the imputation algorithm due to the large fraction of missing data in many variables used [<xref ref-type="bibr" rid="pone.0202344.ref019">19</xref>].</p>
<p>The number of iterations of the imputation process, number of trees, number of split points (<italic>n</italic><sub>split</sub>), and number of variables tested at each split point (<italic>m</italic><sub>try</sub>) were selected with cross-validation.</p>
</sec>
<sec id="sec016">
<title>Variable selection</title>
<p>Performance can be improved by variable selection to isolate the most relevant predictors before fitting the final model. We evaluated various methods for variable selection: ranking by completeness (i.e. retaining those variables with the fewest missing values); by permutation variable importance (which measures a variable’s contribution to the final model performance by randomly shuffling values for that variable and observing the decrease in C-index), similar to varSelRF [<xref ref-type="bibr" rid="pone.0202344.ref038">38</xref>, <xref ref-type="bibr" rid="pone.0202344.ref039">39</xref>]; and by effect on survival, as measured by a logrank test between survival curves of different variable values (either true vs false, categories, or all quartiles for continuous variables). All of these are inherently univariate, and may neglect variables which have a large effect when in combination with others [<xref ref-type="bibr" rid="pone.0202344.ref015">15</xref>, <xref ref-type="bibr" rid="pone.0202344.ref040">40</xref>].</p>
<p>After ranking, we fitted models to decreasing numbers of the highest-ranked variables and cross-validated to find the optimal number based on the C-index.</p>
<p>After selection, the contribution of variables to the final model was assessed by taking the permutation variable importance. There are a number of measures of variable importance in common use [<xref ref-type="bibr" rid="pone.0202344.ref041">41</xref>, <xref ref-type="bibr" rid="pone.0202344.ref042">42</xref>], often using metrics which are specific to random forests, such as the number of times a variable is split upon, or the pureness of nodes following such a split. Permutation variable importance is more readily generalised: it is given by the reduction in performance when values of a particular variable are shuffled randomly within the dataset, thus removing its power to aid predictions—this allows it to be used to assess the importance of variables in other types of model, such as Cox models. The impact on model predictivity is assessed and, the larger the impact, the more important the variable is considered to be. Permutation variable importance was used in this study in order to allow comparisons across different types of model.</p>
</sec>
<sec id="sec017">
<title>Elastic net regression</title>
<p>Elastic net regression extends regression techniques including Cox models to penalise complex models and thus implicitly select variables during the fitting procedure [<xref ref-type="bibr" rid="pone.0202344.ref040">40</xref>, <xref ref-type="bibr" rid="pone.0202344.ref043">43</xref>]. It works by minimising the regularised error function
<disp-formula id="pone.0202344.e004"><alternatives><graphic id="pone.0202344.e004g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0202344.e004" xlink:type="simple"/><mml:math display="block" id="M4"><mml:mo>‖</mml:mo><mml:mi>y</mml:mi><mml:mo>−</mml:mo><mml:mi>X</mml:mi><mml:mi>β</mml:mi><mml:msup><mml:mo>‖</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mi>α</mml:mi><mml:mi>λ</mml:mi><mml:mo>‖</mml:mo><mml:mi>β</mml:mi><mml:mo>‖</mml:mo><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>α</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:mfrac><mml:mi>λ</mml:mi><mml:mo>‖</mml:mo><mml:mi>β</mml:mi><mml:msup><mml:mo>‖</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>,</mml:mo></mml:math></alternatives> <label>(2)</label></disp-formula>
where <italic>y</italic> is the response variable, <italic>X</italic> are the input data, <italic>β</italic> is a vector of coefficients, 0 ≤ <italic>α</italic> ≤ 1 is a hyperparameter for adjusting the weight of the two different regularisation components, and λ is a hyperparameter adjusting the overall magnitude of regularisation. The hyperparameter <italic>α</italic> linearly interpolates between a LASSO (least absolute shrinkage and selection operator) and ridge regression: a value <italic>α</italic> = 0 corresponds to pure LASSO, whilst <italic>α</italic> = 1 corresponds to pure ridge regression, with intermediate values representing a mixture of the two.</p>
<p>The value of λ is determined to some extent by the scale of the data, and so normalisation between parameters is important before fitting. In our case, values were either binary true/false when referring to the presence or absence of particular codes in a patient’s medical history, or represented discretised continuous values with an additional ‘missing’ category, binarised into a one-hot vector for fitting. Thus, since all values are either 0 or 1, no normalisation was undertaken before performing the elastic net procedure.</p>
<p>Hyperparameters <italic>α</italic> and λ were chosen by grid search with ten-fold cross-validation over the training set. Having identified the optimal <italic>α</italic>, this value was used to fit models on bootstrapped samples of the training set to extract model parameters along with an estimate of their variation due to sampling.</p>
</sec>
<sec id="sec018">
<title>Model performance</title>
<p>Model performance was assessed using both the C-index, and a calibration score derived from accuracy of five-year mortality risk predictions.</p>
<p>The C-index measures a model’s discrimination, its ability to discriminate low-risk cases from high-risk ones, by evaluating the probability of the model correctly predicting which of two randomly selected patients will die first. It ranges from 0.5 (chance) to 1 (perfect prediction). The C-index is widely used, but its sensitivity for distinguishing between models of different performance can be low because it is a rank-based measure [<xref ref-type="bibr" rid="pone.0202344.ref044">44</xref>].</p>
<p>In particular, the C-index is unable to assess the calibration of a model, which reflects the accuracy of absolute risk predictions made for individual patients. A well-calibrated model is particularly important in the context of a clinical risk score, where a treatment may be assigned based on whether a patient exceeds a given risk threshold [<xref ref-type="bibr" rid="pone.0202344.ref045">45</xref>]. This was determined by plotting patient outcome (which is binary: dead or alive, with patients censored before the date of interest excluded) against mortality risk predicted by the model [<xref ref-type="bibr" rid="pone.0202344.ref044">44</xref>]. Then, a locally smoothed regression line is calculated, and the area, <italic>a</italic>, between the regression line and the ideal line <italic>y</italic> = <italic>x</italic> provides a measure of model calibration. Since a smaller area is better, we define the calibration score as (1 − <italic>a</italic>) such that a better performance results in a higher score. This means that, like the C-index, it ranges from 0.5 (very poor) to 1 (perfect calibration).</p>
</sec>
<sec id="sec019">
<title>Variable effects</title>
<p>The partial dependency of Cox models on each variable is given by <italic>β</italic><sub><italic>i</italic></sub><italic>x</italic><sub><italic>i</italic></sub>, where <italic>β</italic><sub><italic>i</italic></sub> is the coefficient and <italic>x</italic><sub><italic>i</italic></sub> the standardised value of a variable <italic>i</italic>. The associated 95% confidence interval is given by <italic>β</italic><sub><italic>i</italic></sub><italic>x</italic><sub><italic>i</italic></sub> ± Δ<italic>β</italic><sub><italic>i</italic></sub><italic>x</italic><sub><italic>i</italic></sub> where Δ<italic>β</italic><sub><italic>i</italic></sub> is the uncertainty on that coefficient. Both the risk and the confidence interval are therefore zero at the baseline value where <italic>x</italic><sub><italic>i</italic></sub> = 0. In discrete Cox models, coefficients for a given bin <italic>j</italic> translate directly into an associated <italic>β</italic><sub><italic>ij</italic></sub> value, with an 95% confidence interval <italic>β</italic><sub><italic>ij</italic></sub> ± Δ<italic>β</italic><sub><italic>ij</italic></sub>. The lowest bin is defined as the baseline, and consequently has no associated uncertainty.</p>
<p>Due to their complex structure, interrogation of variable effects in random forests is more involved than for Cox models. They were assessed with partial dependence plots [<xref ref-type="bibr" rid="pone.0202344.ref046">46</xref>] for each variable. These are calculated by drawing many patients at random, and then predicting survival for each of them using the random forest many times, holding all variables constant for each patient except the variable of interest whose value is swept through its possible values for each prediction. This gives a curve of risks which we then normalised by its average, and then averaged over the set of patients to give a response to that variable. Relative changes of this risk can therefore be compared between models, but absolute values are not directly comparable.</p>
</sec>
</sec>
</sec>
<sec id="sec020" sec-type="results">
<title>Results</title>
<sec id="sec021">
<title>Patient population</title>
<p>A summary of the patient population used in this study is shown in <xref ref-type="table" rid="pone.0202344.t002">Table 2</xref>. We initially identified 115,305 patients with coronary disease in CALIBER and, after excluding patients based on criteria relating the timing of the diagnosis and follow-up, 82,197 patients remained in the cohort. Imputation models included all 115,305 patients in order to increase precision, with the exclusion flag as an auxiliary variable in imputation models.</p>
<table-wrap id="pone.0202344.t002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.t002</object-id>
<label>Table 2</label>
<caption>
<title>Cohort summary, including all expert-selected predictors.</title>
</caption>
<alternatives>
<graphic id="pone.0202344.t002g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.t002" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left" style="border-bottom:thick">Characteristic</th>
<th align="left" style="border-bottom:thick">Summary</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Patient population</td>
<td align="left">82,197</td>
</tr>
<tr>
<td align="left">Age</td>
<td align="left">69 (43–90) years</td>
</tr>
<tr>
<td align="left">Gender</td>
<td align="left">35,192 (43%) women, 47,005 (57%) men</td>
</tr>
<tr>
<td align="left">Social deprivation (IMD score)</td>
<td align="left">16 (4.0–52), 0.24% missing</td>
</tr>
<tr>
<td align="left">Diagnosis</td>
<td align="left">57% stable angina<break/>12% unstable angina<break/>1.8% coronary heart disease<break/>2.8% ST-elevated MI<break/>3.1% non-ST-elevated MI<break/>23% MI not otherwise specified</td>
</tr>
<tr>
<td align="left">PCI in last 6 months</td>
<td align="left">4.6%</td>
</tr>
<tr>
<td align="left">CABG in last 6 months</td>
<td align="left">2.0%</td>
</tr>
<tr>
<td align="left">Previous/recurrent MI</td>
<td align="left">29%</td>
</tr>
<tr>
<td align="left">Use of nitrates</td>
<td align="left">27%</td>
</tr>
<tr>
<td align="left">Smoking status</td>
<td align="left">18% current<break/>33% ex-smoker<break/>40% non-smoker<break/>8.2% missing</td>
</tr>
<tr>
<td align="left">Hypertension</td>
<td align="left">88%</td>
</tr>
<tr>
<td align="left">Diabetes mellitus</td>
<td align="left">15%</td>
</tr>
<tr>
<td align="left">Total cholesterol</td>
<td align="left">4.8 (2.8–7.7) mmol l<sup>−1</sup>, 64% missing</td>
</tr>
<tr>
<td align="left">High density lipoprotein cholesterol</td>
<td align="left">1.3 (0.70–2.3) mmol l<sup>−1</sup>, 78% missing</td>
</tr>
<tr>
<td align="left">Heart failure</td>
<td align="left">10%</td>
</tr>
<tr>
<td align="left">Peripheral arterial disease</td>
<td align="left">7.0%</td>
</tr>
<tr>
<td align="left">Atrial fibrillation</td>
<td align="left">12%</td>
</tr>
<tr>
<td align="left">Stroke</td>
<td align="left">5.4%</td>
</tr>
<tr>
<td align="left">Chronic kidney disease</td>
<td align="left">6.7%</td>
</tr>
<tr>
<td align="left">Chronic obstructive pulmonary disease</td>
<td align="left">35%</td>
</tr>
<tr>
<td align="left">Cancer</td>
<td align="left">8.0%</td>
</tr>
<tr>
<td align="left">Chronic liver disease</td>
<td align="left">0.41%</td>
</tr>
<tr>
<td align="left">Depression at diagnosis</td>
<td align="left">19%</td>
</tr>
<tr>
<td align="left">Anxiety at diagnosis</td>
<td align="left">12%</td>
</tr>
<tr>
<td align="left">Heart rate</td>
<td align="left">72 (48–108) bpm, 92% missing</td>
</tr>
<tr>
<td align="left">Creatinine</td>
<td align="left">94 (60–190) μmol l<sup>−1</sup>, 64% missing</td>
</tr>
<tr>
<td align="left">White cell count</td>
<td align="left">7.1 (4.5–11) ×10<sup>−9</sup>, 75% missing</td>
</tr>
<tr>
<td align="left">Haemoglobin</td>
<td align="left">14 (9.9–17) g dl<sup>−1</sup>, 77% missing</td>
</tr>
<tr>
<td align="left">Status at endpoint</td>
<td align="left">18,930 (23%) dead, 63,267 (77%) censored</td>
</tr>
</tbody>
</table>
</alternatives>
<table-wrap-foot>
<fn id="t002fn001">
<p>Values are quoted to two significant figures and may not sum due to rounding. Continuous values are summarised as median (95% confidence interval). IMD = index of multiple deprivation; PCI = percutaneous coronary intervention; CABG = coronary artery bypass graft; MI = myocardial infarction.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec022">
<title>Age-based baseline</title>
<p>A baseline model with age as the only predictor attained a C-index of 0.74. To calculate the risk and assess calibration, the Kaplan–Meier estimator for patients of age <italic>x</italic> years in (a bootstrap sample of) the training set is used, resulting in a well-calibrated model with a calibration score of 0.935 (95% confidence interval: 0.924 to 0.944).</p>
</sec>
<sec id="sec023">
<title>Calibration and discrimination of models</title>
<p>A summary of the performance results obtained for modelling time to all-cause mortality using various statistical methods and data is shown in <xref ref-type="fig" rid="pone.0202344.g001">Fig 1</xref>.</p>
<fig id="pone.0202344.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Overall discrimination and calibration performance for the different models and datasets used.</title>
<p>(A) shows discrimination (C-index) and (B) shows calibration (1 − <italic>a</italic>, where <italic>a</italic> is the area between the observed calibration curve and an idealised one for patient risk at five years). Bar height indicates median performance across bootstrap replicates, with error bars representing 95% confidence intervals. Columns 1–4 represent variations on the Cox proportional hazards models using the 27 expert-selected variables used in Ref. [<xref ref-type="bibr" rid="pone.0202344.ref013">13</xref>]. Column 1 shows a model with missing values included with dummy variables; column 2 shows a model where continuous values have been discretised and missing values included as an additional category; column 3 shows a model where missing values have been imputed; and column 4 shows a model where missing values have been imputed and then all values discretised with the same scheme as column 2. Columns 5–7 show the performance of random survival forests. Column 5 uses the 27 expert-selected variables with missing values included with missingness indicators, and column 6 uses the imputed dataset. Columns 7 and 8 show models based on a subset of the 600 least missing variables across a number of different EHR data sources, selected by cross-validation. 7 is a random forest model with missing values left as-is, while 8 is a Cox proportional hazards model with continuous values discretised. Column 9 is an elastic net regression based on all 600 variables.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.g001" xlink:type="simple"/>
</fig>
<p>Comparing between models, the discrimination performance of random forests is similar to that of the best Cox models, but they performed worse in calibration, underperforming the Cox model with imputed data by 0.098 (0.105 to 0.091, 95% confidence interval on distribution of differences between bootstrap replicates). A calibration curve for this model is shown in <xref ref-type="fig" rid="pone.0202344.g002">Fig 2(A)</xref>. In view of this unexpectedly poor calibration performance, we also attempted to fit five-year survival with a classification forest. The binary outcome of being dead or alive at five years is essentially a classification problem, and it was possible that metrics for node purity during classification may be more reliable than those used in survival modelling. However, this proved to be both worse calibrated and worse discriminating than survival-based models, perhaps due to loss of the large number of censored patients in the dataset. A random survival forest implicitly takes censored patients into account by splitting based on logrank scores of survival curves, whereas a simple classification forest must discard them, perhaps losing valuable information.</p>
<fig id="pone.0202344.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Example calibration curves for mortality at five years.</title>
<p>Curves show calibration of (A) model 5 (poorly-calibrated) and (B) model 9 (well-calibrated) from <xref ref-type="fig" rid="pone.0202344.g001">Fig 1</xref>. Each semi-transparent dot represents a patient in a random sample of 2000 from the test set, the <italic>x</italic>-axis shows their risk of death at five years as predicted by the relevant model, and the <italic>y</italic>-axis shows whether each patient was in fact dead or alive at that time. The curves are smoothed calibration curves derived from these points, and the area between the calibration curve and the black line of perfect calibration is <italic>a</italic> in the calibration score, 1 − <italic>a</italic>.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.g002" xlink:type="simple"/>
</fig>
<p>Imputing missing values has some effect on model performance. Comparing the two continuous Cox models, column 1’s model includes missingness indicators and column 3’s model imputes missing values. Discrimination is very slightly improved in the imputed model, but calibration performance remains the same. Discretising continuous values without imputing has a similar effect, improving C-index slightly with little effect on calibration score.</p>
<p>To assess the impact of imputation on performance model-agnostically, we performed an empirical test by fitting a random forest model to the imputed dataset. The C-index remained similar, but calibration improves dramatically, almost achieving the same performance as the Cox model. This is suggestive but not decisive evidence that some information may leak between training and test sets, or from future observations, during imputation.</p>
<p>On the extended dataset, random forests fitted with all variables performed badly, even when cross-validated to optimise <italic>m</italic><sub>try</sub>. After trying the different techniques discussed in Methods to rank important variables, by far the best random forest performance was achieved when variables were ranked by within-variable logrank test. This gave rise to a model using 98 of the variables, with a C-index of 0.797 (0.796 to 0.798). Unfortunately, even on the extended dataset random forests are similarly poorly calibrated to those fitted to the expert-selected dataset, under-performing the Cox model with imputed data by 0.086 (0.078 to 0.093).</p>
<p>A Cox model was also fitted to the extended dataset, with all continuous variables discretised into deciles. Variables were again ranked by within-variable logrank tests, and cross-validated to give a model comprising 155 variables. Its C-index performance was comparable to other models, and also has the highest median calibration performance of 0.966 (0.961 to 0.970).</p>
<p>Finally, an elastic net model was fitted to the discretised extended dataset. The optimal value of the hyperparameter <italic>α</italic> = 0.90 was determined by grid-search, and fixed for the remainder of the analysis. The hyperparameter λ was selected by ten-fold cross-validation on the full training dataset, returning nonzero coefficients for 270 variables. Bootstrapped replicates were again used to assess model performance, giving a calibration score of 0.075 (0.0679 to 0.0812), and the highest discrimination score of all models with a C-index of 0.801 (0.799 to 0.802). A calibration plot for this model is shown in <xref ref-type="fig" rid="pone.0202344.g002">Fig 2(B)</xref>.</p>
</sec>
<sec id="sec024">
<title>Effect of different methods for missing data</title>
<p>
<xref ref-type="fig" rid="pone.0202344.g003">Fig 3(A)</xref> shows coefficients associated with variables compared between a model based on imputation, and models based on accounting for missing values with missingness indicators, and discretisation. Most of the risks have similar values between models, showing that different methods of dealing with missing data preserve the relationships between variables and outcome.</p>
<fig id="pone.0202344.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Comparison of coefficients for present and missing data in continuous and discrete Cox models.</title>
<p>(A) Fitted coefficients for different Cox models compared. Values of risks from both the continuous model with missingness indicators, and the discretised model are plotted against the continuous imputed Cox model. There are fewer points for the discretised model as coefficients for continuous values are not directly comparable. The very large error bars on four of the points correspond to the risk for diagnoses of STEMI and NSTEMI. This is due to the majority (73%) of heart attack diagnoses being ‘MI (not otherwise specified)’ from which the more specific diagnoses were imputed, introducing significant uncertainty. (B) For the continuous Cox model with missingness indicators, risk ranges for the ranges of values for variables present in the dataset (violin plots) with risk associated with that value being missing (points with error bars). CRN = creatinine, HGB = haemoglobin, HDL = high-density lipoprotein, WBC = white blood cell count, TC = total cholesterol; for smoking status, miss = missing, ex = ex-smoker and curr = current smoker, with non-smokers as the baseline. (C) Survival curves for selected variables, comparing patients with a value recorded for that variable versus patients with a missing value. These can be compared with risks associated with a missing value, seen in (B): HDL and TC show increased risk where values are missing, whilst CRN shows the opposite, which is reflected in the survival curves.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.g003" xlink:type="simple"/>
</fig>
<p>Explicitly coding missing data as a missingness indicator allows us to associate a risk with that variable having no value in a patient’s record. These are examined in <xref ref-type="fig" rid="pone.0202344.g003">Fig 3(B)</xref>, compared to the range of risks implied by the range of values for a given variable. Having a missing value is associated with very different risk depending on which variable is being observed, indicating that missingness carries information about a patient’s prognosis. Where risks fall outside the range of risks associated with values found in the dataset, we validated these associations by plotting survival curves comparing patients with and without a particular value missing. These curves, shown in <xref ref-type="fig" rid="pone.0202344.g003">Fig 3(C)</xref>, agree with the coefficients shown in <xref ref-type="fig" rid="pone.0202344.g003">Fig 3(B)</xref>: variables whose missingness carries a higher risk than the normal range of values show a worse survival curve for patients with that value missing, and vice-versa.</p>
</sec>
<sec id="sec025">
<title>Variable and missing value effects</title>
<p>Next, we examined variable effects between models. These are shown in <xref ref-type="fig" rid="pone.0202344.g004">Fig 4</xref>, with row (A) showing the continuous and discrete Cox models on unimputed data, and row (B) showing partial effects plots derived from the random survival forest models.</p>
<fig id="pone.0202344.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.g004</object-id>
<label>Fig 4</label>
<caption>
<title>Variable effect plots from Cox models and partial dependence plots for random survival forests.</title>
<p>(A) Comparisons between relative log-risks derived from the continuous Cox models (straight blue lines, light blue 95% CI region) against those derived from the discretised models (green steps, light green 95% CI region), together with log-risks associated with those values being missing (points with error bars on the right of plots). Confidence intervals on the continuous models represent Δ<italic>β</italic><sub><italic>i</italic></sub><italic>x</italic><sub><italic>i</italic></sub> for each variable <italic>x</italic><sub><italic>i</italic></sub>, and hence increase from the value taken as the baseline where <italic>x</italic><sub><italic>i</italic></sub> = 0. The lowest-valued bin of the discrete Cox model is taken to be the baseline and has no associated uncertainty. Discrete model lines are shifted on the <italic>y</italic>-axis to align their baseline values with the corresponding value in the continuous model to aid comparison; since these are relative risks, vertical alignment is arbitrary. (B) Partial dependence plots [<xref ref-type="bibr" rid="pone.0202344.ref046">46</xref>] inferred from random forests. Semitransparent black lines show the change in log-risk of death after five years from sweeping across possible values of the variable of interest whilst holding other variables constant. Results are normalised to average mortality across all values of this variable. Thick orange lines show the median of the 1000 replicates, indicating the average response in log-risk to changing this variable.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.g004" xlink:type="simple"/>
</fig>
<p>The partial effects of these variables show varying levels of agreement depending on model used. Risk of death rises exponentially with age, with the discrete Cox model’s response having a very similar gradient to the continuous model. Haemoglobin and total white blood cell count are in close agreement between these models too, with only slight deviations from linearity in the discrete case. Creatinine shows a pronounced increase in mortality at both low and high levels in the discrete model, which the linear model is unable to capture. The risks associated with having a missing value show broad agreement between the Cox models.</p>
<p>Random forest variable effect plots differ somewhat from the equivalent Cox model coefficients. The sign of the overall relationship is always the same, but all display quite pronounced deviations from linearity, especially at extreme values.</p>
</sec>
<sec id="sec026">
<title>Variable selection and importance in data-driven models</title>
<p>
<xref ref-type="fig" rid="pone.0202344.g005">Fig 5</xref> shows the permutation variable importances for the 20 most important variables in the random forest and discrete Cox models fitted to the large dataset, after variable selection. All three models identify several variables which are also present in the expert-selected dataset, including the two most important (age and smoking status). There is also significant agreement within the 20 selected variables between models, with the top three appearing in the same order in the random forest and Cox model, and all appearing in the top five for the elastic net.</p>
<fig id="pone.0202344.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0202344.g005</object-id>
<label>Fig 5</label>
<caption>
<title>Top 20 variables by permutation importance for the three data-driven models, using random survival forests, discrete Cox modelling and elastic net regression.</title>
<p>Variables which are either identical or very similar to those found in the expert-selected dataset are highlighted with a blue dot; variables appearing in two of the models are joined by a pink line (with reduced opacity for those which pass behind the middle graph); whilst variables appearing in all three are joined by green lines. Some variable names have been abbreviated for space: ACE inhibitors = angiotensin-converting enzyme inhibitors; ALP = alkaline phosphatase; analgesics = non-opioid and compound analgesics; ALT = alanine aminotransferase; Beta2 agonists = selective beta2 agonists; blood pressure = diastolic blood pressure; BMI = body mass index; CKD = chronic kidney disease; Insulin = intermediate- and long-acting insulins; LV failure = left ventricular failure; Hb = haemoglobin; MCV = mean corpuscular volume; Na = sodium; PVD = peripheral vascular disease; Records held date = date records held from; WCC = total white blood cell count.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0202344.g005" xlink:type="simple"/>
</fig>
<p>When performing cross-validation to select variables for the Cox and random forest models, we observed large performance differences between ranking methods. Random forest variable importance on the full dataset performed poorly; it was a noisy measure of a variable’s usefulness, with significantly different variable rankings were returned on successive runs of the algorithm. Ranking variables by missingness performed better, but still significantly below the models using expert-selected variables. By far the best performance was achieved when variables were ranked by within-variable logrank test, which was used for both the random forest and discrete Cox models fitted here.</p>
</sec>
</sec>
<sec id="sec027" sec-type="conclusions">
<title>Discussion</title>
<p>We compared Cox modelling and data-driven approaches for all-cause mortality prediction in a cohort of over 80,000 patients in an electronic health record dataset. All models tested had good performance. This suggests that, quantified by discrimination and calibration performance, there is no need to impute EHR data before fitting predictive models, nor to select variables manually. It also invites the use of data-driven models in other contexts where important predictor variables are not known.</p>
<sec id="sec028">
<title>Advantages of data-driven approaches</title>
<p>Data-driven modelling has both practical and theoretical advantages over conventional survival modelling [<xref ref-type="bibr" rid="pone.0202344.ref005">5</xref>, <xref ref-type="bibr" rid="pone.0202344.ref024">24</xref>]. It may require less user input, through automated variable selection. Further, both our method of converting continuous to categorical data (by quantiles), and random forest modelling are robust to arbitrary monotonic transformations of a variable, removing the need to manually normalise or transform data before analysis.</p>
<p>The ability to analyse data without the need for prior imputation also saves researcher time, and allows the effect of missingness in particular variables to be examined. It also has the potential to improve model performance in settings where missingness carries meaning. It is also possible for new patients to be assessed without the need for a clinician to impute, or take additional measurements, in order to obtain a risk score.</p>
<p>Discretisation of continuous values can accommodate nonlinear and nonmonotonic responses to variables, without requiring time-consuming choice and tuning of these relationships [<xref ref-type="bibr" rid="pone.0202344.ref047">47</xref>].</p>
<p>Machine learning techniques can also make use of more data, which in other datasets or settings may give rise to further improved performance over conventional models. Variable selection allows its use in contexts where risk factors are unknown.</p>
</sec>
<sec id="sec029">
<title>Comparison of modelling methods</title>
<p>We found that random forests did not outperform Cox models despite their inherent ability to accommodate nonlinearities and interactions [<xref ref-type="bibr" rid="pone.0202344.ref048">48</xref>, <xref ref-type="bibr" rid="pone.0202344.ref049">49</xref>]. Random forests have a number of shortcomings which may explain this. First, only a random subset of variables (<italic>m</italic><sub>try</sub>) are tried at each split, so datasets that contain a large proportion of uninformative ‘noise’ variables may cause informative variables to be overlooked by chance at many splits. Increasing <italic>m</italic><sub>try</sub> can improve performance, but often at a large cost in computation time. Second, when random forests are used for prediction, the predictions are a weighted average of a subset of the data, and are biased away from the extremes [<xref ref-type="bibr" rid="pone.0202344.ref050">50</xref>]. This may partly explain their poor calibration.</p>
<p>We did not find that discretisation of continuous variables improved model performance, probably because the majority of these variables had associations with prognosis that were close to linear, and the small improvement in fit was offset by the large increase in the number of model parameters.</p>
<p>Elastic nets achieved the highest discrimination performance of any model tested, demonstrating the ability of regularisation to select relevant variables and optimise model coefficients in an EHR context.</p>
</sec>
<sec id="sec030">
<title>Missing data</title>
<p>We also found that missing values may be associated with greater or lesser risk than any of the measured values depending on the potential reasons for a value’s presence or absence. For example, serum creatinine needs to be monitored in people on certain medication, and is less likely to be measured in healthy people. Conversely, missing high-density lipoprotein (HDL) or total cholesterol (TC), used for cardiovascular risk prediction in preventive care, was associated with worse outcomes.</p>
<p>There was little difference in model performance with or without imputation of missing values. It is possible that this did not have a large effect in the expert-selected dataset because only six variables carried missing values and, of these, only three had risks significantly different from the range of measured values. The majority of the variables were Boolean (e.g. presence of diagnoses) and were assumed to be completely recorded, where absence of a record was interpreted as absence of the condition.</p>
<p>One shortcoming of the use of imputation in our analysis is that, to our knowledge, no software is able to build an imputation model on a training set and then apply that model to new data. As a consequence, we performed imputation across the whole dataset, but this violates the principle of keeping training and test data entirely separate to prevent information ‘leaking’ between them. Further, because future values are used in the imputation process, this adds additional potential for introducing bias to models using such data. We investigated this empirically by fitting a random forest model to the imputed dataset (see <xref ref-type="sec" rid="sec020">Results</xref>) and found some evidence that bias may be introduced. If this is the case, the benefits of imputation to model performance may be less than suggested here.</p>
</sec>
<sec id="sec031">
<title>Variable selection</title>
<p>Finally, we found that data-driven modelling with large numbers of variables is sensitive to the modelling and variable selection techniques used. Random forests without variable selection performed poorly, which is evident from both the poor performance of fitted models, and the lack of utility of random forest variable importance as a measure by which to rank variables during selection.</p>
<p>Variable selection using the least missing data performed better, but not as well as expert selection. This may be because many of the variables with low missingness were uninformative. Variable selection using univariate logrank tests was far more successful, allowing models with slightly higher performance than expert selection, and discrete Cox models based on this displayed the best calibration performance.</p>
<p>Elastic net regression offered the best C-index performance, by fitting a Cox model with implicit variable selection. Since this selection process operates simultaneously across coefficients for all variables, it is possible that this explains its improved performance over ranking by univariate measures applied to random forests.</p>
<p>High-ranking variables in our final model (<xref ref-type="fig" rid="pone.0202344.g005">Fig 5</xref>) seem plausible. Well-known prognostic factors such as age and smoking status were strongly associated with mortality, as expected. Many of the other variables identified, such as prescriptions for cardiovascular medications, are proxy indicators of the presence and severity of cardiovascular problems. Finally, some variables are proxies for generalised frailty, such as laxative prescriptions, home visits etc. These may be unlikely to considered for a prognostic model constructed by experts as they are not obviously related to cardiovascular mortality, and this demonstrates a potential benefit of data-driven modelling in EHR to both identify these novel variables, and improve model performance by incorporating them.</p>
<p>The important caveat is that these correlative relationships are not necessarily causal, and may not generalise beyond the study population or EHR system in which they were originally derived [<xref ref-type="bibr" rid="pone.0202344.ref001">1</xref>]. For example, our finding that home visits are a highly ranked variable is likely to be indicative of frailty, but its contribution will change depending on the criteria for such visits and the way they are recorded, which may vary between healthcare systems or over time.</p>
</sec>
<sec id="sec032">
<title>Disadvantages of data-driven approaches</title>
<p>Data-driven and machine learning based techniques come with several disadvantages. Firstly, the degree of automation in these methods is not yet complete; random forests worked showed poor performance on the full 586-variable dataset without variable selection. Use with other datasets or endpoints currently requires some researcher effort to identify optimal algorithms and variable selection methods. It would be useful to develop tools which would automate this, and test them across a multitude of different EHR systems and endpoints.</p>
<p>Secondly, it can be difficult to interpret the final models. Whilst it is possible to use variable effect plots to understand the relationship between a few variables and an outcome, this is prohibitively complex with high-dimensional models. In addition, not imputing missing values makes it difficult to interpret coefficients for partially observed variables, as they include a component of the reason for missingness. When prognostic models are used in clinical practice, it is important to be able to trust that the data are incorporated appropriately into the model so that clinicians can justify decisions based on the models.</p>
<p>Conventional statistical modelling techniques retain advantages and disadvantages which are the converse of these: models are more readily interpretable, and may generalise better, but at the expense of requiring significant expert input to construct, potentially not making use of the richness of available data, and only being applicable to complete data.</p>
</sec>
<sec id="sec033">
<title>Limitations of this study</title>
<p>These methods were compared in a single study dataset, and replication in other settings would be needed to demonstrate the generalisability of the findings.</p>
<p>The simple approach of using the top 100 least-missing variables from each table for development of the data-driven models is unlikely to be the optimal approach. In particular, some highly present variables are administrative and have little prognostic value, while significant but rare variables could be overlooked.</p>
<p>There are many types of model we did not consider in this study, including support vector machines and neural networks, which may provide improved prognostic performance. However, machine learning libraries for survival data are not well-developed, limiting the model types which can be tested without significant time devoted to software and model development.</p>
</sec>
<sec id="sec034">
<title>Conclusion</title>
<p>We have demonstrated that machine learning approaches on routine EHR data can achieve comparable or better performance than expert-selected, imputed data in manually optimised models for risk prediction. Our comparison of a range of machine learning algorithms found that elastic net regression performed best, with cross-validated variable selection based on logrank tests enabling Cox models and random forests to achieve comparable performance.</p>
<p>Data-driven methods have the potential to simplify model construction for researchers and allow novel epidemiologically relevant predictors to be identified, but achieving good performance with machine learning requires careful testing of the methods used. These approaches are also disease-agnostic, which invites their further use in conditions with less well-understood aetiology.</p>
<p>Eventually, machine learning approaches combined with EHR may make it feasible to produce fine-tuned, individualised prognostic models, which will be particularly valuable in patients with conditions or combinations of conditions which would be very difficult for conventional modelling approaches to capture.</p>
</sec>
</sec>
<sec id="sec035">
<title>Code and data</title>
<p>The scripts used for analysis in this paper are available at <ext-link ext-link-type="uri" xlink:href="https://github.com/luslab/MLehealth" xlink:type="simple">https://github.com/luslab/MLehealth</ext-link>.</p>
<p>While our data does not contain any sensitive personal identifiers, it is deemed as sensitive as it contains sufficient clinical information about patients such as dates of clinical events for there to be a potential risk of patient re-identification. This restriction has been imposed by the data owner (CPRD/MHRA) the data sharing agreements between UCL and the CPRD/MHRA. Access to data may be requested via the Clinical Practice Research Datalink (CPRD) and applying to the CPRD’s Independent Scientific Advisory Committee (<ext-link ext-link-type="uri" xlink:href="https://www.cprd.com/researcher/" xlink:type="simple">https://www.cprd.com/researcher/</ext-link>).</p>
</sec>
</body>
<back>
<ack>
<p>The authors thank Sera Aylin Cakiroglu for performing the imputation with MICE, and for her help with the data curation and preparation and the design of the study. This work was supported by the Francis Crick Institute which receives its core funding from Cancer Research UK (FC001110), the UK Medical Research Council (FC001110), and the Wellcome Trust (FC001110). NML and HH were supported by the Medical Research Council Medical Bioinformatics Award eMedLab (MR/L016311/1). The CALIBER programme was supported by the National Institute for Health Research (RP-PG-0407-10314, PI HH); Wellcome Trust (WT 086091/Z/08/Z, PI HH); the Medical Research Prognosis Research Strategy Partnership (G0902393/99558, PI HH) and the Farr Institute of Health Informatics Research, funded by the Medical Research Council (K006584/1, PI HH), in partnership with Arthritis Research UK, the British Heart Foundation, Cancer Research UK, the Economic and Social Research Council, the Engineering and Physical Sciences Research Council, the National Institute of Health Research, the National Institute for Social Care and Health Research (Welsh Assembly Government), the Chief Scientist Office (Scottish Government Health Directorates) and the Wellcome Trust.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0202344.ref001">
<label>1</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Goldstein</surname> <given-names>BA</given-names></name>, <name name-style="western"><surname>Navar</surname> <given-names>AM</given-names></name>, <name name-style="western"><surname>Pencina</surname> <given-names>MJ</given-names></name>, <name name-style="western"><surname>Ioannidis</surname> <given-names>JPA</given-names></name>. <article-title>Opportunities and challenges in developing risk prediction models with electronic health records data: a systematic review</article-title>. <source>J Am Med Inform Assoc</source>. <year>2017</year>;<volume>24</volume>(<issue>1</issue>):<fpage>198</fpage>–<lpage>208</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/jamia/ocw042" xlink:type="simple">10.1093/jamia/ocw042</ext-link></comment> <object-id pub-id-type="pmid">27189013</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref002">
<label>2</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Riley</surname> <given-names>RD</given-names></name>, <name name-style="western"><surname>Ensor</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Snell</surname> <given-names>KIE</given-names></name>, <name name-style="western"><surname>Debray</surname> <given-names>TPA</given-names></name>, <name name-style="western"><surname>Altman</surname> <given-names>DG</given-names></name>, <name name-style="western"><surname>Moons</surname> <given-names>KGM</given-names></name>, <etal>et al</etal>. <article-title>External validation of clinical prediction models using big datasets from e-health records or IPD meta-analysis: opportunities and challenges</article-title>. <source>BMJ</source>. <year>2016</year>;<volume>353</volume>:<fpage>i3140</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/bmj.i3140" xlink:type="simple">10.1136/bmj.i3140</ext-link></comment> <object-id pub-id-type="pmid">27334381</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref003">
<label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Denaxas</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Kunz</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Smeeth</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Gonzalez-Izquierdo</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Boutselakis</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Pikoula</surname> <given-names>M</given-names></name>, <etal>et al</etal>. <article-title>Methods for enhancing the reproducibility of clinical epidemiology research in linked electronic health records: results and lessons learned from the CALIBER platform</article-title>. <source>IJPDS</source>. <year>2017</year>;<volume>1</volume>(<issue>1</issue>). <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.23889/ijpds.v1i1.84" xlink:type="simple">10.23889/ijpds.v1i1.84</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref004">
<label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Casey</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Schwartz</surname> <given-names>BS</given-names></name>, <name name-style="western"><surname>Stewart</surname> <given-names>WF</given-names></name>, <name name-style="western"><surname>Adler</surname> <given-names>NE</given-names></name>. <article-title>Using Electronic Health Records for Population Health Research: A Review of Methods and Applications</article-title>. <source>Annu Rev Public Health</source>. <year>2016</year>;<volume>37</volume>:<fpage>61</fpage>–<lpage>81</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1146/annurev-publhealth-032315-021353" xlink:type="simple">10.1146/annurev-publhealth-032315-021353</ext-link></comment> <object-id pub-id-type="pmid">26667605</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref005">
<label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Denaxas</surname> <given-names>SC</given-names></name>, <name name-style="western"><surname>Morley</surname> <given-names>KI</given-names></name>. <article-title>Big biomedical data and cardiovascular disease research: opportunities and challenges</article-title>. <source>Eur Heart J Qual Care Clin Outcomes</source>. <year>2015</year>;<volume>1</volume>(<issue>1</issue>):<fpage>9</fpage>–<lpage>16</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/ehjqcco/qcv005" xlink:type="simple">10.1093/ehjqcco/qcv005</ext-link></comment> <object-id pub-id-type="pmid">29474568</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref006">
<label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Hripcsak</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Albers</surname> <given-names>DJ</given-names></name>. <article-title>Next-generation phenotyping of electronic health records</article-title>. <source>J Am Med Inform Assoc</source>. <year>2013</year>;<volume>20</volume>(<issue>1</issue>):<fpage>117</fpage>–<lpage>121</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/amiajnl-2012-001145" xlink:type="simple">10.1136/amiajnl-2012-001145</ext-link></comment> <object-id pub-id-type="pmid">22955496</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref007">
<label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Hersh</surname> <given-names>WR</given-names></name>, <name name-style="western"><surname>Weiner</surname> <given-names>MG</given-names></name>, <name name-style="western"><surname>Embi</surname> <given-names>PJ</given-names></name>, <name name-style="western"><surname>Logan</surname> <given-names>JR</given-names></name>, <name name-style="western"><surname>Payne</surname> <given-names>PRO</given-names></name>, <name name-style="western"><surname>Bernstam</surname> <given-names>EV</given-names></name>, <etal>et al</etal>. <article-title>Caveats for the Use of Operational Electronic Health Record Data in Comparative Effectiveness Research</article-title>. <source>Med Care</source>. <year>2013</year>;<volume>51</volume>(<issue>803</issue>):<fpage>S30</fpage>–<lpage>S37</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1097/MLR.0b013e31829b1dbd" xlink:type="simple">10.1097/MLR.0b013e31829b1dbd</ext-link></comment> <object-id pub-id-type="pmid">23774517</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref008">
<label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Lin</surname> <given-names>JH</given-names></name>, <name name-style="western"><surname>Haug</surname> <given-names>PJ</given-names></name>. <article-title>Exploiting missing clinical data in Bayesian network modeling for predicting medical problems</article-title>. <source>J Biomed Inform</source>. <year>2008</year>;<volume>41</volume>(<issue>1</issue>):<fpage>1</fpage>–<lpage>14</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jbi.2007.06.001" xlink:type="simple">10.1016/j.jbi.2007.06.001</ext-link></comment> <object-id pub-id-type="pmid">17625974</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref009">
<label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Lisboa</surname> <given-names>PJG</given-names></name>, <name name-style="western"><surname>Wong</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Harris</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Swindell</surname> <given-names>R</given-names></name>. <article-title>A Bayesian neural network approach for modelling censored data with an application to prognosis after surgery for breast cancer</article-title>. <source>Artif Intell Med</source>. <year>2003</year>;<volume>28</volume>(<issue>1</issue>):<fpage>1</fpage>–<lpage>25</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/S0933-3657(03)00033-2" xlink:type="simple">10.1016/S0933-3657(03)00033-2</ext-link></comment> <object-id pub-id-type="pmid">12850311</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref010">
<label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Bhaskaran</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Forbes</surname> <given-names>HJ</given-names></name>, <name name-style="western"><surname>Douglas</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Leon</surname> <given-names>DA</given-names></name>, <name name-style="western"><surname>Smeeth</surname> <given-names>L</given-names></name>. <article-title>Representativeness and optimal use of body mass index (BMI) in the UK Clinical Practice Research Datalink (CPRD)</article-title>. <source>BMJ Open</source>. <year>2013</year>;<volume>3</volume>(<issue>9</issue>):<fpage>e003389</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/bmjopen-2013-003389" xlink:type="simple">10.1136/bmjopen-2013-003389</ext-link></comment> <object-id pub-id-type="pmid">24038008</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref011">
<label>11</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Roderick J A Little</surname> <given-names>DBR</given-names></name>. <source>Statistical Analysis with Missing Data</source>. <edition>2nd ed</edition>. <publisher-name>Wiley</publisher-name>; <year>2002</year>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref012">
<label>12</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Siontis</surname> <given-names>GCM</given-names></name>, <name name-style="western"><surname>Tzoulaki</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Siontis</surname> <given-names>KC</given-names></name>, <name name-style="western"><surname>Ioannidis</surname> <given-names>JPA</given-names></name>. <article-title>Comparisons of established risk prediction models for cardiovascular disease: systematic review</article-title>. <source>BMJ</source>. <year>2012</year>;<volume>344</volume>:<fpage>e3318</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/bmj.e3318" xlink:type="simple">10.1136/bmj.e3318</ext-link></comment> <object-id pub-id-type="pmid">22628003</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref013">
<label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Rapsomaniki</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Perel</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Denaxas</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>George</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Nicholas</surname> <given-names>O</given-names></name>, <etal>et al</etal>. <article-title>Prognostic models for stable coronary artery disease based on electronic health record cohort of 102 023 patients</article-title>. <source>Eur Heart J</source>. <year>2014</year>;<volume>35</volume>(<issue>13</issue>):<fpage>844</fpage>–<lpage>852</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/eurheartj/eht533" xlink:type="simple">10.1093/eurheartj/eht533</ext-link></comment> <object-id pub-id-type="pmid">24353280</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref014">
<label>14</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Jones</surname> <given-names>MP</given-names></name>. <article-title>Indicator and Stratification Methods for Missing Explanatory Variables in Multiple Linear Regression</article-title>. <source>J Am Stat Assoc</source>. <year>1996</year>;<volume>91</volume>(<issue>433</issue>):<fpage>222</fpage>–<lpage>230</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/01621459.1996.10476680" xlink:type="simple">10.1080/01621459.1996.10476680</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref015">
<label>15</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Chen</surname> <given-names>X</given-names></name>, <name name-style="western"><surname>Ishwaran</surname> <given-names>H</given-names></name>. <article-title>Random forests for genomic data analysis</article-title>. <source>Genomics</source>. <year>2012</year>;<volume>99</volume>(<issue>6</issue>):<fpage>323</fpage>–<lpage>329</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.ygeno.2012.04.003" xlink:type="simple">10.1016/j.ygeno.2012.04.003</ext-link></comment> <object-id pub-id-type="pmid">22546560</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref016">
<label>16</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Wiens</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Shenoy</surname> <given-names>ES</given-names></name>. <article-title>Machine Learning for Healthcare: On the Verge of a Major Shift in Healthcare Epidemiology</article-title>. <source>Clin Infect Dis</source>. <year>2017</year>;</mixed-citation>
</ref>
<ref id="pone.0202344.ref017">
<label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Kattan</surname> <given-names>MW</given-names></name>, <name name-style="western"><surname>Hess</surname> <given-names>KR</given-names></name>, <name name-style="western"><surname>Beck</surname> <given-names>JR</given-names></name>. <article-title>Experiments to determine whether recursive partitioning (CART) or an artificial neural network overcomes theoretical limitations of Cox proportional hazards regression</article-title>. <source>Comput Biomed Res</source>. <year>1998</year>;<volume>31</volume>(<issue>5</issue>):<fpage>363</fpage>–<lpage>373</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1006/cbmr.1998.1488" xlink:type="simple">10.1006/cbmr.1998.1488</ext-link></comment> <object-id pub-id-type="pmid">9790741</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref018">
<label>18</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ross</surname> <given-names>EG</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>NH</given-names></name>, <name name-style="western"><surname>Dalman</surname> <given-names>RL</given-names></name>, <name name-style="western"><surname>Nead</surname> <given-names>KT</given-names></name>, <name name-style="western"><surname>Cooke</surname> <given-names>JP</given-names></name>, <name name-style="western"><surname>Leeper</surname> <given-names>NJ</given-names></name>. <article-title>The use of machine learning for the identification of peripheral artery disease and future mortality risk</article-title>. <source>J Vasc Surg</source>. <year>2016</year>;<volume>64</volume>(<issue>5</issue>):<fpage>1515</fpage>–<lpage>1522.e3</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jvs.2016.04.026" xlink:type="simple">10.1016/j.jvs.2016.04.026</ext-link></comment> <object-id pub-id-type="pmid">27266594</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref019">
<label>19</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ishwaran</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Kogalur</surname> <given-names>UB</given-names></name>, <name name-style="western"><surname>Blackstone</surname> <given-names>EH</given-names></name>, <name name-style="western"><surname>Lauer</surname> <given-names>MS</given-names></name>. <article-title>Random survival forests</article-title>. <source>Ann Appl Stat</source>. <year>2008</year>;<volume>2</volume>(<issue>3</issue>):<fpage>841</fpage>–<lpage>860</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1214/08-AOAS169" xlink:type="simple">10.1214/08-AOAS169</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref020">
<label>20</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Chen</surname> <given-names>HC</given-names></name>, <name name-style="western"><surname>Kodell</surname> <given-names>RL</given-names></name>, <name name-style="western"><surname>Cheng</surname> <given-names>KF</given-names></name>, <name name-style="western"><surname>Chen</surname> <given-names>JJ</given-names></name>. <article-title>Assessment of performance of survival prediction models for cancer prognosis</article-title>. <source>BMC Med Res Methodol</source>. <year>2012</year>;<volume>12</volume>:<fpage>102</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2288-12-102" xlink:type="simple">10.1186/1471-2288-12-102</ext-link></comment> <object-id pub-id-type="pmid">22824262</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref021">
<label>21</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Schmid</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Wright</surname> <given-names>MN</given-names></name>, <name name-style="western"><surname>Ziegler</surname> <given-names>A</given-names></name>. <article-title>On the use of Harrell’s C for clinical risk prediction via random survival forests</article-title>. <source>Expert Syst Appl</source>. <year>2016</year>;<volume>63</volume>:<fpage>450</fpage>–<lpage>459</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.eswa.2016.07.018" xlink:type="simple">10.1016/j.eswa.2016.07.018</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref022">
<label>22</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Miotto</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Li</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Kidd</surname> <given-names>BA</given-names></name>, <name name-style="western"><surname>Dudley</surname> <given-names>JT</given-names></name>. <article-title>Deep Patient: An Unsupervised Representation to Predict the Future of Patients from the Electronic Health Records</article-title>. <source>Sci Rep</source>. <year>2016</year>;<volume>6</volume>:<fpage>26094</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/srep26094" xlink:type="simple">10.1038/srep26094</ext-link></comment> <object-id pub-id-type="pmid">27185194</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref023">
<label>23</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Weng</surname> <given-names>SF</given-names></name>, <name name-style="western"><surname>Reps</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Kai</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Garibaldi</surname> <given-names>JM</given-names></name>, <name name-style="western"><surname>Qureshi</surname> <given-names>N</given-names></name>. <article-title>Can machine-learning improve cardiovascular risk prediction using routine clinical data?</article-title> <source>PLoS One</source>. <year>2017</year>;<volume>12</volume>(<issue>4</issue>):<fpage>e0174944</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0174944" xlink:type="simple">10.1371/journal.pone.0174944</ext-link></comment> <object-id pub-id-type="pmid">28376093</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref024">
<label>24</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Wu</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Roy</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Stewart</surname> <given-names>WF</given-names></name>. <article-title>Prediction modeling using EHR data: challenges, strategies, and a comparison of machine learning approaches</article-title>. <source>Med Care</source>. <year>2010</year>;<volume>48</volume>(<issue>6 Suppl</issue>):<fpage>S106</fpage>–<lpage>13</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1097/MLR.0b013e3181de9e17" xlink:type="simple">10.1097/MLR.0b013e3181de9e17</ext-link></comment> <object-id pub-id-type="pmid">20473190</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref025">
<label>25</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Brisimi</surname> <given-names>TS</given-names></name>, <name name-style="western"><surname>Chen</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Mela</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Olshevsky</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Paschalidis</surname> <given-names>IC</given-names></name>, <name name-style="western"><surname>Shi</surname> <given-names>W</given-names></name>. <article-title>Federated learning of predictive models from federated Electronic Health Records</article-title>. <source>Int J Med Inform</source>. <year>2018</year>; <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.ijmedinf.2018.01.007" xlink:type="simple">10.1016/j.ijmedinf.2018.01.007</ext-link></comment> <object-id pub-id-type="pmid">29500022</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref026">
<label>26</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Denaxas</surname> <given-names>SC</given-names></name>, <name name-style="western"><surname>George</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Herrett</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>AD</given-names></name>, <name name-style="western"><surname>Kalra</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Hingorani</surname> <given-names>AD</given-names></name>, <etal>et al</etal>. <article-title>Data resource profile: cardiovascular disease research using linked bespoke studies and electronic health records (CALIBER)</article-title>. <source>Int J Epidemiol</source>. <year>2012</year>;<volume>41</volume>(<issue>6</issue>):<fpage>1625</fpage>–<lpage>1638</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/ije/dys188" xlink:type="simple">10.1093/ije/dys188</ext-link></comment> <object-id pub-id-type="pmid">23220717</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref027">
<label>27</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Herrett</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>AD</given-names></name>, <name name-style="western"><surname>Boggon</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Denaxas</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Smeeth</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>van Staa</surname> <given-names>T</given-names></name>, <etal>et al</etal>. <article-title>Completeness and diagnostic validity of recording acute myocardial infarction events in primary care, hospital care, disease registry, and national mortality records: cohort study</article-title>. <source>BMJ</source>. <year>2013</year>;<volume>346</volume>:<fpage>f2350</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/bmj.f2350" xlink:type="simple">10.1136/bmj.f2350</ext-link></comment> <object-id pub-id-type="pmid">23692896</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref028">
<label>28</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Rapsomaniki</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Perel</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Denaxas</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>George</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Nicholas</surname> <given-names>O</given-names></name>, <etal>et al</etal>. <article-title>Prognostic models for stable coronary artery disease based on electronic health record cohort of 102 023 patients</article-title>. <source>Eur Heart J</source>. <year>2014</year>;<volume>35</volume>(<issue>13</issue>):<fpage>844</fpage>–<lpage>852</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/eurheartj/eht533" xlink:type="simple">10.1093/eurheartj/eht533</ext-link></comment> <object-id pub-id-type="pmid">24353280</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref029">
<label>29</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>van Buuren</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Groothuis-Oudshoorn</surname> <given-names>K</given-names></name>. <article-title>mice: Multivariate Imputation by Chained Equations in R</article-title>. <source>Journal of Statistical Software, Articles</source>. <year>2011</year>;<volume>45</volume>(<issue>3</issue>):<fpage>1</fpage>–<lpage>67</lpage>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref030">
<label>30</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Levey</surname> <given-names>AS</given-names></name>, <name name-style="western"><surname>Stevens</surname> <given-names>LA</given-names></name>, <name name-style="western"><surname>Schmid</surname> <given-names>CH</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>YL</given-names></name>, <name name-style="western"><surname>Castro</surname> <given-names>AF</given-names> <suffix>3rd</suffix></name>, <name name-style="western"><surname>Feldman</surname> <given-names>HI</given-names></name>, <etal>et al</etal>. <article-title>A new equation to estimate glomerular filtration rate</article-title>. <source>Ann Intern Med</source>. <year>2009</year>;<volume>150</volume>(<issue>9</issue>):<fpage>604</fpage>–<lpage>612</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7326/0003-4819-150-9-200905050-00006" xlink:type="simple">10.7326/0003-4819-150-9-200905050-00006</ext-link></comment> <object-id pub-id-type="pmid">19414839</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref031">
<label>31</label>
<mixed-citation publication-type="other" xlink:type="simple">Miao F, Cai YP, Zhang YT, Li CY. Is Random Survival Forest an Alternative to Cox Proportional Model on Predicting Cardiovascular Disease? In: 6th European Conference of the International Federation for Medical and Biological Engineering. Springer, Cham; 2015. p. 740–743. Available from: <ext-link ext-link-type="uri" xlink:href="https://link.springer.com/chapter/10.1007/978-3-319-11128-5_184" xlink:type="simple">https://link.springer.com/chapter/10.1007/978-3-319-11128-5_184</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref032">
<label>32</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Breiman</surname> <given-names>L</given-names></name>. <article-title>Random Forests</article-title>. <source>Mach Learn</source>. <year>2001</year>;<volume>45</volume>(<issue>1</issue>):<fpage>5</fpage>–<lpage>32</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1023/A:1010933404324" xlink:type="simple">10.1023/A:1010933404324</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref033">
<label>33</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Loh</surname> <given-names>WY</given-names></name>, <name name-style="western"><surname>Shih</surname> <given-names>YS</given-names></name>. <article-title>Split selection methods for classification trees</article-title>. <source>Stat Sin</source>. <year>1997</year>;<volume>7</volume>(<issue>4</issue>):<fpage>815</fpage>–<lpage>840</lpage>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref034">
<label>34</label>
<mixed-citation publication-type="other" xlink:type="simple">Wright MN, Dankowski T, Ziegler A. Random forests for survival analysis using maximally selected rank statistics. 2016;.</mixed-citation>
</ref>
<ref id="pone.0202344.ref035">
<label>35</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ishwaran</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Kogalur</surname> <given-names>UB</given-names></name>. <article-title>Random survival forests for R</article-title>. <source>Rnews</source>. <year>2007</year>;<volume>7</volume>(<issue>2</issue>):<fpage>25</fpage>–<lpage>31</lpage>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref036">
<label>36</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Breiman</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Friedman</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Stone</surname> <given-names>CJ</given-names></name>, <name name-style="western"><surname>Olshen</surname> <given-names>RA</given-names></name>. <source>Classification and Regression Trees (Wadsworth Statistics/Probability)</source>. <edition>1st ed</edition>. <publisher-name>Chapman and Hall/CRC</publisher-name>; <year>1984</year>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref037">
<label>37</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Quinlan</surname> <given-names>JR</given-names></name>. <source>C4.5: Programs for Machine Learning</source>. <publisher-loc>San Francisco, CA, USA</publisher-loc>: <publisher-name>Morgan Kaufmann Publishers Inc</publisher-name>.; <year>1993</year>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dl.acm.org/citation.cfm?id=152181" xlink:type="simple">http://dl.acm.org/citation.cfm?id=152181</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref038">
<label>38</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Díaz-Uriarte</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Alvarez de Andrés</surname> <given-names>S</given-names></name>. <article-title>Gene selection and classification of microarray data using random forest</article-title>. <source>BMC Bioinformatics</source>. <year>2006</year>;<volume>7</volume>:<fpage>3</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-7-3" xlink:type="simple">10.1186/1471-2105-7-3</ext-link></comment> <object-id pub-id-type="pmid">16398926</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref039">
<label>39</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Diaz-Uriarte</surname> <given-names>R</given-names></name>. <article-title>GeneSrF and varSelRF: a web-based tool and R package for gene selection and classification using random forest</article-title>. <source>BMC Bioinformatics</source>. <year>2007</year>;<volume>8</volume>:<fpage>328</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-8-328" xlink:type="simple">10.1186/1471-2105-8-328</ext-link></comment> <object-id pub-id-type="pmid">17767709</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref040">
<label>40</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Fan</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Feng</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Wu</surname> <given-names>Y</given-names></name>. <source>High-dimensional variable selection for Cox’s proportional hazards model</source>. <year>2010</year>;.</mixed-citation>
</ref>
<ref id="pone.0202344.ref041">
<label>41</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Strobl</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Boulesteix</surname> <given-names>AL</given-names></name>, <name name-style="western"><surname>Kneib</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Augustin</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Zeileis</surname> <given-names>A</given-names></name>. <article-title>Conditional variable importance for random forests</article-title>. <source>BMC Bioinformatics</source>. <year>2008</year>;<volume>9</volume>:<fpage>307</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-9-307" xlink:type="simple">10.1186/1471-2105-9-307</ext-link></comment> <object-id pub-id-type="pmid">18620558</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref042">
<label>42</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Strobl</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Boulesteix</surname> <given-names>AL</given-names></name>, <name name-style="western"><surname>Zeileis</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Hothorn</surname> <given-names>T</given-names></name>. <article-title>Bias in random forest variable importance measures: illustrations, sources and a solution</article-title>. <source>BMC Bioinformatics</source>. <year>2007</year>;<volume>8</volume>:<fpage>25</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-8-25" xlink:type="simple">10.1186/1471-2105-8-25</ext-link></comment> <object-id pub-id-type="pmid">17254353</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref043">
<label>43</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Perperoglou</surname> <given-names>A</given-names></name>. <article-title>Cox models with dynamic ridge penalties on time-varying effects of the covariates</article-title>. <source>Stat Med</source>. <year>2014</year>;<volume>33</volume>(<issue>1</issue>):<fpage>170</fpage>–<lpage>180</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/sim.5921" xlink:type="simple">10.1002/sim.5921</ext-link></comment> <object-id pub-id-type="pmid">23913655</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref044">
<label>44</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Harrell</surname> <given-names>FE</given-names> <suffix>Jr</suffix></name>, <name name-style="western"><surname>Lee</surname> <given-names>KL</given-names></name>, <name name-style="western"><surname>Mark</surname> <given-names>DB</given-names></name>. <article-title>Multivariable prognostic models: issues in developing models, evaluating assumptions and adequacy, and measuring and reducing errors</article-title>. <source>Stat Med</source>. <year>1996</year>;<volume>15</volume>(<issue>4</issue>):<fpage>361</fpage>–<lpage>387</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/(SICI)1097-0258(19960229)15:4&lt;361::AID-SIM168&gt;3.0.CO;2-4" xlink:type="simple">10.1002/(SICI)1097-0258(19960229)15:4&lt;361::AID-SIM168&gt;3.0.CO;2-4</ext-link></comment> <object-id pub-id-type="pmid">8668867</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref045">
<label>45</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<collab>Diverse Populations Collaborative Group</collab>. <article-title>Prediction of mortality from coronary heart disease among diverse populations: is there a common predictive function?</article-title> <source>Heart</source>. <year>2002</year>;<volume>88</volume>(<issue>3</issue>):<fpage>222</fpage>–<lpage>228</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/heart.88.3.222" xlink:type="simple">10.1136/heart.88.3.222</ext-link></comment> <object-id pub-id-type="pmid">12181209</object-id></mixed-citation>
</ref>
<ref id="pone.0202344.ref046">
<label>46</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Hastie</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Tibshirani</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Friedman</surname> <given-names>J</given-names></name>. <source>The elements of statistical learning</source>;. Available from: <ext-link ext-link-type="uri" xlink:href="https://statweb.stanford.edu/~tibs/ElemStatLearn/" xlink:type="simple">https://statweb.stanford.edu/~tibs/ElemStatLearn/</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref047">
<label>47</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Poppe</surname> <given-names>KK</given-names></name>, <name name-style="western"><surname>Doughty</surname> <given-names>RN</given-names></name>, <name name-style="western"><surname>Wells</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Gentles</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Hemingway</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Jackson</surname> <given-names>R</given-names></name>, <etal>et al</etal>. <article-title>Developing and validating a cardiovascular risk score for patients in the community with prior cardiovascular disease</article-title>. <source>Heart</source>. <year>2017</year>; <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1136/heartjnl-2016-310668" xlink:type="simple">10.1136/heartjnl-2016-310668</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0202344.ref048">
<label>48</label>
<mixed-citation publication-type="other" xlink:type="simple">Wainer J. Comparison of 14 different families of classification algorithms on 115 binary datasets. 2016;.</mixed-citation>
</ref>
<ref id="pone.0202344.ref049">
<label>49</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Fernández-Delgado</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Cernadas</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Barro</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Amorim</surname> <given-names>D</given-names></name>. <article-title>Do we Need Hundreds of Classifiers to Solve Real World Classification Problems?</article-title> <source>J Mach Learn Res</source>. <year>2014</year>;<volume>15</volume>:<fpage>3133</fpage>–<lpage>3181</lpage>.</mixed-citation>
</ref>
<ref id="pone.0202344.ref050">
<label>50</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Zhang</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Lu</surname> <given-names>Y</given-names></name>. <article-title>Bias-corrected random forests in regression</article-title>. <source>J Appl Stat</source>. <year>2012</year>;<volume>39</volume>(<issue>1</issue>):<fpage>151</fpage>–<lpage>160</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/02664763.2011.578621" xlink:type="simple">10.1080/02664763.2011.578621</ext-link></comment></mixed-citation>
</ref>
</ref-list>
</back>
</article>