<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id>
<journal-title-group>
<journal-title>PLOS Computational Biology</journal-title>
</journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-20-00012</article-id>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Epidemiology</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Epidemiology</subject><subj-group><subject>Infectious disease epidemiology</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Infectious diseases</subject><subj-group><subject>Infectious disease epidemiology</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Infectious diseases</subject><subj-group><subject>Viral diseases</subject><subj-group><subject>Influenza</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Population biology</subject><subj-group><subject>Population dynamics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Mathematical and statistical techniques</subject><subj-group><subject>Statistical methods</subject><subj-group><subject>Forecasting</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Statistics</subject><subj-group><subject>Statistical methods</subject><subj-group><subject>Forecasting</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Infectious diseases</subject><subj-group><subject>Viral diseases</subject><subj-group><subject>SARS</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Infectious diseases</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Data visualization</subject><subj-group><subject>Infographics</subject><subj-group><subject>Graphs</subject></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Using information theory to optimise epidemic models for real-time prediction and estimation</article-title>
<alt-title alt-title-type="running-head">Information theoretic epidemic model selection</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-7806-3605</contrib-id>
<name name-style="western">
<surname>Parag</surname> <given-names>Kris V.</given-names></name>
<role content-type="https://casrai.org/credit/">Conceptualization</role>
<role content-type="https://casrai.org/credit/">Formal analysis</role>
<role content-type="https://casrai.org/credit/">Investigation</role>
<role content-type="https://casrai.org/credit/">Methodology</role>
<role content-type="https://casrai.org/credit/">Project administration</role>
<role content-type="https://casrai.org/credit/">Software</role>
<role content-type="https://casrai.org/credit/">Validation</role>
<role content-type="https://casrai.org/credit/">Writing – original draft</role>
<role content-type="https://casrai.org/credit/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-0195-2463</contrib-id>
<name name-style="western">
<surname>Donnelly</surname> <given-names>Christl A.</given-names></name>
<role content-type="https://casrai.org/credit/">Supervision</role>
<role content-type="https://casrai.org/credit/">Validation</role>
<role content-type="https://casrai.org/credit/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>MRC Centre for Global Infectious Disease Analysis, Imperial College London, London, W2 1PG, United Kingdom</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Department of Statistics, University of Oxford, Oxford, OX1 3LB, United Kingdom</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Ferrari</surname> <given-names>Matthew (Matt)</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>The Pennsylvania State University, UNITED STATES</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">k.parag@imperial.ac.uk</email></corresp>
</author-notes>
<pub-date pub-type="collection">
<month>7</month>
<year>2020</year>
</pub-date>
<pub-date pub-type="epub">
<day>1</day>
<month>7</month>
<year>2020</year>
</pub-date>
<volume>16</volume>
<issue>7</issue>
<elocation-id>e1007990</elocation-id>
<history>
<date date-type="received">
<day>3</day>
<month>1</month>
<year>2020</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>5</month>
<year>2020</year>
</date>
</history>
<permissions>
<copyright-year>2020</copyright-year>
<copyright-holder>Parag, Donnelly</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pcbi.1007990"/>
<abstract>
<p>The effective reproduction number, <italic>R</italic><sub><italic>t</italic></sub>, is a key time-varying prognostic for the growth rate of any infectious disease epidemic. Significant changes in <italic>R</italic><sub><italic>t</italic></sub> can forewarn about new transmissions within a population or predict the efficacy of interventions. Inferring <italic>R</italic><sub><italic>t</italic></sub> reliably and in real-time from observed time-series of infected (demographic) data is an important problem in population dynamics. The renewal or branching process model is a popular solution that has been applied to Ebola and Zika virus disease outbreaks, among others, and is currently being used to investigate the ongoing COVID-19 pandemic. This model estimates <italic>R</italic><sub><italic>t</italic></sub> using a heuristically chosen piecewise function. While this facilitates real-time detection of statistically significant <italic>R</italic><sub><italic>t</italic></sub> changes, inference is highly sensitive to the function choice. Improperly chosen piecewise models might ignore meaningful changes or over-interpret noise-induced ones, yet produce visually reasonable estimates. No principled piecewise selection scheme exists. We develop a practical yet rigorous scheme using the accumulated prediction error (APE) metric from information theory, which deems the model capable of describing the observed data using the fewest bits as most justified. We derive exact posterior prediction distributions for infected population size and integrate these within an APE framework to obtain an exact and reliable method for identifying the piecewise function best supported by available epidemic data. We find that this choice optimises short-term prediction accuracy and can rapidly detect salient fluctuations in <italic>R</italic><sub><italic>t</italic></sub>, and hence the infected population growth rate, in real-time over the course of an unfolding epidemic. Moreover, we emphasise the need for formal selection by exposing how common heuristic choices, which seem sensible, can be misleading. Our APE-based method is easily computed and broadly applicable to statistically similar models found in phylogenetics and macroevolution, for example. Our results explore the relationships among estimate precision, forecast reliability and model complexity.</p>
</abstract>
<abstract abstract-type="summary">
<title>Author summary</title>
<p>Understanding how the population of infected individuals (which may be humans, animals or plants) fluctuates in size over the course of an epidemic is an important problem in epidemiology and ecology. The effective reproduction number, <italic>R</italic>, provides an intuitive and useful way of describing these fluctuations by characterising the growth rate of the infected population. An <italic>R</italic> &gt; 1 signifies a burgeoning epidemic whereas <italic>R</italic> &lt; 1 indicates a declining one. Public health agencies often use <italic>R</italic> to inform or corroborate vaccination and quarantine policies. However, popular approaches to inferring <italic>R</italic> from epidemic data make heuristic choices, which may lead to visually reasonable estimates that are deceptive or unreliable. By adapting mathematical tools from information theory, we develop a general and principled scheme for estimating <italic>R</italic> in a data-justified way. Our method exposes the pitfalls of heuristic estimates and provides an easily computable correction that also maximises our ability to predict upcoming population fluctuations. Our work is widely applicable to similar inference problems found in evolution and genetics, demonstrably useful for reliably analysing emerging epidemics in real time and highlights how abstract mathematical concepts can inspire novel and practical biological solutions, showcasing the importance of multidisciplinary research.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000265</institution-id>
<institution>Medical Research Council</institution>
</institution-wrap>
</funding-source>
<award-id>MR/R015600/1</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0002-7806-3605</contrib-id>
<name name-style="western">
<surname>Parag</surname> <given-names>Kris V.</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>KVP and CAD acknowledge joint Centre funding from the UK Medical Research Council and Department for International Development under grant reference MR/R015600/1. CAD thanks the UK National Institute for Health Research Health Protection Research Unit (NIHR HPRU) in Modelling Methodology at Imperial College London in partnership with Public Health England (PHE) for funding (grant HPRU-2012–10080). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="10"/>
<table-count count="0"/>
<page-count count="20"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>PLOS Publication Stage</meta-name>
<meta-value>vor-update-to-uncorrected-proof</meta-value>
</custom-meta>
<custom-meta>
<meta-name>Publication Update</meta-name>
<meta-value>2020-07-14</meta-value>
</custom-meta>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All code and data are available at <ext-link ext-link-type="uri" xlink:href="https://github.com/kpzoo/model-selection-for-epidemic-renewal-models" xlink:type="simple">https://github.com/kpzoo/model-selection-for-epidemic-renewal-models</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>The time-series of newly infected cases (infecteds) observed over the course of an infectious disease epidemic is known as an incidence curve or epi-curve. These curves offer prospective insight into the spread of a disease within an animal or human population by informing on the effective reproduction number, which defines the average number of secondary infections induced by a primary one [<xref ref-type="bibr" rid="pcbi.1007990.ref001">1</xref>]. This reproduction number, denoted <italic>R</italic><sub><italic>t</italic></sub> at time <italic>t</italic>, is an important prognostic of the demographic behaviour of an epidemic. If <italic>R</italic><sub><italic>t</italic></sub> &gt; 1, for example, we can expect and hence prepare for exponentially increasing incidence, whereas if <italic>R</italic><sub><italic>t</italic></sub> &lt; 1, we can be reasonably confident that the epidemic has been arrested [<xref ref-type="bibr" rid="pcbi.1007990.ref001">1</xref>].</p>
<p>Reliably estimating meaningful changes in <italic>R</italic><sub><italic>t</italic></sub> is an important problem in epidemiology and population biology, since it can forewarn about the growth rate of an outbreak and signify the level of control effort that must be initiated or sustained [<xref ref-type="bibr" rid="pcbi.1007990.ref002">2</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>]. While we explicitly consider epidemic applications here, similar reproduction numbers (and growth rates) can be defined for many ecological problems [<xref ref-type="bibr" rid="pcbi.1007990.ref004">4</xref>], for example in species conservation where we might aim to infer species population dynamics from time-series of sample counts.</p>
<p>The renewal model [<xref ref-type="bibr" rid="pcbi.1007990.ref005">5</xref>] is a popular approach for inferring salient fluctuations in <italic>R</italic><sub><italic>t</italic></sub> that is based on the fundamental Euler-Lotka reproduction equation from ecology and evolution [<xref ref-type="bibr" rid="pcbi.1007990.ref004">4</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref006">6</xref>]. This model has been used to predict Ebola virus disease case counts and assess the transmission potential of pandemic influenza and Zika virus, among others [<xref ref-type="bibr" rid="pcbi.1007990.ref002">2</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref007">7</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref008">8</xref>]. Currently, it is in widespread use as a tool for tracking the progress of interventions across countries in response to the COVID-19 pandemic [<xref ref-type="bibr" rid="pcbi.1007990.ref009">9</xref>]. The renewal model may be applied retrospectively to understand the past behaviour of an epidemic or prospectively to gain real-time insight into ongoing outbreak dynamics [<xref ref-type="bibr" rid="pcbi.1007990.ref005">5</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. We only consider the latter here (see [<xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>] for an investigation of the former).</p>
<p>The prospective approach approximates <italic>R</italic><sub><italic>t</italic></sub> with a piecewise-constant function i.e. <italic>R</italic><sub><italic>t</italic></sub> is constant (stable) over some sliding window of <italic>k</italic> time units (e.g. days or weeks) into the past, beyond which a discontinuous change is assumed [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. This formulation models the non-stationary nature of epidemics, capturing the idea that different population dynamics are expected during distinct phases (e.g. onset, growth, control) of the epidemic lifetime. The window length, <italic>k</italic>, is essentially a hypothesis about the stability of <italic>R</italic><sub><italic>t</italic></sub> underpinning the observed epi-curve. It is critical to reliably characterising the epidemic because it controls the time-scale over which <italic>R</italic><sub><italic>t</italic></sub> fluctuations are deemed significant.</p>
<p>Too large or small a <italic>k</italic>-value can respectively lead to over-smoothing (which ignores important changes) or to random noise being misinterpreted as meaningful. Inferring under a wrong <italic>k</italic> can appreciably affect our understanding of an epidemic, as observed in [<xref ref-type="bibr" rid="pcbi.1007990.ref005">5</xref>], where case reproduction numbers (a common <italic>k</italic>-choice) were found to over-smooth significant changes in HIV transmission, for example. Surprisingly, no principled method for optimising <italic>k</italic> exists. Current best practice either relies on heuristic choices [<xref ref-type="bibr" rid="pcbi.1007990.ref007">7</xref>] or provides implicit bounds on <italic>k</italic> [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. Here we adapt the minimum description length (MDL) principle from information theory to develop a simple but rigorous selection framework.</p>
<p>The description length is defined as the number of bits needed to communicate a model (<inline-formula id="pcbi.1007990.e001"><alternatives><graphic id="pcbi.1007990.e001g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e001" xlink:type="simple"/><mml:math display="inline" id="M1"><mml:mi mathvariant="script">M</mml:mi></mml:math></alternatives></inline-formula>) and the data given that model (<inline-formula id="pcbi.1007990.e002"><alternatives><graphic id="pcbi.1007990.e002g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e002" xlink:type="simple"/><mml:math display="inline" id="M2"><mml:mrow><mml:mi mathvariant="script">D</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:mi mathvariant="script">M</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>) on some channel. More complex models increase <inline-formula id="pcbi.1007990.e003"><alternatives><graphic id="pcbi.1007990.e003g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e003" xlink:type="simple"/><mml:math display="inline" id="M3"><mml:mrow><mml:mi>L</mml:mi> <mml:mo>(</mml:mo> <mml:mi mathvariant="script">M</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> (more bits are needed to represent its extra parameters) but decrease <inline-formula id="pcbi.1007990.e004"><alternatives><graphic id="pcbi.1007990.e004g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e004" xlink:type="simple"/><mml:math display="inline" id="M4"><mml:mrow><mml:mi>L</mml:mi> <mml:mo>(</mml:mo> <mml:mi mathvariant="script">D</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:mi mathvariant="script">M</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> (as it can better fit the data) for example. Here <italic>L</italic>(.) indicates length. The MDL model choice minimises <inline-formula id="pcbi.1007990.e005"><alternatives><graphic id="pcbi.1007990.e005g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e005" xlink:type="simple"/><mml:math display="inline" id="M5"><mml:mrow><mml:mi>L</mml:mi> <mml:mo>(</mml:mo> <mml:mi mathvariant="script">M</mml:mi> <mml:mo>)</mml:mo> <mml:mo>+</mml:mo> <mml:mi>L</mml:mi> <mml:mo>(</mml:mo> <mml:mi mathvariant="script">D</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:mi mathvariant="script">M</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> and is considered the model most justified by the data <inline-formula id="pcbi.1007990.e006"><alternatives><graphic id="pcbi.1007990.e006g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e006" xlink:type="simple"/><mml:math display="inline" id="M6"><mml:mi mathvariant="script">D</mml:mi></mml:math></alternatives></inline-formula> [<xref ref-type="bibr" rid="pcbi.1007990.ref012">12</xref>]. We adapt an approximation to MDL known as the accumulated prediction error (APE) [<xref ref-type="bibr" rid="pcbi.1007990.ref013">13</xref>] to identify the <italic>k</italic> best justified by the available epi-curve, <italic>k</italic>* (see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref> for further details and for the general definition of APE).</p>
<p>We analytically derive the posterior predictive incidence distribution of the renewal model, which allows us to evaluate cumulative log-loss prediction scores at any <italic>k</italic> exactly. The APE choice, <italic>k</italic>*, minimises these scores and, additionally, optimises short-term prediction accuracy [<xref ref-type="bibr" rid="pcbi.1007990.ref014">14</xref>]. Our method is valid at all sample sizes, easily computed for arbitrary model dimensions and, unlike many selection criteria of similar computability, includes parametric complexity [<xref ref-type="bibr" rid="pcbi.1007990.ref012">12</xref>]. Parametric complexity measures how functional relationships among parameters influence complexity, and when ignored (as in the Bayesian or Akaike information criteria) can lead to biased renewal model selection [<xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>].</p>
<p>The APE metric designates the model that best predicts unseen data from the same underlying process, and not the one that best fits existing data, as optimal and of justified complexity [<xref ref-type="bibr" rid="pcbi.1007990.ref013">13</xref>]. This is equivalent to minimising what is called the generalisation or out-of-sample error in machine learning, and is known to balance under and overfitting. Our APE-based approach therefore pinpoints and characterises only those <italic>R</italic><sub><italic>t</italic></sub> fluctuations that are integral to achieving reliable short-term incidence growth predictions.</p>
<p>The performance, speed and computational ease of our approach makes it suitable for real-time forecasting and model selection. It could therefore serve as a stand-alone computational tool or be integrated within existing real-time frameworks, such as in [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>] or [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>], to provide emerging insights into infected population dynamics, or to assess the prospective efficacy of implemented interventions (e.g. vaccination or quarantine). Public health policy decisions or preparedness plans based on improperly specified <italic>k</italic>-windows could be misinformed or overconfident. Our method hopefully limits these risks.</p>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Methods</title>
<sec id="sec003">
<title>Renewal model window-sizing problem</title>
<p>Let the incidence or number of newly infected cases in an epidemic at present time <italic>t</italic>, be <italic>I</italic><sub><italic>t</italic></sub>. The incidence curve is a historical record of these case counts from the start of the outbreak and is summarised as <inline-formula id="pcbi.1007990.e007"><alternatives><graphic id="pcbi.1007990.e007g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e007" xlink:type="simple"/><mml:math display="inline" id="M7"><mml:mrow><mml:msubsup><mml:mi>I</mml:mi> <mml:mn>1</mml:mn> <mml:mi>t</mml:mi></mml:msubsup> <mml:mo>=</mml:mo> <mml:mrow><mml:mo>{</mml:mo> <mml:msub><mml:mi>I</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>:</mml:mo> <mml:mn>1</mml:mn> <mml:mo>≤</mml:mo> <mml:mi>s</mml:mi> <mml:mo>≤</mml:mo> <mml:mi>t</mml:mi> <mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula>, with <italic>s</italic> as a time-indexing variable. For convenience, we assume that incidence is available on a daily scale so that <inline-formula id="pcbi.1007990.e008"><alternatives><graphic id="pcbi.1007990.e008g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e008" xlink:type="simple"/><mml:math display="inline" id="M8"><mml:msubsup><mml:mi>I</mml:mi> <mml:mn>1</mml:mn> <mml:mi>t</mml:mi></mml:msubsup></mml:math></alternatives></inline-formula> is a vector of <italic>t</italic> daily counts (weeks or months could be used instead). The associated effective reproduction number and total infectiousness of the epidemic are denoted <italic>R</italic><sub><italic>t</italic></sub> and Λ<sub><italic>t</italic></sub>, respectively.</p>
<p>Here <italic>R</italic><sub><italic>t</italic></sub> is the number of secondary cases that are induced, on average, by a single primary case at <italic>t</italic> [<xref ref-type="bibr" rid="pcbi.1007990.ref001">1</xref>], while Λ<sub><italic>t</italic></sub> measures the cumulative impact of past cases, from the epidemic origin at <italic>s</italic> = 1, on the present. The generation time distribution of the epidemic, which describes the elapsed time between a primary and secondary case, controls Λ<sub><italic>t</italic></sub> (see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>). In the problems we consider, <inline-formula id="pcbi.1007990.e009"><alternatives><graphic id="pcbi.1007990.e009g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e009" xlink:type="simple"/><mml:math display="inline" id="M9"><mml:msubsup><mml:mi>I</mml:mi> <mml:mn>1</mml:mn> <mml:mi>t</mml:mi></mml:msubsup></mml:math></alternatives></inline-formula> and <inline-formula id="pcbi.1007990.e010"><alternatives><graphic id="pcbi.1007990.e010g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e010" xlink:type="simple"/><mml:math display="inline" id="M10"><mml:msubsup><mml:mo>Λ</mml:mo> <mml:mn>1</mml:mn> <mml:mi>t</mml:mi></mml:msubsup></mml:math></alternatives></inline-formula> form our data, the generation time distribution is assumed known, and <inline-formula id="pcbi.1007990.e011"><alternatives><graphic id="pcbi.1007990.e011g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e011" xlink:type="simple"/><mml:math display="inline" id="M11"><mml:msubsup><mml:mi>R</mml:mi> <mml:mn>1</mml:mn> <mml:mi>t</mml:mi></mml:msubsup></mml:math></alternatives></inline-formula> or groupings of this vector are the parameters to be inferred.</p>
<p>The renewal model [<xref ref-type="bibr" rid="pcbi.1007990.ref005">5</xref>] derives from the classic Euler-Lotka equation [<xref ref-type="bibr" rid="pcbi.1007990.ref006">6</xref>] (see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>), and defines the Poisson distributed (Poiss) relationship <italic>I</italic><sub><italic>t</italic></sub> ∼ Poiss(<italic>R</italic><sub><italic>t</italic></sub>Λ<sub><italic>t</italic></sub>). This means that <inline-formula id="pcbi.1007990.e012"><alternatives><graphic id="pcbi.1007990.e012g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e012" xlink:type="simple"/><mml:math display="inline" id="M12"><mml:mrow><mml:mi>ℙ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mn>1</mml:mn><mml:mrow><mml:mi>x</mml:mi><mml:mo>!</mml:mo></mml:mrow></mml:mfrac></mml:mstyle><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:msub><mml:mo>Λ</mml:mo><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>R</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:msub><mml:mo>Λ</mml:mo><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mi>x</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, with <italic>x</italic> indexing possible <italic>I</italic><sub><italic>t</italic></sub> values. This formulation assumes maximum epidemic non-stationarity (i.e. that demographic transmission statistics change at every time unit) and hence only uses the most recent data (<italic>I</italic><sub><italic>t</italic></sub>, Λ<sub><italic>t</italic></sub>) to infer the current <italic>R</italic><sub><italic>t</italic></sub>. While this maximises the fitting flexibility of the renewal model, it often results in noisy and unreliable estimates that possess many spurious reproduction number changes [<xref ref-type="bibr" rid="pcbi.1007990.ref007">7</xref>]. Consequently, grouping is employed.</p>
<p>This hypothesises that the reproduction number is constant over a <italic>k</italic>-day window into the past and results in a piecewise-constant function that separates salient fluctuations (change-points) from negligible ones (the constant segments) [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>]. We use <italic>R</italic><sub><italic>τ</italic>(<italic>t</italic>)</sub> to indicate that the present reproduction number to be inferred is stationary (constant) over the last <italic>k</italic> points of the incidence curve i.e. the time window <italic>τ</italic>(<italic>t</italic>) ≔ {<italic>t</italic>, <italic>t</italic> − 1, …, <italic>t</italic> − <italic>k</italic> + 1}. The data used to estimate <italic>R</italic><sub><italic>τ</italic>(<italic>t</italic>)</sub> is then <inline-formula id="pcbi.1007990.e013"><alternatives><graphic id="pcbi.1007990.e013g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e013" xlink:type="simple"/><mml:math display="inline" id="M13"><mml:mrow><mml:mo>(</mml:mo> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>t</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>t</mml:mi></mml:msubsup> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>t</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>t</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>. This construction allows us to filter noise and increase estimate reliability but elevates bias. In the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref> we show precisely how grouping achieves this bias-variance trade-off.</p>
<p>Choosing <italic>k</italic> therefore amounts to selecting a belief about the scale over which the epidemic statistics are meaningfully varying and can significantly influence our understanding of the population dynamics of the outbreak. Thus, it is necessary to find a principled method for balancing <italic>k</italic>. Ultimately, we want to find a <italic>k</italic>, denoted <italic>k</italic>*, that (i) is best supported by the epi-curve and (ii) maximises our confidence in making real-time, short-term predictions that can inform prospective epidemic responses. This is no trivial task as <italic>k</italic>* would be sensitive to both the specifically observed stochastic incidence of an epidemic and to how past infections propagate forward in time.</p>
<p>We solve (i)-(ii) by applying the accumulated prediction error (APE) metric from information theory [<xref ref-type="bibr" rid="pcbi.1007990.ref013">13</xref>], which values models on their capacity to predict unseen data from the generating process instead of their ability to fit existing data [<xref ref-type="bibr" rid="pcbi.1007990.ref014">14</xref>]. The properties and mathematical definition of the APE are provided in the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref> and in the subsequent section. The APE uses the window of data preceding time <italic>s</italic>, <inline-formula id="pcbi.1007990.e014"><alternatives><graphic id="pcbi.1007990.e014g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e014" xlink:type="simple"/><mml:math display="inline" id="M14"><mml:mrow><mml:mo>(</mml:mo> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, to predict the incidence at <italic>s</italic> + 1 and assigns a log-score to this prediction. This procedure is repeated over <italic>s</italic> ≤ <italic>t</italic> and for possible <italic>k</italic>-values. The <italic>k</italic> achieving the minimum cumulated log-score is deemed optimal. <xref ref-type="fig" rid="pcbi.1007990.g001">Fig 1</xref> summarises and illustrates the APE algorithm.</p>
<fig id="pcbi.1007990.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Optimal window selection using APE.</title>
<p>(A) An observed incidence curve (blue dots) is sequentially and causally predicted over time <italic>s</italic> ≤ <italic>t</italic> using effective reproduction number estimates based on two possible windows lengths of <italic>k</italic><sub>1</sub> and <italic>k</italic><sub>2</sub> (blue shaded). Predictive distributions are summarised by red error bars (shown only for times <italic>t</italic><sub>1</sub> and <italic>t</italic><sub>2</sub>, respectively for <italic>k</italic><sub>1</sub> and <italic>k</italic><sub>2</sub>), (B) The true reproduction number (<italic>R</italic><sub><italic>s</italic></sub>, dashed black) is estimated under each window length as <inline-formula id="pcbi.1007990.e015"><alternatives><graphic id="pcbi.1007990.e015g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e015" xlink:type="simple"/><mml:math display="inline" id="M15"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:msub><mml:mi>τ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> (blue) and <inline-formula id="pcbi.1007990.e016"><alternatives><graphic id="pcbi.1007990.e016g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e016" xlink:type="simple"/><mml:math display="inline" id="M16"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:msub><mml:mi>τ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> (grey). Large windows (<italic>k</italic><sub>1</sub>) smooth over fluctuations. Small ones (<italic>k</italic><sub>2</sub>) recover more changes but are noisy. (C) The APE assesses <italic>k</italic><sub>1</sub> and <italic>k</italic><sub>2</sub> via the log-loss of their sequential predictions (i.e. from red error bars across time). The window with the smaller APE is better supported by this incidence curve. See <xref ref-type="sec" rid="sec002">Methods</xref> for more mathematical details.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g001" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec004">
<title>Exact model selection using predictive distributions</title>
<p>To adapt APE, we require the posterior predictive incidence distribution of the renewal model, which at time <italic>s</italic> is <inline-formula id="pcbi.1007990.e017"><alternatives><graphic id="pcbi.1007990.e017g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e017" xlink:type="simple"/><mml:math display="inline" id="M17"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, with <italic>x</italic> indexing the space of possible one-step-ahead predictions at <italic>s</italic> + 1, and <inline-formula id="pcbi.1007990.e018"><alternatives><graphic id="pcbi.1007990.e018g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e018" xlink:type="simple"/><mml:math display="inline" id="M18"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:mi mathvariant="double-struck">E</mml:mi> <mml:mrow><mml:mo>[</mml:mo> <mml:mi>x</mml:mi> <mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula>. We assume a gamma conjugate prior distribution on <italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> as is commonly done in the renewal model frameworks of [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>] and because the gamma distribution can fit many unimodal variables on the positive real line. For some hyperparameters <italic>a</italic> and <italic>c</italic> this is <inline-formula id="pcbi.1007990.e019"><alternatives><graphic id="pcbi.1007990.e019g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e019" xlink:type="simple"/><mml:math display="inline" id="M19"><mml:mrow><mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>∼</mml:mo> <mml:mtext>Gam</mml:mtext> <mml:mo>(</mml:mo> <mml:mi>a</mml:mi> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mn>1</mml:mn><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:mfrac></mml:mstyle> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, with Gam as a shape-scale parametrised gamma distribution. The posterior distribution of <italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub>, <inline-formula id="pcbi.1007990.e020"><alternatives><graphic id="pcbi.1007990.e020g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e020" xlink:type="simple"/><mml:math display="inline" id="M20"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, is then
<disp-formula id="pcbi.1007990.e021"><alternatives><graphic id="pcbi.1007990.e021g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e021" xlink:type="simple"/><mml:math display="block" id="M21"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mrow><mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>∼</mml:mo> <mml:mtext>Gam</mml:mtext> <mml:mo>(</mml:mo> <mml:mi>a</mml:mi> <mml:mo>+</mml:mo> <mml:msub><mml:mi>i</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:mfrac><mml:mn>1</mml:mn> <mml:mrow><mml:mi>c</mml:mi> <mml:mo>+</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mfrac> <mml:mo>)</mml:mo> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(1)</label></disp-formula></p>
<p>For convenience we define <inline-formula id="pcbi.1007990.e022"><alternatives><graphic id="pcbi.1007990.e022g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e022" xlink:type="simple"/><mml:math display="inline" id="M22"><mml:mrow><mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>≔</mml:mo> <mml:mi>a</mml:mi> <mml:mo>+</mml:mo> <mml:msub><mml:mi>i</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:msub><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>≔</mml:mo> <mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mn>1</mml:mn><mml:mrow><mml:mi>c</mml:mi> <mml:mo>+</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></alternatives></inline-formula> with <italic>i</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> and λ<sub><italic>τ</italic>(<italic>s</italic>)</sub> are the sum of incidence (<italic>I</italic><sub><italic>s</italic></sub>) and total infectiousness (Λ<sub><italic>s</italic></sub>) over the window <italic>τ</italic>(<italic>s</italic>) (see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>). If a variable <italic>y</italic> ∼ Gam(<italic>α</italic>, <italic>β</italic>) then <inline-formula id="pcbi.1007990.e023"><alternatives><graphic id="pcbi.1007990.e023g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e023" xlink:type="simple"/><mml:math display="inline" id="M23"><mml:mrow><mml:mi>ℙ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mi>α</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi>y</mml:mi><mml:mo>/</mml:mo><mml:mi>β</mml:mi></mml:mrow></mml:msup><mml:msub><mml:mo>/</mml:mo><mml:mrow><mml:msup><mml:mi>β</mml:mi><mml:mi>α</mml:mi></mml:msup><mml:mo>Γ</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>α</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>
 and <inline-formula id="pcbi.1007990.e024"><alternatives><graphic id="pcbi.1007990.e024g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e024" xlink:type="simple"/><mml:math display="inline" id="M24"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi> <mml:mo>[</mml:mo> <mml:mi>y</mml:mi> <mml:mo>]</mml:mo> <mml:mo>=</mml:mo> <mml:mi>α</mml:mi> <mml:mi>β</mml:mi></mml:mrow></mml:math></alternatives></inline-formula>. The posterior mean estimate is therefore <inline-formula id="pcbi.1007990.e025"><alternatives><graphic id="pcbi.1007990.e025g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e025" xlink:type="simple"/><mml:math display="inline" id="M25"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>t</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>t</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:msub><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>t</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>. Applying Bayes formula and marginalising yields the posterior predictive distribution of the number of infecteds at <italic>s</italic> + 1 as
<disp-formula id="pcbi.1007990.e026"><alternatives><graphic id="pcbi.1007990.e026g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e026" xlink:type="simple"/><mml:math display="block" id="M26"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo> <mml:mo>=</mml:mo> <mml:msubsup><mml:mo>∫</mml:mo> <mml:mrow><mml:mn>0</mml:mn></mml:mrow> <mml:mi>∞</mml:mi></mml:msubsup> <mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>)</mml:mo> <mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mrow><mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo> <mml:mspace width="0.166667em"/><mml:mtext>d</mml:mtext> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(2)</label></disp-formula></p>
<p>In <xref ref-type="disp-formula" rid="pcbi.1007990.e026">Eq (2)</xref> we used the conditional independence of future incidence data from the past epi-curve, given the reproduction number to reduce <inline-formula id="pcbi.1007990.e027"><alternatives><graphic id="pcbi.1007990.e027g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e027" xlink:type="simple"/><mml:math display="inline" id="M27"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> to <inline-formula id="pcbi.1007990.e028"><alternatives><graphic id="pcbi.1007990.e028g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e028" xlink:type="simple"/><mml:math display="inline" id="M28"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mo>(</mml:mo> <mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, which expresses the renewal model relation <italic>x</italic> ∼ Poiss(<italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub>Λ<sub><italic>s</italic>+1</sub>). As Λ<sub><italic>s</italic>+1</sub> only depends on <inline-formula id="pcbi.1007990.e029"><alternatives><graphic id="pcbi.1007990.e029g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e029" xlink:type="simple"/><mml:math display="inline" id="M29"><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:math></alternatives></inline-formula> there are no unknowns. Solving using this and <xref ref-type="disp-formula" rid="pcbi.1007990.e021">Eq (1)</xref> gives <inline-formula id="pcbi.1007990.e030"><alternatives><graphic id="pcbi.1007990.e030g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e030" xlink:type="simple"/><mml:math display="inline" id="M30"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msub><mml:mi>ϕ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msubsup><mml:mi>ϕ</mml:mi> <mml:mn>2</mml:mn> <mml:mrow><mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> where
<disp-formula id="pcbi.1007990.e031"><alternatives><graphic id="pcbi.1007990.e031g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e031" xlink:type="simple"/><mml:math display="block" id="M31"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd columnalign="left"><mml:mrow><mml:msub><mml:mi>ϕ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mo>≔</mml:mo> <mml:msubsup><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>x</mml:mi></mml:msubsup> <mml:msubsup><mml:mo>∫</mml:mo> <mml:mrow><mml:mn>0</mml:mn></mml:mrow> <mml:mi>∞</mml:mi></mml:msubsup> <mml:msubsup><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mrow><mml:mi>x</mml:mi> <mml:mo>+</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msubsup> <mml:msup><mml:mi>e</mml:mi> <mml:mrow><mml:mo>-</mml:mo> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>(</mml:mo> <mml:msub><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>+</mml:mo> <mml:mfrac><mml:mn>1</mml:mn> <mml:msub><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow></mml:msup> <mml:mspace width="0.166667em"/><mml:mtext>d</mml:mtext> <mml:msub><mml:mi>R</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mspace width="4pt"/><mml:mtext>and</mml:mtext></mml:mrow></mml:mtd></mml:mtr> <mml:mtr><mml:mtd/><mml:mtd columnalign="left"><mml:mrow><mml:msub><mml:mi>ϕ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mo>≔</mml:mo> <mml:msubsup><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:msubsup> <mml:mo>Γ</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>Γ</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mtext>,</mml:mtext> <mml:mspace width="4pt"/><mml:mtext>with</mml:mtext> <mml:mspace width="4pt"/><mml:mo>Γ</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>z</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>≔</mml:mo> <mml:msubsup><mml:mo>∫</mml:mo> <mml:mn>0</mml:mn> <mml:mi>∞</mml:mi></mml:msubsup> <mml:msup><mml:mi>θ</mml:mi> <mml:mrow><mml:mi>z</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msup> <mml:msup><mml:mi>e</mml:mi> <mml:mrow><mml:mo>-</mml:mo> <mml:mi>z</mml:mi></mml:mrow></mml:msup> <mml:mspace width="0.166667em"/><mml:mtext>d</mml:mtext> <mml:mi>θ</mml:mi> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(3)</label></disp-formula></p>
<p>Since <inline-formula id="pcbi.1007990.e032"><alternatives><graphic id="pcbi.1007990.e032g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e032" xlink:type="simple"/><mml:math display="inline" id="M32"><mml:mrow><mml:mo>∫</mml:mo><mml:msup><mml:mi>θ</mml:mi><mml:mrow><mml:mi>z</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi>y</mml:mi><mml:mi>θ</mml:mi></mml:mrow></mml:msup><mml:mtext>d</mml:mtext><mml:mi>θ</mml:mi><mml:mo>≡</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mrow><mml:mo>Γ</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mi>y</mml:mi><mml:mi>z</mml:mi></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></alternatives></inline-formula> the integral in <italic>ϕ</italic><sub>1</sub> simplifies to <inline-formula id="pcbi.1007990.e033"><alternatives><graphic id="pcbi.1007990.e033g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e033" xlink:type="simple"/><mml:math display="inline" id="M33"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Γ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow><mml:mi>τ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mo>Λ</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mn>1</mml:mn><mml:mrow><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>τ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow><mml:mi>τ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:math></alternatives></inline-formula>. Some algebra then reveals the negative binomial (NB) distribution:
<disp-formula id="pcbi.1007990.e034"><alternatives><graphic id="pcbi.1007990.e034g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e034" xlink:type="simple"/><mml:math display="block" id="M34"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>∼</mml:mo> <mml:mtext>NB</mml:mtext> <mml:mo>(</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:msub><mml:mi>p</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>≔</mml:mo> <mml:mfrac><mml:mrow><mml:msub><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:msub><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow> <mml:mrow><mml:mn>1</mml:mn> <mml:mo>+</mml:mo> <mml:msub><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:msub><mml:mi>β</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mfrac> <mml:mo>)</mml:mo> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(4)</label></disp-formula></p>
<p>If some variable <italic>y</italic> ∼ NB(<italic>α</italic>, <italic>p</italic>) then <inline-formula id="pcbi.1007990.e035"><alternatives><graphic id="pcbi.1007990.e035g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e035" xlink:type="simple"/><mml:math display="inline" id="M35"><mml:mrow><mml:mi mathvariant="double-struck">P</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>y</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>≔</mml:mo> <mml:mo>(</mml:mo> <mml:mfrac linethickness="0pt"><mml:mrow><mml:mi>α</mml:mi> <mml:mo>+</mml:mo> <mml:mi>y</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>y</mml:mi></mml:mfrac> <mml:mo>)</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mn>1</mml:mn> <mml:mo>-</mml:mo> <mml:mi>p</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mi>α</mml:mi></mml:msup> <mml:msup><mml:mi>p</mml:mi> <mml:mi>y</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> and <inline-formula id="pcbi.1007990.e036"><alternatives><graphic id="pcbi.1007990.e036g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e036" xlink:type="simple"/><mml:math display="inline" id="M36"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi><mml:mrow><mml:mo>[</mml:mo> <mml:mi>y</mml:mi> <mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mrow><mml:mi>p</mml:mi><mml:mi>α</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></alternatives></inline-formula>. <xref ref-type="disp-formula" rid="pcbi.1007990.e034">Eq (4)</xref> is a key result and relates to a framework developed in [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>]. It completely describes the one-step-ahead prediction uncertainty and has mean <inline-formula id="pcbi.1007990.e037"><alternatives><graphic id="pcbi.1007990.e037g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e037" xlink:type="simple"/><mml:math display="inline" id="M37"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi> <mml:mrow><mml:mo>[</mml:mo> <mml:mi>x</mml:mi> <mml:mo>]</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mo>Λ</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>, and variance <inline-formula id="pcbi.1007990.e038"><alternatives><graphic id="pcbi.1007990.e038g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e038" xlink:type="simple"/><mml:math display="inline" id="M38"><mml:mrow><mml:mtext>var</mml:mtext> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mn>1</mml:mn> <mml:mo>+</mml:mo> <mml:mfrac><mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mrow><mml:msub><mml:mi>i</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>+</mml:mo> <mml:mi>a</mml:mi></mml:mrow></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula>. These relations explicate how the current estimate of <italic>R</italic><sub><italic>τ</italic></sub>(<italic>s</italic>) influences our ability to predict upcoming incidence points. At <italic>s</italic> = <italic>t</italic> the above expressions yield prediction statistics for the next (unobserved) time-point beyond the present.</p>
<p>The APE metric is generally defined as <inline-formula id="pcbi.1007990.e039"><alternatives><graphic id="pcbi.1007990.e039g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e039" xlink:type="simple"/><mml:math display="inline" id="M39"><mml:mrow><mml:msub><mml:mtext>APE</mml:mtext><mml:mi>k</mml:mi></mml:msub> <mml:mo>≔</mml:mo> <mml:msubsup><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mi>t</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msubsup> <mml:mo>-</mml:mo> <mml:mo form="prefix">log</mml:mo> <mml:mi mathvariant="double-struck">P</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula> [<xref ref-type="bibr" rid="pcbi.1007990.ref013">13</xref>] (see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>) and is controlled by the shape of the posterior predictive distribution over time. We specialise this metric for renewal models and integrate <xref ref-type="disp-formula" rid="pcbi.1007990.e034">Eq (4)</xref> into the algorithm of <xref ref-type="fig" rid="pcbi.1007990.g001">Fig 1</xref> to derive
<disp-formula id="pcbi.1007990.e040"><alternatives><graphic id="pcbi.1007990.e040g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e040" xlink:type="simple"/><mml:math display="block" id="M40"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:msub><mml:mtext>APE</mml:mtext> <mml:mi>k</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mi>t</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:munderover> <mml:msub><mml:mi>B</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>+</mml:mo> <mml:msub><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo form="prefix">log</mml:mo> <mml:msub><mml:mi>p</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>+</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo form="prefix">log</mml:mo> <mml:mrow><mml:mo>(</mml:mo> <mml:mn>1</mml:mn> <mml:mo>-</mml:mo> <mml:msub><mml:mi>p</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(5)</label></disp-formula>
with <italic>I</italic><sub><italic>s</italic>+1</sub> as the (true) observed incidence at time <italic>s</italic> + 1, which is evaluated within the context of the predictive space of <italic>x</italic>, and <inline-formula id="pcbi.1007990.e041"><alternatives><graphic id="pcbi.1007990.e041g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e041" xlink:type="simple"/><mml:math display="inline" id="M41"><mml:mrow><mml:msub><mml:mi>B</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>≔</mml:mo> <mml:mo form="prefix">log</mml:mo> <mml:mo>(</mml:mo> <mml:mfrac linethickness="0pt"><mml:mrow><mml:msub><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mo>+</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:msub><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>. <xref ref-type="disp-formula" rid="pcbi.1007990.e040">Eq (5)</xref> is exact, easy to evaluate and offers a simple and direct means of finding the optimal data-justified window length as <italic>k</italic>* ≔ arg min<sub><italic>k</italic></sub>APE<sub><italic>k</italic></sub>.</p>
<p>The summation in <xref ref-type="disp-formula" rid="pcbi.1007990.e040">Eq (5)</xref> also clarifies why our approach is useful for real-time applications. As an epidemic unfolds and more data accumulate the upper limit on <italic>s</italic> will increase. We can update APE<sub><italic>k</italic></sub> to account for emerging data by simply including an additional term for each new data point on top of the previously calculated sum. For example, a new datum at <italic>t</italic> + 1 requires the addition of <italic>B</italic><sub><italic>t</italic>+1</sub> + <italic>I</italic><sub><italic>t</italic>+1</sub> log <italic>p</italic><sub><italic>τ</italic>(<italic>t</italic>)</sub> + <italic>α</italic><sub><italic>τ</italic>(<italic>t</italic>)</sub> log (1 − <italic>p</italic><sub><italic>τ</italic>(<italic>t</italic>)</sub>) to <xref ref-type="disp-formula" rid="pcbi.1007990.e040">Eq (5)</xref>. In the Results section we will demonstrate this by successively computing <inline-formula id="pcbi.1007990.e042"><alternatives><graphic id="pcbi.1007990.e042g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e042" xlink:type="simple"/><mml:math display="inline" id="M42"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula>, the optimal window length for data up to time <italic>s</italic>.</p>
<p>APE<sub><italic>k</italic></sub> can also be computed using the in-built NB routines of many software (some parametrise this distribution differently so that <xref ref-type="disp-formula" rid="pcbi.1007990.e034">Eq (4)</xref> may need to be implemented as NB(<italic>α</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub>, 1 − <italic>p</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub>)). When computing <xref ref-type="disp-formula" rid="pcbi.1007990.e040">Eq (5)</xref> directly, the most difficult term is <italic>B</italic><sub><italic>s</italic>+1</sub> when <italic>I</italic><sub><italic>s</italic>+1</sub> and <italic>α</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> are large. In these cases Stirling approximations may be applied. We provide Matlab and R implementations of our renewal model APE metric at <ext-link ext-link-type="uri" xlink:href="https://github.com/kpzoo/model-selection-for-epidemic-renewal-models" xlink:type="simple">https://github.com/kpzoo/model-selection-for-epidemic-renewal-models</ext-link>.</p>
</sec>
</sec>
<sec id="sec005" sec-type="results">
<title>Results</title>
<sec id="sec006">
<title>Optimal window selection for dynamic outbreaks</title>
<p>We apply our APE metric to select <italic>k</italic>* for several epidemic examples, which examine reproduction number profiles featuring stable (<xref ref-type="fig" rid="pcbi.1007990.g002">Fig 2A</xref>) and seasonally fluctuating (<xref ref-type="fig" rid="pcbi.1007990.g002">Fig 2B</xref>) transmission, as well as exponential (<xref ref-type="fig" rid="pcbi.1007990.g003">Fig 3A</xref>) and step-changing (<xref ref-type="fig" rid="pcbi.1007990.g003">Fig 3B</xref>)) control measures. These examples explore dynamically diverse epidemic scenarios and consider both large (Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2A</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3A</xref>) and small outbreaks (Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2B</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3B</xref>). Small outbreaks are fundamentally more difficult to estimate. We simulated epidemics for <italic>t</italic> = 200 days using a generation time distribution analogous to that used for Ebola virus disease predictions in [<xref ref-type="bibr" rid="pcbi.1007990.ref015">15</xref>]. We investigated a window search space of <inline-formula id="pcbi.1007990.e043"><alternatives><graphic id="pcbi.1007990.e043g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e043" xlink:type="simple"/><mml:math display="inline" id="M43"><mml:mrow><mml:mn>2</mml:mn><mml:mo>≤</mml:mo><mml:mi>k</mml:mi><mml:mo>≤</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mi>t</mml:mi><mml:mn>2</mml:mn></mml:mfrac></mml:mstyle></mml:mrow></mml:math></alternatives></inline-formula> and computed the APE at each <italic>k</italic>, over the epidemic duration (1 ≤ <italic>s</italic> ≤ <italic>t</italic>). Reproduction number estimates, <inline-formula id="pcbi.1007990.e044"><alternatives><graphic id="pcbi.1007990.e044g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e044" xlink:type="simple"/><mml:math display="inline" id="M44"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula>, and demographic incidence predictions <inline-formula id="pcbi.1007990.e045"><alternatives><graphic id="pcbi.1007990.e045g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e045" xlink:type="simple"/><mml:math display="inline" id="M45"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>, are presented in Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3</xref>. Predictions over the first <italic>s</italic> &lt; <italic>k</italic> times use all <italic>s</italic> data points.</p>
<fig id="pcbi.1007990.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Selection for stable and fluctuating epidemics.</title>
<p>Left graphs compare <inline-formula id="pcbi.1007990.e046"><alternatives><graphic id="pcbi.1007990.e046g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e046" xlink:type="simple"/><mml:math display="inline" id="M46"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates (blue with 95% confidence intervals) at the APE window length <italic>k</italic>* to those when <italic>k</italic> is set to its upper and lower limits. Right graphs give corresponding one-step-ahead predictions <inline-formula id="pcbi.1007990.e047"><alternatives><graphic id="pcbi.1007990.e047g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e047" xlink:type="simple"/><mml:math display="inline" id="M47"><mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mi>s</mml:mi></mml:msub></mml:math></alternatives></inline-formula> given the window of data <italic>I</italic><sub><italic>τ</italic>(<italic>s</italic>−1)</sub> (blue with 95% prediction intervals). Dashed lines are the true <italic>R</italic><sub><italic>s</italic></sub> numbers (left) and dots are the true <italic>I</italic><sub><italic>s</italic></sub> counts (right). The panels examine (A) stable (constant) and (B) periodically varying changes in <italic>R</italic><sub><italic>s</italic></sub>.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g002" xlink:type="simple"/>
</fig>
<fig id="pcbi.1007990.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Selection for epidemics with interventions.</title>
<p>Left graphs compare <inline-formula id="pcbi.1007990.e048"><alternatives><graphic id="pcbi.1007990.e048g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e048" xlink:type="simple"/><mml:math display="inline" id="M48"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates (blue with 95% confidence intervals) at the APE window length <italic>k</italic>* to those when <italic>k</italic> is set to its upper or lower limits. Right graphs give corresponding one-step-ahead predictions <inline-formula id="pcbi.1007990.e049"><alternatives><graphic id="pcbi.1007990.e049g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e049" xlink:type="simple"/><mml:math display="inline" id="M49"><mml:msub><mml:mover accent="true"><mml:mi>I</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mi>s</mml:mi></mml:msub></mml:math></alternatives></inline-formula> given the window of data <italic>I</italic><sub><italic>τ</italic>(<italic>s</italic>−1)</sub> (blue with 95% prediction intervals). Dashed lines are the true <italic>R</italic><sub><italic>s</italic></sub> numbers (left) and dots are the true <italic>I</italic><sub><italic>s</italic></sub> counts (right). The panels examine (A) exponentially rising and decaying and (B) piecewise falling <italic>R</italic><sub><italic>s</italic></sub> due to differing interventions.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g003" xlink:type="simple"/>
</fig>
<p>We find that the APE metric balances <inline-formula id="pcbi.1007990.e050"><alternatives><graphic id="pcbi.1007990.e050g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e050" xlink:type="simple"/><mml:math display="inline" id="M50"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimate accuracy against one-step-ahead predictive coverage (i.e. the proportion of time that the true <italic>I</italic><sub><italic>s</italic>+1</sub> lies within the 95% prediction intervals of <inline-formula id="pcbi.1007990.e051"><alternatives><graphic id="pcbi.1007990.e051g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e051" xlink:type="simple"/><mml:math display="inline" id="M51"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>) for a range of <italic>R</italic><sub><italic>s</italic></sub> dynamics (top graphs of Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3</xref>). This dual optimisation is central to this work. Moreover, vastly different <italic>I</italic><sub><italic>s</italic>+1</sub> predictions and <inline-formula id="pcbi.1007990.e052"><alternatives><graphic id="pcbi.1007990.e052g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e052" xlink:type="simple"/><mml:math display="inline" id="M52"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates can result when <italic>k</italic> is misspecified (middle and bottom graphs). This can be especially misleading when attempting to identify the significant changes in epidemic transmissibility and can support substantially different beliefs about the infection population biology. Estimation and prediction performance also depend on the actual incidence at any time; small <italic>I</italic><sub><italic>s</italic></sub> values lead to wider confidence intervals in <inline-formula id="pcbi.1007990.e053"><alternatives><graphic id="pcbi.1007990.e053g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e053" xlink:type="simple"/><mml:math display="inline" id="M53"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> at any <italic>k</italic> (see Eq. (S3) of the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>).</p>
<p>When <italic>k</italic> is unjustifiably large not only do we observe systematic prediction and estimation errors, but alarmingly, we tend to be overconfident in them. When <italic>k</italic> is too small, we infer rapidly and randomly fluctuating <inline-formula id="pcbi.1007990.e054"><alternatives><graphic id="pcbi.1007990.e054g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e054" xlink:type="simple"/><mml:math display="inline" id="M54"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> values, which sometimes can deceptively underlie reasonably looking incidence predictions. Consequently, optimal <italic>k</italic>-selection is integral for trustworthy inference and prediction. Observe that small <italic>k</italic>, which implies a more complex renewal model (i.e. there are more parameters to be inferred), does not generally result in better causally predictive one-step-ahead incidence predictions. Had we instead naively picked the <italic>k</italic> that best fits the existing epi-curve, then the smallest <italic>k</italic> would always be favoured (overfitting) [<xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>].</p>
<p>We emphasize and validate the predictive performance of APE in <xref ref-type="fig" rid="pcbi.1007990.g004">Fig 4</xref>. This shows, for all simulated examples above that minimising the APE also approximately minimises the percentage of the time <italic>s</italic> ≤ <italic>t</italic> that true incidence values fall outside the 95% prediction intervals of <xref ref-type="disp-formula" rid="pcbi.1007990.e034">Eq (4)</xref>. This validates the APE approach and evidences its proficiency at optimising for short-term forecasting accuracy in real time. We recommend using APE to define <italic>k</italic>* for an existing epi-curve up to the present <italic>t</italic>. Applying <xref ref-type="disp-formula" rid="pcbi.1007990.e034">Eq (4)</xref> with this <italic>k</italic>* should then best predict the number of cases on the (<italic>t</italic> + 1)<sup>th</sup> day. This entire procedure should be repeated for later forecasts, with <italic>k</italic>* being progressively updated. The real-time behaviour of <italic>k</italic>* as data accumulate over the course of an outbreak is investigated in the next section.</p>
<fig id="pcbi.1007990.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g004</object-id>
<label>Fig 4</label>
<caption>
<title>APE prediction accuracy.</title>
<p>We compare the APE metric (blue, left y axes) to the percentage of true incidence values, <italic>I</italic><sub><italic>s</italic>+1</sub> that fall outside the 95% prediction intervals of <inline-formula id="pcbi.1007990.e055"><alternatives><graphic id="pcbi.1007990.e055g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e055" xlink:type="simple"/><mml:math display="inline" id="M55"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> (red, right y axes) for various window sizes <italic>k</italic>. The dashed line is <italic>k</italic>*. Panels correspond to those of Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3</xref>.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g004" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec007">
<title>Real-time performance for emerging outbreaks</title>
<p>During an unfolding epidemic, renewal models can support response efforts by providing real-time estimates of infectious disease transmissibility and short-term forecasts of the expected incidence of cases [<xref ref-type="bibr" rid="pcbi.1007990.ref009">9</xref>]. Here we show how the APE metric can be used within this framework as a rapid and reliable real-time tool. We focus on the key problem of diagnosing and verifying the efficacy of implemented control measures over the course of an emerging outbreak [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>]. Knowing whether an intervention is working or not can inform preparedness and resource allocation. Estimating whether <italic>R</italic><sub><italic>s</italic></sub> &lt; 1 or <italic>R</italic><sub><italic>s</italic></sub> ≥ 1 is one simple means of assessing efficacy. Effective control measures lead to the former.</p>
<p>We investigate how the APE metric responds in real time to swift swings in reproduction number and quantify our results over 10<sup>3</sup> simulated epi-curves with <italic>t</italic> = 150 days. We focus on step-changes in <italic>R</italic><sub><italic>s</italic></sub> and examine how the successively optimal window length, denoted <inline-formula id="pcbi.1007990.e056"><alternatives><graphic id="pcbi.1007990.e056g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e056" xlink:type="simple"/><mml:math display="inline" id="M56"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> when computed with data up to time <italic>s</italic>, responds. If <inline-formula id="pcbi.1007990.e057"><alternatives><graphic id="pcbi.1007990.e057g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e057" xlink:type="simple"/><mml:math display="inline" id="M57"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> is sensitive to these changes with minimum delay then it is likely a dependable means of diagnosing control efficacy in real time. In previous sections we have been computing <inline-formula id="pcbi.1007990.e058"><alternatives><graphic id="pcbi.1007990.e058g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e058" xlink:type="simple"/><mml:math display="inline" id="M58"><mml:mrow><mml:msup><mml:mi>k</mml:mi> <mml:mo>*</mml:mo></mml:msup> <mml:mo>=</mml:mo> <mml:msubsup><mml:mi>k</mml:mi> <mml:mi>t</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>. We consider four models and present our main results in Figs <xref ref-type="fig" rid="pcbi.1007990.g005">5</xref> and <xref ref-type="fig" rid="pcbi.1007990.g006">6</xref>.</p>
<fig id="pcbi.1007990.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g005</object-id>
<label>Fig 5</label>
<caption>
<title>Real-time APE sensitivity to increasing transmission.</title>
<p>We simulate 10<sup>3</sup> independent epi-curves under renewal models with sharply (A) increasing and (B) recovering epidemics. Top graphs give the true (green) and predicted (blue) incidence ranges, the middle ones provide estimates of <italic>R</italic><sub><italic>s</italic></sub> under the final <inline-formula id="pcbi.1007990.e059"><alternatives><graphic id="pcbi.1007990.e059g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e059" xlink:type="simple"/><mml:math display="inline" id="M59"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>t</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> and the bottom graphs illustrate how successive <inline-formula id="pcbi.1007990.e060"><alternatives><graphic id="pcbi.1007990.e060g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e060" xlink:type="simple"/><mml:math display="inline" id="M60"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> choices from APE vary across time and are sensitive to real-time elevations in transmission.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g005" xlink:type="simple"/>
</fig>
<fig id="pcbi.1007990.g006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g006</object-id>
<label>Fig 6</label>
<caption>
<title>Real-time APE sensitivity to rapid epidemic control.</title>
<p>We simulate 10<sup>3</sup> independent epi-curves under renewal models with (A) one effective and (B) two partially effective interventions. Top graphs give the true (green) and predicted (blue) incidence ranges, the middle ones provide estimates of <italic>R</italic><sub><italic>s</italic></sub> under the final <inline-formula id="pcbi.1007990.e061"><alternatives><graphic id="pcbi.1007990.e061g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e061" xlink:type="simple"/><mml:math display="inline" id="M61"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>t</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> and the bottom graphs illustrate how successive <inline-formula id="pcbi.1007990.e062"><alternatives><graphic id="pcbi.1007990.e062g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e062" xlink:type="simple"/><mml:math display="inline" id="M62"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> choices from APE can reliably and rapidly detect the impact of real-time control actions.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g006" xlink:type="simple"/>
</fig>
<p>In <xref ref-type="fig" rid="pcbi.1007990.g005">Fig 5</xref> we examine cases of increasing transmission for (A) an outbreak that is uncontrolled and worsening (e.g. if the infectious disease acquires an additional route of spread) and (B) an ineffectively controlled epidemic with a late-stage elevation in <italic>R</italic><sub><italic>s</italic></sub> (e.g. if quarantine measures were relaxed too quickly). <xref ref-type="fig" rid="pcbi.1007990.g006">Fig 6</xref> then investigates scenarios involving effective interventions where (A) rapid control is employed (e.g. if a wide-scale lockdown is enacted) and (B) the initial action is only partially effective and so later additional controls are needed (e.g. if flight restrictions were first applied but then further transport shutdowns or mobility restrictions were required).</p>
<p>In both figures, top graphs overlay true <italic>I</italic><sub><italic>s</italic>+1</sub> and predicted <inline-formula id="pcbi.1007990.e063"><alternatives><graphic id="pcbi.1007990.e063g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e063" xlink:type="simple"/><mml:math display="inline" id="M63"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mi>k</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> across time and middle ones show the best <inline-formula id="pcbi.1007990.e064"><alternatives><graphic id="pcbi.1007990.e064g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e064" xlink:type="simple"/><mml:math display="inline" id="M64"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates under the final <inline-formula id="pcbi.1007990.e065"><alternatives><graphic id="pcbi.1007990.e065g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e065" xlink:type="simple"/><mml:math display="inline" id="M65"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>t</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula>. We see that the APE-selected model properly distinguishes significant changes from stable periods in all scenarios. The model in <xref ref-type="fig" rid="pcbi.1007990.g005">Fig 5A</xref> is the most difficult to infer as the increasing <italic>R</italic><sub><italic>s</italic></sub> does not visibly change the shape of the incidence population curve. Bottom graphs show <inline-formula id="pcbi.1007990.e066"><alternatives><graphic id="pcbi.1007990.e066g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e066" xlink:type="simple"/><mml:math display="inline" id="M66"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> sequentially across the ongoing epidemic. Here we observe how the APE metric uses data to justify its choices in real time. As data accumulate under a stable reproduction number APE increases <italic>k</italic>*. This makes sense as there is increasing support for stationary epidemic behaviour. This is especially obvious in <xref ref-type="fig" rid="pcbi.1007990.g006">Fig 6A</xref>.</p>
<p>However, on facing a step-change the APE immediately responds by drastically reducing <inline-formula id="pcbi.1007990.e067"><alternatives><graphic id="pcbi.1007990.e067g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e067" xlink:type="simple"/><mml:math display="inline" id="M67"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula> − both in scenarios where <italic>R</italic><sub><italic>s</italic></sub> is changing from above to below 1 and vice versa. This reduction is less visible in <xref ref-type="fig" rid="pcbi.1007990.g005">Fig 5A</xref> because the observed epi-curve is not as dramatically altered as in the other scenarios. This rapid response recommends the APE as a suitable and sensitive real-time diagnostic tool. We dissect the impact of change-times on the actual APE values at different <italic>k</italic> in <xref ref-type="fig" rid="pcbi.1007990.g007">Fig 7</xref>. There we find that when the epi-curve is reasonably stable there is not much difference between the performance at various window lengths and so the APE curves are neighbouring. In contrast, when a non-stationary change occurs there is a clear unravelling of the curves and potentially large gains to be made by optimising <italic>k</italic>. Failing to properly select <italic>k</italic> here would potentially mislead our understanding of the unfolding outbreak. This increases the impetus for formal <italic>k</italic>-selection metrics such as the APE.</p>
<fig id="pcbi.1007990.g007" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g007</object-id>
<label>Fig 7</label>
<caption>
<title>APE scores in real time.</title>
<p>The successive (in time) APE scores for each window length, <italic>k</italic>, are shown in grey. The scores for the smallest and largest <italic>k</italic> are in cyan and magenta respectively. Graphs correspond to the models in Figs <xref ref-type="fig" rid="pcbi.1007990.g005">5</xref> and <xref ref-type="fig" rid="pcbi.1007990.g006">6</xref>. Dashed lines are the change-times of each model. In (5A) the change-time does not significantly affect the epi-curve shape and so the APE scores are close together. In (5B), (6A) and (6B) the change-times notably alter the epi-curve shape so that the choice of <italic>k</italic> becomes critical to performance.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g007" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec008">
<title>Window selection on empirical data</title>
<p>We test our APE approach on two well-studied empirical epidemic datasets: pandemic H1N1 influenza in Baltimore from 1918 [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>] and SARS in Hong Kong from 2003 [<xref ref-type="bibr" rid="pcbi.1007990.ref017">17</xref>]. For each epidemic we extract the incidence curve, total infectiousness and generation time distributions from the EpiEstim R package [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>] and apply the APE metric over <inline-formula id="pcbi.1007990.e068"><alternatives><graphic id="pcbi.1007990.e068g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e068" xlink:type="simple"/><mml:math display="inline" id="M68"><mml:mrow><mml:mn>2</mml:mn><mml:mo>≤</mml:mo><mml:mi>k</mml:mi><mml:mo>≤</mml:mo><mml:mstyle scriptlevel="+1"><mml:mfrac bevelled="true"><mml:mi>t</mml:mi><mml:mn>2</mml:mn></mml:mfrac></mml:mstyle></mml:mrow></mml:math></alternatives></inline-formula> with <italic>t</italic> as the last available incidence time point. We compare the one-step-ahead <italic>I</italic><sub><italic>s</italic>+1</sub> prediction fidelity and the <italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> estimation accuracy obtained from the renewal model under the APE-selected <italic>k</italic>* to that from the model used in [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>], which recommended weekly windows (i.e. <italic>k</italic> = 7) after visually examining several window lengths.</p>
<p>Our main results are in Figs <xref ref-type="fig" rid="pcbi.1007990.g008">8</xref>–<xref ref-type="fig" rid="pcbi.1007990.g010">10</xref>. We benchmarked our estimates against those directly provided by EpiEstim to confirm our implementation and restrict <italic>k</italic> &gt; 1 to avoid model identifiability issues [<xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref018">18</xref>]. Intriguingly, we find <italic>k</italic>* = min <italic>k</italic> = 2 for both datasets. This yields appreciably improved prediction fidelity (i.e. the coverage of observed incidence values by the 95% prediction intervals), relative to the weekly window choice, as seen in <xref ref-type="fig" rid="pcbi.1007990.g008">Fig 8</xref>. Between 8−10% of incidence data points are better covered by using <italic>k</italic>* over <italic>k</italic> = 7. This improvement is apparent in the right graphs of Figs <xref ref-type="fig" rid="pcbi.1007990.g009">9A</xref> and <xref ref-type="fig" rid="pcbi.1007990.g010">10A</xref>, where the <italic>k</italic> = 7 case produces stiffer incidence predictions that cannot properly reproduce the observed epi-curve.</p>
<fig id="pcbi.1007990.g008" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g008</object-id>
<label>Fig 8</label>
<caption>
<title>Empirical prediction accuracy.</title>
<p>We compare the APE metric (dotted blue, left y axis) to the percentage of true incidence values, <italic>I</italic><sub><italic>s</italic>+1</sub>, which fall outside the 95% prediction intervals of <inline-formula id="pcbi.1007990.e069"><alternatives><graphic id="pcbi.1007990.e069g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e069" xlink:type="simple"/><mml:math display="inline" id="M69"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> (dotted red, right y axis) across the window search space <italic>k</italic>. The dashed line gives <italic>k</italic>* (black) and <italic>k</italic> = 7 (grey). The top graph presents results for the influenza 1918 dataset, while the bottom one is for SARS 2003 data. We find that heuristic weekly windows lead to appreciably larger forecasting error than the APE selections.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g008" xlink:type="simple"/>
</fig>
<fig id="pcbi.1007990.g009" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g009</object-id>
<label>Fig 9</label>
<caption>
<title>Selection for pandemic influenza (1918).</title>
<p>Left graphs compare <inline-formula id="pcbi.1007990.e070"><alternatives><graphic id="pcbi.1007990.e070g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e070" xlink:type="simple"/><mml:math display="inline" id="M70"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates (blue with 95% confidence intervals) at optimal APE window length <italic>k</italic>* to weekly sliding windows (<italic>k</italic> = 7), which were recommended in [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. Right graphs give corresponding one-step-ahead predictions <inline-formula id="pcbi.1007990.e071"><alternatives><graphic id="pcbi.1007990.e071g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e071" xlink:type="simple"/><mml:math display="inline" id="M71"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> (blue with 95% prediction intervals). Dashed lines are the <italic>R</italic> = 1 threshold (left) and dots are the true incidence <italic>I</italic><sub><italic>s</italic></sub> (right). Panel (A) directly uses the empirical influenza (1918) data [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>] while (B) smooths outliers in the data as in [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>].</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g009" xlink:type="simple"/>
</fig>
<fig id="pcbi.1007990.g010" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1007990.g010</object-id>
<label>Fig 10</label>
<caption>
<title>Selection for SARS (2003).</title>
<p>Left graphs compare <inline-formula id="pcbi.1007990.e072"><alternatives><graphic id="pcbi.1007990.e072g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e072" xlink:type="simple"/><mml:math display="inline" id="M72"><mml:msub><mml:mover accent="true"><mml:mi>R</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mi>τ</mml:mi> <mml:mo>(</mml:mo> <mml:mi>s</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:msub></mml:math></alternatives></inline-formula> estimates (blue with 95% confidence intervals) at optimal APE window length <italic>k</italic>* to weekly sliding windows (<italic>k</italic> = 7), which were recommended in [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. Right graphs give corresponding one-step-ahead predictions <inline-formula id="pcbi.1007990.e073"><alternatives><graphic id="pcbi.1007990.e073g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e073" xlink:type="simple"/><mml:math display="inline" id="M73"><mml:mrow><mml:mrow><mml:mi>x</mml:mi> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/></mml:mrow> <mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>-</mml:mo> <mml:mi>k</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>s</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> (blue with 95% prediction intervals). Dashed lines are the <italic>R</italic> = 1 threshold (left) and dots are the true incidence <italic>I</italic><sub><italic>s</italic></sub> (right). Panel (A) directly uses the empirical SARS (2003) data [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>] while (B) smooths outliers (with a 5-day moving average) in the data as in [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>].</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.g010" xlink:type="simple"/>
</fig>
<p>Weekly windows misjudge the SARS epidemic peak, predict a multimodal SARS incidence curve that is not reflected by the actual data and systematically underestimate influenza case counts when it matters most (i.e. around the high incidence phase). However, the smaller <italic>k</italic>* window results in noisier <italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> estimates (left graphs of Figs <xref ref-type="fig" rid="pcbi.1007990.g009">9A</xref> and <xref ref-type="fig" rid="pcbi.1007990.g010">10A</xref>, which may be questionable. These rapidly fluctuating reproduction numbers likely motivated the adoption of weekly windows in previous analyses. While the APE-based estimates are indeed more uncertain (a consequence of shorter windows), we argue that they are formally justified by the available data, especially given the weaker predictive capacity of weekly windows. While the noisier <italic>R</italic><sub><italic>τ</italic>(<italic>s</italic>)</sub> estimates do not mislead our understanding of the efficacy of implemented control measures (it is still clear that the influenza epidemic is only partially under control between 40 ≤ <italic>s</italic> ≤ 65 days before recovering, while the SARS outbreak is largely arrested from <italic>s</italic> &gt; 50 days), they likely overestimate the peak transmissibility of these diseases.</p>
<p>We might suspect that certain artefacts of the data could resolve this issue, rendering a more believable combination of estimated reproduction number and predicted incidence. Particularly, the influenza data seem considerably more affected by the smaller look-back windows. These <italic>k</italic>* = 2 windows are needed to help predictions get close to the peak incidence values of the data. However, these peaks seem reminiscent of outliers and in the original analysis of [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>] they were attributed to possible recollection bias in patients that were questioned. Removing these biases might be expected to lead to smoother APE-justified reproduction numbers. We test this hypothesis in Figs <xref ref-type="fig" rid="pcbi.1007990.g009">9B</xref> and <xref ref-type="fig" rid="pcbi.1007990.g010">10B</xref>.</p>
<p>There we find that applying simple 5-day moving average filters to the data, as done in [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>] to ameliorate outliers, still does not support a <italic>k</italic>* &gt; 2 in either scenario, though smoother estimates do result. Weekly windows are still too inflexible to properly predict this averaged incidence. While other signal processing techniques, such as using autocorrelated smoothing prior distributions instead of independent gamma ones over the reproduction numbers, could be applied to further investigate if <italic>k</italic>* &gt; 2 is justifiable we consider this beyond the scope of this work and somewhat biologically unmotivated. Instead we conjecture that these results support the interesting alternative hypothesis that the epi-curve population data are not Poisson distributed. We expand on this conjecture in the subsequent section.</p>
</sec>
</sec>
<sec id="sec009" sec-type="conclusions">
<title>Discussion</title>
<p>Inferring the dynamics of the effective reproduction number, <italic>R</italic><sub><italic>t</italic></sub>, in real time is crucial for forecasting transmissibility and the efficacy of implemented control actions and for assessing the growth of an unfolding infectious disease outbreak [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>]. Renewal models provide a popular platform for prospectively estimating these reproduction numbers, which can then be used to generate incidence projections [<xref ref-type="bibr" rid="pcbi.1007990.ref002">2</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref007">7</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref015">15</xref>]. However, the dependence of these estimates and predictions on the look-back window size, <italic>k</italic>, which determines the renewal model dimensionality, has never been formally investigated.</p>
<p>Previous methods for selecting <italic>k</italic> have generally been heuristic or ad-hoc [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. Here we have devised and validated a new, rigorous, information-theoretic approach for optimising <italic>k</italic> to available incidence data. Our method was founded on deriving an analytical expression for the renewal model posterior predictive distribution. Integrating this into the general APE metric of [<xref ref-type="bibr" rid="pcbi.1007990.ref013">13</xref>] led to <xref ref-type="disp-formula" rid="pcbi.1007990.e040">Eq (5)</xref>, a simple, easily-computed yet theoretically justified metric for <italic>k</italic>-selection that balances estimation fidelity with prediction accuracy.</p>
<p>This APE approach not only accounts for parametric complexity and is formally linked to MDL (and Bayesian model selection − see the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref>), but it also has several desirable properties that render it suitable for handling the eccentricities of real-time infectious disease applications:</p>
<p>(a) Small outbreak sample size. Early-on in an epidemic data are scarce and uncertainty is large. The APE approach is valid and computable at all sample sizes at which the renewal model is identifiable and hence is applicable to emerging outbreaks [<xref ref-type="bibr" rid="pcbi.1007990.ref014">14</xref>]. Its emphasis on using the full predictive distribution and its Bayesian formulation allow it to properly account for large uncertainties and to handle different (prior) hypotheses about <italic>R</italic><sub><italic>t</italic></sub>, as the epidemic progresses from small (e.g. onset) to large (e.g. establishment) data regimes.</p>
<p>(b) Non-stationary transmission. Incidence time series can change rapidly across an unfolding outbreak and hence are not independent and identically distributed [<xref ref-type="bibr" rid="pcbi.1007990.ref007">7</xref>]. Instead they are sequential, autocorrelated and possess non-stationary (time-varying) statistics. The APE formally accounts for these properties. Leave-one-out cross validation (CV) is a popular model selection approach that is related to APE. It can be defined as <inline-formula id="pcbi.1007990.e074"><alternatives><graphic id="pcbi.1007990.e074g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e074" xlink:type="simple"/><mml:math display="inline" id="M74"><mml:mrow><mml:mtext>CV</mml:mtext><mml:mo>≔</mml:mo> <mml:msubsup><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mi>t</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msubsup> <mml:mo>-</mml:mo> <mml:mo form="prefix">log</mml:mo> <mml:mi mathvariant="double-struck">P</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msub> <mml:mspace width="0.166667em"/><mml:mo>|</mml:mo> <mml:mspace width="0.166667em"/><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mspace width="0.166667em"/><mml:mi mathvariant="sans-serif">c</mml:mi></mml:mrow></mml:msubsup> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula>, where <inline-formula id="pcbi.1007990.e075"><alternatives><graphic id="pcbi.1007990.e075g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e075" xlink:type="simple"/><mml:math display="inline" id="M75"><mml:msubsup><mml:mi>I</mml:mi> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>+</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mspace width="0.166667em"/><mml:mi mathvariant="sans-serif">c</mml:mi></mml:mrow></mml:msubsup></mml:math></alternatives></inline-formula> means that all incidence points except <italic>I</italic><sub><italic>s</italic>+1</sub> are included. Comparing this to Eq. (S4) of the <xref ref-type="supplementary-material" rid="pcbi.1007990.s001">S1 Text</xref> we see that APE is a modified CV that is explicitly specialised for non-stationary, accumulating time-series.</p>
<p>(c) Lack of a ‘true model’. It is highly unlikely that the underlying <italic>R</italic><sub><italic>t</italic></sub> is truly described by one of the piecewise-constant functions assumed within renewal models. Unlike many model selection criteria, the APE does not require a true model to exist within the set being evaluated [<xref ref-type="bibr" rid="pcbi.1007990.ref012">12</xref>]. It is only interested in finding the model that best predicts the data and so emphasises accurate forecasting and data-justified complexity. However, should a true model exist, the APE is statistically consistent i.e. as data accumulate it will select the data-generating model with probability 1 [<xref ref-type="bibr" rid="pcbi.1007990.ref012">12</xref>].</p>
<p>While we have specialised our results to epidemiology, our method is widely-applicable. Several popular models in macroevolution, phylogeography and genetics, such as the skyline plot, structured coalescent and sequential Markovian coalescent, all possess piecewise-Poisson statistical formulations analogous to the renewal model [<xref ref-type="bibr" rid="pcbi.1007990.ref018">18</xref>]. As shown in [<xref ref-type="bibr" rid="pcbi.1007990.ref011">11</xref>], this leads to an almost plug-and-play usability of MDL-based methodology. Moreover, the renewal model itself can be adapted to other animal and plant ecology problems where the interest is to infer a demographic growth rate from a time-series of species samples [<xref ref-type="bibr" rid="pcbi.1007990.ref004">4</xref>].</p>
<p>We tested our method on both simulated and empirical datasets. We started by examining various distinct, time-varying reproduction number profiles in Figs <xref ref-type="fig" rid="pcbi.1007990.g002">2</xref> and <xref ref-type="fig" rid="pcbi.1007990.g003">3</xref>, for which no true model existed. By comparing the APE-optimised <italic>k</italic>* against long and short window sizes, we found that not only does APE meaningfully balance <italic>R</italic><sub><italic>t</italic></sub> estimates to achieve good prediction capacity, but also that getting <italic>k</italic> wrong could promote strikingly different conclusions about the infection population dynamics from the same dataset. This behaviour held consistent for outbreaks with both small and large infected case numbers, stable and seasonal transmission and under different dynamics of control.</p>
<p>We then investigated several step-changing reproduction number examples in Figs <xref ref-type="fig" rid="pcbi.1007990.g005">5</xref> and <xref ref-type="fig" rid="pcbi.1007990.g006">6</xref> to better expose the underlying mechanics of our method, and to showcase its value as a real-time inference tool for emerging epidemics with strong non-stationary transmission. These examples featured rapid changes, also known as events in information theory, caused by effective and ineffective countermeasures [<xref ref-type="bibr" rid="pcbi.1007990.ref003">3</xref>]. In all cases our metric responded rapidly whilst maintaining reliable, real-time incidence predictions. Such rapid, event-triggered responses (as in <xref ref-type="fig" rid="pcbi.1007990.g007">Fig 7</xref>) are known to be time-efficient [<xref ref-type="bibr" rid="pcbi.1007990.ref019">19</xref>]. A key objective of short-term forecasting is the speedy diagnosis of intervention efficacy. Our results confirmed the practical value of our method towards this objective. Importantly, we found in Figs <xref ref-type="fig" rid="pcbi.1007990.g005">5</xref> and <xref ref-type="fig" rid="pcbi.1007990.g006">6</xref> that the APE metric computed successively and causally during an unfolding outbreak, <inline-formula id="pcbi.1007990.e076"><alternatives><graphic id="pcbi.1007990.e076g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e076" xlink:type="simple"/><mml:math display="inline" id="M76"><mml:msubsup><mml:mi>k</mml:mi> <mml:mi>s</mml:mi> <mml:mo>*</mml:mo></mml:msubsup></mml:math></alternatives></inline-formula>, can quickly and correctly inform on changes in transmission and growth of an epidemic in real time.</p>
<p>While our metric is promising, we caution that much work remains to be done. Standard Poisson renewal models, such as those we have considered here, make several limiting assumptions including that (i) all cases are detected (i.e. there is no significant sampling bias), (ii) the serial interval and generation time distribution coincide and do not change over the epidemic lifetime and (iii) that heterogeneities in transmission within the infected population have negligible effect [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]. Our analysis of H1N1 influenza (1918) and SARS (2003) flagged some of these concerns. Comparing the <italic>k</italic>* to previously recommended weekly windows from [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>], as in Figs <xref ref-type="fig" rid="pcbi.1007990.g009">9</xref> and <xref ref-type="fig" rid="pcbi.1007990.g010">10</xref>, led to some interesting revelations.</p>
<p>Our APE approach selected a notably shorter window (<italic>k</italic>* = 2 days) for both datasets. While this produced noisy <italic>R</italic><sub><italic>s</italic></sub> estimates that seemed less reliable than those obtained with weekly windows, the predicted incidence values were central to understanding this discrepancy. The weekly windows were considerably worse at forecasting the observed epi-curve, often systematically biased around the peak of the epidemic and sometimes predicting multimodal curves that were not reflected by the existing data. From a model selection perspective (among all EpiEstim-type models [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>]), the shorter windows are therefore justified.</p>
<p>However, the noisy <italic>R</italic><sub><italic>t</italic></sub> estimates are still undesirable. Both datasets are known to potentially contain super-spreading heterogeneities and other biases that may lead to outliers [<xref ref-type="bibr" rid="pcbi.1007990.ref010">10</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>]. Moving-average filters were applied in [<xref ref-type="bibr" rid="pcbi.1007990.ref016">16</xref>] to remove these artefacts from the H1N1 data. We used the same technique to regularise both datasets and then re-applied the APE. Results remained consistent, suggesting that if a Poisson model is valid then <italic>k</italic>* = 2 is indeed the correct window choice. It may also be that the mean-variance equality of Poisson models is too restrictive for these datasets and so APE compensated for this inflexibility with short windows. A negative binomial renewal model where <inline-formula id="pcbi.1007990.e077"><alternatives><graphic id="pcbi.1007990.e077g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1007990.e077" xlink:type="simple"/><mml:math display="inline" id="M77"><mml:mrow><mml:msub><mml:mi>I</mml:mi> <mml:mi>t</mml:mi></mml:msub> <mml:mo>∼</mml:mo> <mml:mtext>NB</mml:mtext> <mml:mo>(</mml:mo> <mml:mi>κ</mml:mi> <mml:mo>,</mml:mo> <mml:mspace width="0.166667em"/><mml:mfrac><mml:mrow><mml:msub><mml:mo>Λ</mml:mo> <mml:mi>t</mml:mi></mml:msub> <mml:msub><mml:mi>R</mml:mi> <mml:mi>t</mml:mi></mml:msub></mml:mrow> <mml:mrow><mml:msub><mml:mo>Λ</mml:mo> <mml:mi>t</mml:mi></mml:msub> <mml:msub><mml:mi>R</mml:mi> <mml:mi>t</mml:mi></mml:msub> <mml:mo>+</mml:mo> <mml:mi>κ</mml:mi></mml:mrow></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>, with <italic>κ</italic> as a noise parameter, would relax this equality yet still embody the Euler-Lotka dynamics of the original model. The extension of APE to these generalised models will form an upcoming study.</p>
<p>Real-time, model-based epidemic forecasts are quickly becoming an integral prognostic for resource allocation and strategic intervention planning [<xref ref-type="bibr" rid="pcbi.1007990.ref020">20</xref>]. As global infectious disease threats elevate, model-supported predictions, which can inform decision making as an outbreak progresses, have become imperative. In the ongoing COVID-19 pandemic, for example, modelling is being extensively used to inform policy. However, much is still unknown about the fundamentals of epidemic prediction and this uncertainty has inspired some reluctance in public health applications [<xref ref-type="bibr" rid="pcbi.1007990.ref020">20</xref>]. As a result, recent studies have called for careful evaluation of the predictive capacity of models, and demonstrated the importance of having standard metrics to compare models and qualify their uncertainties [<xref ref-type="bibr" rid="pcbi.1007990.ref021">21</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref022">22</xref>].</p>
<p>Here we have presented one such metric for use in real time. Short-term forecasts aid immediate decision making and are more actionable than long-term ones (which are less reliable) [<xref ref-type="bibr" rid="pcbi.1007990.ref020">20</xref>, <xref ref-type="bibr" rid="pcbi.1007990.ref021">21</xref>]. While we focussed on renewal models, the general APE metric (see <xref ref-type="fig" rid="pcbi.1007990.g001">Fig 1</xref>) can select among diverse model types and facilitate comparisons of their short-term predictive capacity and reliability. Its direct use of posterior predictive distributions within an honest, causal framework allows proper and problem-specific inclusion of uncertainty [<xref ref-type="bibr" rid="pcbi.1007990.ref014">14</xref>]. Common metrics such as the mean absolute error lack this adaptation [<xref ref-type="bibr" rid="pcbi.1007990.ref021">21</xref>]. Given its information-theoretic links and demonstrated performance we hope that our APE-based method can serve as a rigorous benchmark for model-based forecasts and help refine ongoing estimates of transmission as outbreaks unfold.</p>
</sec>
<sec id="sec010">
<title>Supporting information</title>
<supplementary-material id="pcbi.1007990.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.s001" xlink:type="simple">
<label>S1 Text</label>
<caption>
<title>Further details and more general definitions and equations under sub-headings: Epidemic renewal models and Prospective model selection.</title>
<p>(PDF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="pcbi.1007990.ref001">
<label>1</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Anderson</surname> <given-names>R</given-names></name> and <name name-style="western"><surname>May</surname> <given-names>R</given-names></name>. <source><italic>Infectious diseases of humans: dynamics and control</italic></source>. <publisher-name>Oxford University Press</publisher-name>; <year>1991</year>.</mixed-citation>
</ref>
<ref id="pcbi.1007990.ref002">
<label>2</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<collab>WHO Ebola Response Team</collab>. <article-title>Ebola virus disease in West Africa—the first 9 months of the epidemic and forward projections</article-title>. <source><italic>N. Engl. J. Med</italic></source>. <year>2014</year>;<volume>371</volume>(<issue>16</issue>):<fpage>1481</fpage>–<lpage>95</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1056/NEJMoa1411100" xlink:type="simple">10.1056/NEJMoa1411100</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref003">
<label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Cauchemez</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Boelle</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Thomas</surname> <given-names>G</given-names></name>, and <name name-style="western"><surname>Valleron</surname> <given-names>A</given-names></name>. <article-title>Estimating in real time the efficacy of measures to control emerging communicable diseases</article-title>. <source><italic>Am. J. Epidemiol</italic></source>. <year>2006</year>;<volume>164</volume>(<issue>6</issue>):<fpage>591</fpage>–<lpage>7</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/aje/kwj274" xlink:type="simple">10.1093/aje/kwj274</ext-link></comment> <object-id pub-id-type="pmid">16887892</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref004">
<label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Cortes</surname> <given-names>E</given-names></name>. <article-title>Perspectives on the intrinsic rate of population growth</article-title>. <source><italic>Methods Ecol. Evol</italic></source>. <year>2016</year>;<volume>7</volume>:<fpage>1136</fpage>–<lpage>45</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1111/2041-210X.12592" xlink:type="simple">10.1111/2041-210X.12592</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref005">
<label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Fraser</surname> <given-names>C</given-names></name>. <article-title>Estimating individual and household reproduction numbers in an emerging epidemic</article-title>. <source><italic>PLOS One</italic></source>. <year>2007</year>;<volume>8</volume>:<fpage>e758</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0000758" xlink:type="simple">10.1371/journal.pone.0000758</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref006">
<label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Wallinga</surname> <given-names>J</given-names></name> and <name name-style="western"><surname>Lipsitch</surname> <given-names>M</given-names></name>. <article-title>How generation intervals shape the relationship between growth rates and reproductive numbers</article-title>. <source><italic>Proc. R. Soc. B</italic></source>. <year>2007</year>;<volume>274</volume>:<fpage>599</fpage>–<lpage>604</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1098/rspb.2006.3754" xlink:type="simple">10.1098/rspb.2006.3754</ext-link></comment> <object-id pub-id-type="pmid">17476782</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref007">
<label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Fraser</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Cummings</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Klinkenberg</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Burke</surname> <given-names>D</given-names></name> and <name name-style="western"><surname>Ferguson</surname> <given-names>N</given-names></name>. <article-title>Influenza transmission in households during the 1918 pandemic</article-title>. <source><italic>Am. J. Epidemiol</italic></source>. <year>2011</year>;<volume>174</volume>(<issue>5</issue>):<fpage>505</fpage>–<lpage>14</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/aje/kwr122" xlink:type="simple">10.1093/aje/kwr122</ext-link></comment> <object-id pub-id-type="pmid">21749971</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref008">
<label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ferguson</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Cucunba</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Dorigatti</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Nedjati-Gilani</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Donnelly</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Basanez</surname> <given-names>M</given-names></name>, <etal>et al</etal>. <article-title>Countering the Zika epidemic in Latin America</article-title>. <source><italic>Science</italic></source>. <year>2016</year>;<volume>353</volume>(<issue>6297</issue>):<fpage>353</fpage>–<lpage>4</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1126/science.aag0219" xlink:type="simple">10.1126/science.aag0219</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref009">
<label>9</label>
<mixed-citation publication-type="other" xlink:type="simple">S Bhatia, A Cori, K Parag, S Mishra, L Cooper, K Ainslie, et al. Short-term forecasts of COVID-19 deaths in multiple countries. <italic>MRC Centre for Global Infectious Disease Analysis COVID-19 planning tools</italic>. 2020 [cited 8 June 2020]. Available from <ext-link ext-link-type="uri" xlink:href="https://mrc-ide.github.io/covid19-short-term-forecasts/index.html" xlink:type="simple">https://mrc-ide.github.io/covid19-short-term-forecasts/index.html</ext-link></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref010">
<label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Cori</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Ferguson</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Fraser</surname> <given-names>C</given-names></name> and <name name-style="western"><surname>Cauchemez</surname> <given-names>S</given-names></name>. <article-title>A new framework and software to estimate time-varying reproduction numbers during epidemics</article-title>. <source><italic>Am. J. Epidemiol</italic></source>. <year>2013</year>;<volume>178</volume>(<issue>9</issue>):<fpage>1505</fpage>–<lpage>12</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/aje/kwt133" xlink:type="simple">10.1093/aje/kwt133</ext-link></comment> <object-id pub-id-type="pmid">24043437</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref011">
<label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Parag</surname> <given-names>K</given-names></name> and <name name-style="western"><surname>Donnelly</surname> <given-names>C</given-names></name>. <article-title>Adaptive estimation for epidemic renewal and phylogenetic skyline models</article-title>. <source><italic>Syst. Biol</italic></source>. <year>2020</year>;<fpage>syaa035</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/sysbio/syaa035" xlink:type="simple">10.1093/sysbio/syaa035</ext-link></comment> <object-id pub-id-type="pmid">32333789</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref012">
<label>12</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Grunwald</surname> <given-names>P</given-names></name>. <source><italic>The Minimum Description Length Principle</italic></source>. <publisher-name>The MIT Press</publisher-name>; <year>2007</year>.</mixed-citation>
</ref>
<ref id="pcbi.1007990.ref013">
<label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Rissanen</surname> <given-names>J</given-names></name>. <article-title>Order estimation by accumulated prediction errors</article-title>. <source><italic>J. Appl. Prob</italic></source>. <year>1986</year>;<volume>23</volume>:<fpage>55</fpage>–<lpage>61</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/3214342" xlink:type="simple">10.2307/3214342</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref014">
<label>14</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Wagenmakers</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Grunwald</surname> <given-names>P</given-names></name>, and <name name-style="western"><surname>Steyvers</surname> <given-names>M</given-names></name>. <article-title>Accumulative prediction error and the selection of time series models</article-title>. <source><italic>J. Math. Psychol</italic></source>. <year>2006</year>;<volume>50</volume>:<fpage>149</fpage>–<lpage>166</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jmp.2006.01.004" xlink:type="simple">10.1016/j.jmp.2006.01.004</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref015">
<label>15</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Nouvellet</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Cori</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Garske</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Blake</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Dorigatti</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Hinsley</surname> <given-names>W</given-names></name>, <etal>et al</etal>. <article-title>A simple approach to measure transmissibility and forecast incidence</article-title>. <source><italic>Epidemics</italic></source>. <year>2018</year>;<volume>22</volume>:<fpage>29</fpage>–<lpage>35</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.epidem.2017.02.012" xlink:type="simple">10.1016/j.epidem.2017.02.012</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref016">
<label>16</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Frost</surname> <given-names>W</given-names></name> and <name name-style="western"><surname>Sydenstricker</surname> <given-names>E</given-names></name>. <article-title>Influenza in Maryland: preliminary statistics of certain localities</article-title>. <source><italic>Public Health Rep</italic></source>. <year>1919</year>;<volume>34</volume>:<fpage>491</fpage>–<lpage>504</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/4575056" xlink:type="simple">10.2307/4575056</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref017">
<label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Leung</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Hedley</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Ho</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Chau</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Wong</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Thach</surname> <given-names>T</given-names></name>, <etal>et al</etal>. <article-title>The epidemiology of severe acute respiratory syndrome in the 2003 Hong Kong epidemic: An analysis of all 1755 patients</article-title>. <source><italic>Ann. Intern. Med</italic></source>. <year>2004</year>;<volume>141</volume>(<issue>9</issue>):<fpage>662</fpage>–<lpage>73</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7326/0003-4819-141-9-200411020-00006" xlink:type="simple">10.7326/0003-4819-141-9-200411020-00006</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref018">
<label>18</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Parag</surname> <given-names>K</given-names></name> and <name name-style="western"><surname>Pybus</surname> <given-names>O</given-names></name>. <article-title>Robust design for coalescent model inference</article-title>. <source><italic>Syst. Biol</italic></source>. <year>2019</year>;<volume>68</volume>(<issue>5</issue>):<fpage>730</fpage>–<lpage>43</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/sysbio/syz008" xlink:type="simple">10.1093/sysbio/syz008</ext-link></comment> <object-id pub-id-type="pmid">30726979</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref019">
<label>19</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Parag</surname> <given-names>K</given-names></name>. <article-title>On signalling and estimation limits for molecular birth-processes</article-title>. <source><italic>J. Theor. Biol</italic></source>. <year>2019</year>;<volume>480</volume>:<fpage>262</fpage>–<lpage>73</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jtbi.2019.07.007" xlink:type="simple">10.1016/j.jtbi.2019.07.007</ext-link></comment> <object-id pub-id-type="pmid">31299332</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref020">
<label>20</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Viboud</surname> <given-names>C</given-names></name> and <name name-style="western"><surname>Vespignani</surname> <given-names>A</given-names></name>. <article-title>The future of influenza forecasts</article-title>. <source><italic>PNAS</italic></source>. <year>2019</year>;<volume>116</volume>(<issue>8</issue>):<fpage>2802</fpage>–<lpage>4</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1822167116" xlink:type="simple">10.1073/pnas.1822167116</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref021">
<label>21</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Funk</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Camacho</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kucharski</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Lowe</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Eggo</surname> <given-names>R</given-names></name> and <name name-style="western"><surname>Edmunds</surname> <given-names>W</given-names></name>. <article-title>Assessing the performance of real-time epidemic forecasts: A case study of Ebola in the western area region of Sierra Leone, 2014-15</article-title>. <source><italic>PLOS Comput. Biol</italic></source>. <year>2019</year>;<volume>15</volume>(<issue>2</issue>):<fpage>e1006785</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1006785" xlink:type="simple">10.1371/journal.pcbi.1006785</ext-link></comment> <object-id pub-id-type="pmid">30742608</object-id></mixed-citation>
</ref>
<ref id="pcbi.1007990.ref022">
<label>22</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Reich</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Brooks</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Fox</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Kandula</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>McGowan</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Moore</surname> <given-names>E</given-names></name>, <etal>et al</etal>. <article-title>A collaborative multiyear, multimodel assessment of seasonal influenza forecasting in the United States</article-title>. <source><italic>PNAS</italic></source>. <year>2019</year>;<volume>116</volume>(<issue>8</issue>):<fpage>3146</fpage>–<lpage>54</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1812594116" xlink:type="simple">10.1073/pnas.1812594116</ext-link></comment></mixed-citation>
</ref>
</ref-list>
</back>
<sub-article article-type="aggregated-review-documents" id="pcbi.1007990.r001" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r001</article-id>
<title-group>
<article-title>Decision Letter 0</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ferrari</surname>
<given-names>Matthew (Matt)</given-names>
</name>
<role>Associate Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Pitzer</surname>
<given-names>Virginia E.</given-names>
</name>
<role>Deputy Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2020</copyright-year>
<copyright-holder>Ferrari, Pitzer</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="article" id="rel-obj001" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>0</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">1 Mar 2020</named-content>
</p>
<p>Dear Dr Parag,</p>
<p>Thank you very much for submitting your manuscript "Using information theory to optimise epidemic models for real-time prediction and estimation" for consideration at PLOS Computational Biology. As with all papers reviewed by the journal, your manuscript was reviewed by members of the editorial board and by several independent reviewers. The reviewers appreciated the attention to an important topic. Based on the reviews, we are likely to accept this manuscript for publication, providing that you modify the manuscript according to the review recommendations.</p>
<p>I congratulate the author's on a very nice manuscript that was well received by the reviewers. Both reviewers were complimentary of the methods, though they raised a number of concerns about the presentation. I would encourage the authors to revisit the figures, as both reviewers found that they were challenging to interpret. Specifically:</p>
<p>1. Please increase the font size for figure labeling</p>
<p>2. If possible, consider limiting the number of scenarios presented in the multi-panel figures (particularly Fig 2). This could be reduced to a smaller number of exemplary scenarios (e.g. a,f,c,e where the latter two reflect scenarios exemplary of interventions)</p>
<p>3. Figure 1 is quite dense and I agree with R2 that breaking this up into panels that could be labelled, and this referenced in the legend, would help with interpretation.</p>
<p>4. consider using a color palettes that have greater contrasts.</p>
<p>Both reviewers comment on the tendency for k* to be estimated at the minimal (both reviewers) or maximal (R2) values. Please elaborate on this phenomenon.</p>
<p>Please consider the suggestion 1 of R2. If it is possible to clarify the applicability in real-time, I suspect that would appeal to readers.</p>
<p>Please address the minor comments of R2 and comments 1&amp;2 of R1.</p>
<p>Please prepare and submit your revised manuscript within 30 days. If you anticipate any delay, please let us know the expected resubmission date by replying to this email. </p>
<p>When you are ready to resubmit, please upload the following:</p>
<p>[1] A letter containing a detailed list of your responses to all review comments, and a description of the changes you have made in the manuscript. Please note while forming your response, if your article is accepted, you may have the opportunity to make the peer review history publicly available. The record will include editor decision letters (with reviews) and your responses to reviewer comments. If eligible, we will contact you to opt in or out</p>
<p>[2] Two versions of the revised manuscript: one with either highlights or tracked changes denoting where the text has been changed; the other a clean version (uploaded as the manuscript file).</p>
<p>Important additional instructions are given below your reviewer comments.</p>
<p>Thank you again for your submission to our journal. We hope that our editorial process has been constructive so far, and we welcome your feedback at any time. Please don't hesitate to contact us if you have any questions or comments.</p>
<p>Sincerely,</p>
<p>Matthew (Matt) Ferrari</p>
<p>Associate Editor</p>
<p>PLOS Computational Biology</p>
<p>Virginia Pitzer</p>
<p>Deputy Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************</p>
<p>A link appears below if there are any accompanying review attachments. If you believe any reviews to be missing, please contact <email xlink:type="simple">ploscompbiol@plos.org</email> immediately:</p>
<p>[LINK]</p>
<p>I congratulate the author's on a very nice manuscript that was well received by the reviewers. Both reviewers were complimentary of the methods, though they raised a number of concerns about the presentation. I would encourage the authors to revisit the figures, as both reviewers found that they were challenging to interpret. Specifically:</p>
<p>1. Please increase the font size for figure labeling</p>
<p>2. If possible, consider limiting the number of scenarios presented in the multi-panel figures (particularly Fig 2). This could be reduced to a smaller number of exemplary scenarios (e.g. a,f,c,e where the latter two reflect scenarios exemplary of interventions)</p>
<p>3. Figure 1 is quite dense and I agree with R2 that breaking this up into panels that could be labelled, and this referenced in the legend, would help with interpretation.</p>
<p>4. consider using a color palettes that have greater contrasts.</p>
<p>Both reviewers comment on the tendency for k* to be estimated at the minimal (both reviewers) or maximal (R2) values. Please elaborate on this phenomenon.</p>
<p>Please consider the suggestion 1 of R2. If it is possible to clarify the applicability in real-time, I suspect that would appeal to readers.</p>
<p>Please address the minor comments of R2 and comments 1&amp;2 of R1.</p>
<p>Reviewer's Responses to Questions</p>
<p><bold>Comments to the Authors:</bold></p>
<p><bold>Please note here if the review is uploaded as an attachment.</bold></p>
<p>Reviewer #1: This was an interesting paper that adds to the suite of methods for analysing case report data. In particular it provides a useful measure of the “window size” over which the reproductive ratio Rt should be calculated. As such, this could prove to be a powerful tool.</p>
<p>On the whole I found the paper to be extremely well written, and I only have a few minor comments.</p>
<p>1. In the abstract, I find the phrase “most succinctly describes” to be rather vague and confusing, could the authors seek a more informative description?</p>
<p>2. Page 9, the choice of a Gamma distribution for the posterior should be better motivated.</p>
<p>3. Figures: I found these hard to read. I realise that most people view on-line and therefore can greatly magnify figures, but a printed version is unreadable. In figure 7, I wonder if plotting Rt on a log-scale would help?</p>
<p>4. The paper ends on a whimper, with it being unclear if the k=2 found from real data is true or an artifact of a noisy system. I’m surprised that the authors didn’t test the method against simulations that were made sequentially noisier, this would seem an obvious test. I also feel that the speculation in the last paragraph of the results would sit far better in the discussion.</p>
<p>Reviewer #2: See attached</p>
<p>**********</p>
<p><bold>Have all data underlying the figures and results presented in the manuscript been provided?</bold></p>
<p>Large-scale datasets should be made available via a public repository as described in the <italic>PLOS Computational Biology</italic> <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/ploscompbiol/s/data-availability" xlink:type="simple">data availability policy</ext-link>, and numerical data that underlies graphs or summary statistics should be provided in spreadsheet form as supporting information.</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.</p>
<p>Reviewer #1: No</p>
<p>Reviewer #2: No</p>
<p><underline>Figure Files:</underline></p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <underline><ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com</ext-link></underline>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email us at <underline><email xlink:type="simple">figures@plos.org</email></underline>.</p>
<p><underline>Data Requirements:</underline></p>
<p>Please note that, as a condition of publication, PLOS' data policy requires that you make available all data used to draw the conclusions outlined in your manuscript. Data must be deposited in an appropriate repository, included within the body of the manuscript, or uploaded as supporting information. This includes all numerical values that were used to generate graphs, histograms etc.. For an example in PLOS Biology see here: <ext-link ext-link-type="uri" xlink:href="http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5" xlink:type="simple">http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5</ext-link>.</p>
<p><underline>Reproducibility:</underline></p>
<p>To enhance the reproducibility of your results, PLOS recommends that you deposit laboratory protocols in protocols.io, where a protocol can be assigned its own identifier (DOI) such that it can be cited independently in the future. For instructions see <underline><ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plospathogens/s/submission-guidelines" xlink:type="simple">http://journals.plos.org/ploscompbiol/s/submission-guidelines#loc-materials-and-methods</ext-link></underline></p>
<supplementary-material id="pcbi.1007990.s002" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.s002" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">plos_renewal_review.docx</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pcbi.1007990.r002">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r002</article-id>
<title-group>
<article-title>Author response to Decision Letter 0</article-title>
</title-group>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj002" link-type="rebutted-decision-letter" object-id="10.1371/journal.pcbi.1007990.r001" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">22 Apr 2020</named-content>
</p>
<supplementary-material id="pcbi.1007990.s003" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.s003" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">plos_revisions.docx</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1007990.r003" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r003</article-id>
<title-group>
<article-title>Decision Letter 1</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ferrari</surname>
<given-names>Matthew (Matt)</given-names>
</name>
<role>Associate Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Pitzer</surname>
<given-names>Virginia E.</given-names>
</name>
<role>Deputy Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2020</copyright-year>
<copyright-holder>Ferrari, Pitzer</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="article" id="rel-obj003" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">13 May 2020</named-content>
</p>
<p>Dear Dr Parag,</p>
<p>Thank you very much for submitting your manuscript "Using information theory to optimise epidemic models for real-time prediction and estimation" for consideration at PLOS Computational Biology. As with all papers reviewed by the journal, your manuscript was reviewed by members of the editorial board and by several independent reviewers. The reviewers appreciated the attention to an important topic. Based on the reviews, we are likely to accept this manuscript for publication, providing that you modify the manuscript according to the review recommendations.</p>
<p>I thank the authors for their careful revisions. I do not see the need to send this back to review, but I would ask the authors to correct the legend for Figure 8 before I recommend this for publication. I assume the "(solid blue, left y axis)" and "(dotted red, right y axis)" refer to the points in the figure, but it isn't clear if I am assuming correctly or if there are elements missing from the figure. With this clarified, I will submit a final recommendation of "accept".</p>
<p>Please prepare and submit your revised manuscript within 30 days. If you anticipate any delay, please let us know the expected resubmission date by replying to this email. </p>
<p>When you are ready to resubmit, please upload the following:</p>
<p>[1] A letter containing a detailed list of your responses to all review comments, and a description of the changes you have made in the manuscript. Please note while forming your response, if your article is accepted, you may have the opportunity to make the peer review history publicly available. The record will include editor decision letters (with reviews) and your responses to reviewer comments. If eligible, we will contact you to opt in or out</p>
<p>[2] Two versions of the revised manuscript: one with either highlights or tracked changes denoting where the text has been changed; the other a clean version (uploaded as the manuscript file).</p>
<p>Important additional instructions are given below your reviewer comments.</p>
<p>Thank you again for your submission to our journal. We hope that our editorial process has been constructive so far, and we welcome your feedback at any time. Please don't hesitate to contact us if you have any questions or comments.</p>
<p>Sincerely,</p>
<p>Matthew (Matt) Ferrari</p>
<p>Associate Editor</p>
<p>PLOS Computational Biology</p>
<p>Virginia Pitzer</p>
<p>Deputy Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************</p>
<p>A link appears below if there are any accompanying review attachments. If you believe any reviews to be missing, please contact <email xlink:type="simple">ploscompbiol@plos.org</email> immediately:</p>
<p>[LINK]</p>
<p>I thank the authors for their careful revisions. I do not see the need to send this back to review, but I would ask the authors to correct the legend for Figure 8 before I recommend this for publication. I assume the "(solid blue, left y axis)" and "(dotted red, right y axis)" refer to the points in the figure, but it isn't clear if I am assuming correctly or if there are elements missing from the figure. With this clarified, I will submit a final recommendation of "accept".</p>
<p><underline>Figure Files:</underline></p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <underline><ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com</ext-link></underline>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email us at <underline><email xlink:type="simple">figures@plos.org</email></underline>.</p>
<p><underline>Data Requirements:</underline></p>
<p>Please note that, as a condition of publication, PLOS' data policy requires that you make available all data used to draw the conclusions outlined in your manuscript. Data must be deposited in an appropriate repository, included within the body of the manuscript, or uploaded as supporting information. This includes all numerical values that were used to generate graphs, histograms etc.. For an example in PLOS Biology see here: <ext-link ext-link-type="uri" xlink:href="http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5" xlink:type="simple">http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5</ext-link>.</p>
<p><underline>Reproducibility:</underline></p>
<p>To enhance the reproducibility of your results, PLOS recommends that you deposit laboratory protocols in protocols.io, where a protocol can be assigned its own identifier (DOI) such that it can be cited independently in the future. For instructions see <underline><ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plospathogens/s/submission-guidelines" xlink:type="simple">http://journals.plos.org/ploscompbiol/s/submission-guidelines#loc-materials-and-methods</ext-link></underline></p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pcbi.1007990.r004">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r004</article-id>
<title-group>
<article-title>Author response to Decision Letter 1</article-title>
</title-group>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj004" link-type="rebutted-decision-letter" object-id="10.1371/journal.pcbi.1007990.r003" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">13 May 2020</named-content>
</p>
<supplementary-material id="pcbi.1007990.s004" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1007990.s004" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">plos_revisions.docx</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1007990.r005" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r005</article-id>
<title-group>
<article-title>Decision Letter 2</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ferrari</surname>
<given-names>Matthew (Matt)</given-names>
</name>
<role>Associate Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Pitzer</surname>
<given-names>Virginia E.</given-names>
</name>
<role>Deputy Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2020</copyright-year>
<copyright-holder>Ferrari, Pitzer</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="article" id="rel-obj005" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">27 May 2020</named-content>
</p>
<p>Dear Dr Parag,</p>
<p>We are pleased to inform you that your manuscript 'Using information theory to optimise epidemic models for real-time prediction and estimation' has been provisionally accepted for publication in PLOS Computational Biology.</p>
<p>Before your manuscript can be formally accepted you will need to complete some formatting changes, which you will receive in a follow up email. A member of our team will be in touch with a set of requests.</p>
<p>Please note that your manuscript will not be scheduled for publication until you have made the required changes, so a swift response is appreciated.</p>
<p>IMPORTANT: The editorial review process is now complete. PLOS will only permit corrections to spelling, formatting or significant scientific errors from this point onwards. Requests for major changes, or any which affect the scientific understanding of your work, will cause delays to the publication date of your manuscript.</p>
<p>Should you, your institution's press office or the journal office choose to press release your paper, you will automatically be opted out of early publication. We ask that you notify us now if you or your institution is planning to press release the article. All press must be co-ordinated with PLOS.</p>
<p>Thank you again for supporting Open Access publishing; we are looking forward to publishing your work in PLOS Computational Biology. </p>
<p>Best regards,</p>
<p>Matthew (Matt) Ferrari</p>
<p>Associate Editor</p>
<p>PLOS Computational Biology</p>
<p>Virginia Pitzer</p>
<p>Deputy Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************************************************</p>
<p>I thank the authors for their attention the reviewer comments and commend them on a fine manuscript.</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1007990.r006" specific-use="acceptance-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1007990.r006</article-id>
<title-group>
<article-title>Acceptance letter</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ferrari</surname>
<given-names>Matthew (Matt)</given-names>
</name>
<role>Associate Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Pitzer</surname>
<given-names>Virginia E.</given-names>
</name>
<role>Deputy Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2020</copyright-year>
<copyright-holder>Ferrari, Pitzer</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1007990" document-id-type="doi" document-type="article" id="rel-obj006" link-type="peer-reviewed-article"/>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">24 Jun 2020</named-content>
</p>
<p>PCOMPBIOL-D-20-00012R2 </p>
<p>Using information theory to optimise epidemic models for real-time prediction and estimation</p>
<p>Dear Dr Parag,</p>
<p>I am pleased to inform you that your manuscript has been formally accepted for publication in PLOS Computational Biology. Your manuscript is now with our production department and you will be notified of the publication date in due course.</p>
<p>The corresponding author will soon be receiving a typeset proof for review, to ensure errors have not been introduced during production. Please review the PDF proof of your manuscript carefully, as this is the last chance to correct any errors. Please note that major changes, or those which affect the scientific understanding of the work, will likely cause delays to the publication date of your manuscript. </p>
<p>Soon after your final files are uploaded, unless you have opted out, the early version of your manuscript will be published online. The date of the early version will be your article's publication date. The final article will be published to the same URL, and all versions of the paper will be accessible to readers.</p>
<p>Thank you again for supporting PLOS Computational Biology and open-access publishing. We are looking forward to publishing your work! </p>
<p>With kind regards,</p>
<p>Sarah Hammond</p>
<p>PLOS Computational Biology | Carlyle House, Carlyle Road, Cambridge CB4 3DN | United Kingdom <email xlink:type="simple">ploscompbiol@plos.org</email> | Phone +44 (0) 1223-442824 | <ext-link ext-link-type="uri" xlink:href="http://ploscompbiol.org" xlink:type="simple">ploscompbiol.org</ext-link> | @PLOSCompBiol</p>
</body>
</sub-article>
</article>