<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS ONE</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PONE-D-16-08387</article-id>
<article-id pub-id-type="doi">10.1371/journal.pone.0166694</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Communications</subject><subj-group><subject>Social communication</subject><subj-group><subject>Social media</subject><subj-group><subject>Twitter</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Computer and information sciences</subject><subj-group><subject>Network analysis</subject><subj-group><subject>Social networks</subject><subj-group><subject>Social media</subject><subj-group><subject>Twitter</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Social networks</subject><subj-group><subject>Social media</subject><subj-group><subject>Twitter</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Computer and information sciences</subject><subj-group><subject>Network analysis</subject><subj-group><subject>Social networks</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Social networks</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Communications</subject><subj-group><subject>Social communication</subject><subj-group><subject>Social media</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Computer and information sciences</subject><subj-group><subject>Network analysis</subject><subj-group><subject>Social networks</subject><subj-group><subject>Social media</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Social networks</subject><subj-group><subject>Social media</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Collective human behavior</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Collective human behavior</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Behavior</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Earth sciences</subject><subj-group><subject>Natural disasters</subject><subj-group><subject>Tornadoes</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Earth sciences</subject><subj-group><subject>Atmospheric science</subject><subj-group><subject>Meteorology</subject><subj-group><subject>Wind</subject><subj-group><subject>Tornadoes</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Behavior</subject><subj-group><subject>Animal behavior</subject><subj-group><subject>Collective animal behavior</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Biology and life sciences</subject><subj-group><subject>Zoology</subject><subj-group><subject>Animal behavior</subject><subj-group><subject>Collective animal behavior</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3"><subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Communications</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Prediction and Characterization of High-Activity Events in Social Media Triggered by Real-World News</article-title>
<alt-title alt-title-type="running-head">Early Prediction of High-Activity Events</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<name name-style="western">
<surname>Kalyanam</surname> <given-names>Janani</given-names></name>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Quezada</surname> <given-names>Mauricio</given-names></name>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Poblete</surname> <given-names>Barbara</given-names></name>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Lanckriet</surname> <given-names>Gert</given-names></name>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>Department of Electrical and Computer Engineering, University of California San Diego, La Jolla, California, United States of America</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Department of Computer Science, University of Chile, Santiago, Chile</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Lambiotte</surname> <given-names>Renaud</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>Universite de Namur, BELGIUM</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<fn fn-type="con">
<p>
<list list-type="simple">
<list-item>
<p><bold>Conceptualization:</bold> BP GL.</p>
</list-item>
<list-item>
<p><bold>Data curation:</bold> MQ.</p>
</list-item>
<list-item>
<p><bold>Formal analysis:</bold> JK MQ.</p>
</list-item>
<list-item>
<p><bold>Funding acquisition:</bold> BP GL.</p>
</list-item>
<list-item>
<p><bold>Investigation:</bold> JK MQ.</p>
</list-item>
<list-item>
<p><bold>Methodology:</bold> BP GL.</p>
</list-item>
<list-item>
<p><bold>Project administration:</bold> BP.</p>
</list-item>
<list-item>
<p><bold>Resources:</bold> BP MQ JK.</p>
</list-item>
<list-item>
<p><bold>Software:</bold> JK MQ.</p>
</list-item>
<list-item>
<p><bold>Supervision:</bold> BP GL.</p>
</list-item>
<list-item>
<p><bold>Validation:</bold> JK MQ.</p>
</list-item>
<list-item>
<p><bold>Visualization:</bold> MQ JK.</p>
</list-item>
<list-item>
<p><bold>Writing – original draft:</bold> BP MQ JK.</p>
</list-item>
<list-item>
<p><bold>Writing – review &amp; editing:</bold> BP GL.</p>
</list-item>
</list>
</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">jkalyana@ucsd.edu</email></corresp>
</author-notes>
<pub-date pub-type="collection">
<year>2016</year>
</pub-date>
<pub-date pub-type="epub">
<day>16</day>
<month>12</month>
<year>2016</year>
</pub-date>
<volume>11</volume>
<issue>12</issue>
<elocation-id>e0166694</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>2</month>
<year>2016</year>
</date>
<date date-type="accepted">
<day>2</day>
<month>11</month>
<year>2016</year>
</date>
</history>
<permissions>
<copyright-year>2016</copyright-year>
<copyright-holder>Kalyanam et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0166694"/>
<abstract>
<p>On-line social networks publish information on a high volume of real-world events almost instantly, becoming a primary source for breaking news. Some of these real-world events can end up having a very strong impact on on-line social networks. The effect of such events can be analyzed from several perspectives, one of them being the intensity and characteristics of the collective activity that it produces in the social platform. We research 5,234 real-world news events encompassing 43 million messages discussed on the Twitter microblogging service for approximately 1 year. We show empirically that exogenous news events naturally create collective patterns of bursty behavior in combination with long periods of inactivity in the network. This type of behavior agrees with other patterns previously observed in other types of natural collective phenomena, as well as in individual human communications. In addition, we propose a methodology to classify news events according to the different levels of intensity in activity that they produce. In particular, we analyze the most highly active events and observe a consistent and strikingly different collective reaction from users when they are exposed to such events. This reaction is independent of an event’s reach and scope. We further observe that extremely high-activity events have characteristics that are quite distinguishable at the beginning stages of their outbreak. This allows us to predict with high precision, the top 8% of events that will have the most impact in the social network by just using the first 5% of the information of an event’s lifetime evolution. This strongly implies that high-activity events are naturally prioritized collectively by the social network, engaging users early on, way before they are brought to the mainstream audience.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution>CCF</institution>
</funding-source>
<award-id>0830535</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Lanckriet</surname> <given-names>Gert</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award002">
<funding-source>
<institution>IIS</institution>
</funding-source>
<award-id>1054960</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Lanckriet</surname> <given-names>Gert</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award003">
<funding-source>
<institution>FONDECYT</institution>
</funding-source>
<award-id>11121511</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Poblete</surname> <given-names>Barbara</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award004">
<funding-source>
<institution>Millennium Nucleus Center for Semantic Web Research</institution>
</funding-source>
<award-id>NC120004.</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Poblete</surname> <given-names>Barbara</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award005">
<funding-source>
<institution>CONICYT</institution>
</funding-source>
<award-id>2015/21151445</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Quezada</surname> <given-names>Mauricio</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award006">
<funding-source>
<institution>Yahoo Faculty Research Engagement Program</institution>
</funding-source>
<principal-award-recipient>
<name name-style="western">
<surname>Kalyanam</surname> <given-names>Janani</given-names></name>
</principal-award-recipient>
</award-group>
<award-group id="award007">
<funding-source>
<institution>Yahoo Faculty Research Engagement Program</institution>
</funding-source>
<principal-award-recipient>
<name name-style="western">
<surname>Lanckriet</surname> <given-names>Gert</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>This work was supported by National Science Foundation CCF 0830535, GL; National Science Foundation, IIS 1054960, GL; Fondo Nacional de Desarrollo Científico y Tecnológico, 11121511, BP; Millennium Nucleus Center for Semantic Web Research, NC120004, BP; Comision Nacional de Ciencia y Tecnología, 2015/21151445, MQ; and Yahoo Faculty Research Engagement Program, JK, GL.</funding-statement>
</funding-group>
<counts>
<fig-count count="6"/>
<table-count count="5"/>
<page-count count="13"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All data files are available here: <ext-link ext-link-type="uri" xlink:href="https://dx.doi.org/10.6084/m9.figshare.3465974" xlink:type="simple">https://dx.doi.org/10.6084/m9.figshare.3465974</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://users.dcc.uchile.cl/~mquezada/breakingnews/" xlink:type="simple">https://users.dcc.uchile.cl/~mquezada/breakingnews/</ext-link>. A description of the data collection methodology is provided in <xref ref-type="supplementary-material" rid="pone.0166694.s001">S1 Appendix</xref>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Social media is now a primary source of breaking news information for millions of users all over the world [<xref ref-type="bibr" rid="pone.0166694.ref001">1</xref>]. On-line social networks along with mobile internet devices have crowdsourced the task of disseminating real-time information. As a result, both news media and news consumers have become inundated with much more information than they can process. One possible way of handling this data overload, is to find ways to filter and prioritize information that has the potential of creating a strong collective impact. Understanding and quickly identifying the type of reaction that certain exogenous events will produce in on-line social networks, at both global and local scales, can help in the understanding of collective human behavior, as well as improve information delivery, journalistic coverage and crisis management, among other things. We address this challenge by analyzing the properties of real-world news events in on-line social networks, showing that they corroborate patterns previously identified in other case studies of human communications. In addition, we present our main findings of how news events that produce extremely high-activity can be clearly identified in the early stages of their outbreak.</p>
<p>The study of information propagation on the Web has sparked tremendous interest in recent years. Current literature on the subject primarily considers the process through which a <italic>meme</italic>, usually a piece of media (like a video, an image, or a specific Web article), gains popularity [<xref ref-type="bibr" rid="pone.0166694.ref002">2</xref>–<xref ref-type="bibr" rid="pone.0166694.ref009">9</xref>]. However, a meme represents a simple information unit and its propagation behavior does not necessarily correspond to that of more complex information such as news events. News events are usually diffused in the network in many different formats, e.g., a particular news story such as an <italic>earthquake in Japan</italic> can be communicated through images, URLs, tweets, videos, etc. Therefore, current research can benefit from analyzing the effects of more high-level forms of information.</p>
<p>Traditionally, the impact of information in on-line social networks has been measured in relation to the total amount of attention that this subject receives [<xref ref-type="bibr" rid="pone.0166694.ref010">10</xref>–<xref ref-type="bibr" rid="pone.0166694.ref014">14</xref>]. That is, if a content posted in the network receives votes/comments/shares above a certain threshold it is usually deemed as <italic>viral</italic> or <italic>popular</italic>. Nevertheless, this notion of popularity or impact will favor only information that produces very large volumes of social media messages. Naturally, global breaking news that has world-wide coverage and that produces a high volume of activity in a short time should be considered as having a strong impact on the network. However, there are other types of events that can produce a similar reaction in smaller on-line communities such as, for example, on users from a particular country (e.g., the withdrawal of the main right wing presidential candidate in Chile due to psychiatric problems, just before elections [<xref ref-type="bibr" rid="pone.0166694.ref015">15</xref>]). Clearly, events of local scope do not produce as much social media activity as events of global scope, but they can create a strong and immediate reaction from users in local networks [<xref ref-type="bibr" rid="pone.0166694.ref016">16</xref>]. Conversely, there are large events which do not produce an intense reaction, such as <italic>The Oscars</italic> (<xref ref-type="fig" rid="pone.0166694.g001">Fig 1</xref>), which span a long period of time and are discussed by social network users for weeks or even months, but do not spark intense user activity. Therefore, it is reasonable to consider additional dimensions, than just volume, when analyzing the impact of information in on-line communities.</p>
<fig id="pone.0166694.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Examples of interarrival time histograms of two real-world news events discussed on Twitter.</title>
<p>The event [nelson, mandela] (top) was collected on 12/05/2013. Since there is a high concentration in the first histogram bin, we conclude that most of the social media posts for this event occur in one or more successions of high-activity bursts (therefore, considered a high-activity event). The second event, [may, oscar] (bottom) was collected on 03/23/2014 about The Oscars event that was held a few weeks before. The arrival times of these posts are much more spread out, displaying much less concentration of bursty activity.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g001" xlink:type="simple"/>
</fig>
<p>Prior research has shown that certain types of individual activities, such as communications (studied in email exchanges), work patterns and entertainment, follow a behavior of bursts of rapidly occurring actions followed by long periods of inactivity [<xref ref-type="bibr" rid="pone.0166694.ref017">17</xref>], referred to as temporally inhomogeneous behavior [<xref ref-type="bibr" rid="pone.0166694.ref018">18</xref>]. This type of behavior initially observed in individual activities, has also been observed in relation to other naturally occurring types of collective phenomena in human dynamics similar to processes seen in self-organized criticality [<xref ref-type="bibr" rid="pone.0166694.ref018">18</xref>]. In particular, extremely high-activity bursty behavior seems to also occur in critical situations, observed from the information flow in cell phone networks during emergencies [<xref ref-type="bibr" rid="pone.0166694.ref019">19</xref>]. Although, there is research towards modeling this type of collective behavior [<xref ref-type="bibr" rid="pone.0166694.ref020">20</xref>] in on-line social networks, to the best of our knowledge, it has not yet been analyzed quantitatively.</p>
<p>Our work focuses on high-activity events in social media produced by real-world news, with the following contributions:</p>
<list list-type="order">
<list-item>
<p>We introduce a methodology for modeling and classifying events in social media, based on the intensity of the activity that they produce. This methodology is independent of the size and scope of the event, and is an indicator of the impact that the event information had on the social network.</p>
</list-item>
<list-item>
<p>We show empirically that real-world news events produce collective patterns of bursty behavior in the social network, in combination with long periods of inactivity. Furthermore, we identify events for which most of their activity is concentrated into very high-activity periods, we call these events <italic>high-activity events</italic>.</p>
</list-item>
<list-item>
<p>We determine the existence of unique characteristics that differentiate how high-activity events propagate in the social network.</p>
</list-item>
<list-item>
<p>We show that an important portion of high-activity events can be predicted very early in their lifecycle, indicating that this type of information is spontaneously identified and filtered collectively, early on, by social network users.</p>
</list-item>
</list>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Materials and Methods</title>
<p>We define an event as a conglomerate of information that encompasses all of the social media content related to a real-world news occurrence. Using this specification, which considers an event as a complex unit of information, we study the type of collective reaction produced by the event on the social network. In particular, we analyze the intensity or immediacy of the social network’s response. By analyzing the levels of intensity in activity induced by different exogenous events to the network, we are implicitly studying the priority that has been collectively assigned to the event by groups of independent individuals [<xref ref-type="bibr" rid="pone.0166694.ref017">17</xref>, <xref ref-type="bibr" rid="pone.0166694.ref018">18</xref>].</p>
<p>We characterize an event’s discrete activity dynamics by using <italic>interarrival times</italic> between consecutive social media messages within an event (e.g., <italic>d</italic><sub><italic>i</italic></sub> = <italic>t</italic><sub><italic>i</italic>+1</sub> − <italic>t</italic><sub><italic>i</italic></sub>, where <italic>d</italic><sub><italic>i</italic></sub> denotes the interarrival time between two consecutive social media messages <italic>i</italic> and <italic>i</italic> + 1 that arrived in moments <italic>t</italic><sub><italic>i</italic></sub> and <italic>t</italic><sub><italic>i</italic>+1</sub>, respectively).</p>
<p>We introduce a novel vectorial representation based on a <italic>vector quantization of the interarrival time distribution</italic>, which we call <italic>“VQ-event model”</italic>. This model is designed to filter events based on the distribution of the interarrival times between consecutive messages. This approach is inspired by the <italic>codebook-based representation</italic> from the field of multimedia content analysis, which has been used in audio processing and computer vision [<xref ref-type="bibr" rid="pone.0166694.ref021">21</xref>, <xref ref-type="bibr" rid="pone.0166694.ref022">22</xref>]. In our proposed approach, our method learns a set of the most representative interarrival times from a large training corpus of events; each one of the representative interarrival times is known as a <italic>codeword</italic> and the complete learned set is known as the <italic>codebook</italic> [<xref ref-type="bibr" rid="pone.0166694.ref022">22</xref>]. Each event is then modeled using a vector quantization (VQ) that converts the interarrival times of an event into a discrete set of values, each value corresponding to the closest codeword in the codebook (details in supplementary material). The resulting VQ-event model is then a vector in which each dimension contains the percentage of interarrival times of the event that were assigned a particular codeword in the codebook.</p>
<p>The VQ-event representation is relative to an event’s overall size since the model is normalized with respect to the number of messages in the event. Therefore the only criteria that are considered in the model are the interarrival times of each particular event. This model allows us to group events based on the <italic>similarity of the distribution</italic> of their interarrival times. In those terms, we consider as high-activity events those events for which the distribution of interarrival times is most heavily skewed towards the smallest possible interval, zero. In other words, events for which the overall activity is extremely intense in comparison with other events.</p>
<p>To illustrate events with different levels of intensity in activity we present two examples taken from our analysis of Twitter data. These examples show the interarrival time histograms for the entire lifecycle of the two events. In the first example, the majority of the messages about the death of political leader Nelson Mandela (<xref ref-type="fig" rid="pone.0166694.g001">Fig 1</xref>) arrive within almost zero seconds of each other. On the contrary, the messages about The Oscars (<xref ref-type="fig" rid="pone.0166694.g001">Fig 1</xref>) are much more spread out in time.</p>
<p>We note that, by using interarrival times to describe the intensity of the activity of an event, we make our analysis independent of the particular evolution of each event. By doing this, we put no restrictions on how high-activity events unfold in time, for example, they could be: (a) events that start out slowly and suddenly gain momentum, (b) events that go viral soon after they appear on social media and then decay in intensity over a long (or short) period of time, (c) events that from the beginning produce large amounts of interest and sustain that interest throughout their long (or short) lifespan, or (d) events that are a concatenation of any of the above, etc.</p>
<p>We study a dataset of news events gathered from news headlines from a <italic>manually curated</italic> list of well-known news media accounts (e.g., @CNN, @BreakingNews, @BBCNews, etc.) in the microblogging platform Twitter [<xref ref-type="bibr" rid="pone.0166694.ref023">23</xref>] (a full list of all the news media accounts is provided in the supplementary material). Headlines were collected periodically every hour, over the course of approximately one year. In parallel, all the Twitter messages (called <italic>tweets</italic>) were extracted for each news event using the public API [<xref ref-type="bibr" rid="pone.0166694.ref024">24</xref>]. This process was performed by automatically extracting descriptive sets of keywords for each event using a variation of frequent itemset extraction [<xref ref-type="bibr" rid="pone.0166694.ref025">25</xref>] over the event’s headlines. These sets of keywords were then used to retrieve corresponding user tweets for each event. We validate the events gathered in our data collection process to ensure that each group of social media posts corresponds to a meaningful and cohesive news event. We provide a detailed description of the collection methodology and of the validation of event cohesiveness in the supplementary material. Overall, the resulting dataset contains 43,256,261 tweets that account for 5,234 events (<xref ref-type="table" rid="pone.0166694.t001">Table 1</xref>).</p>
<table-wrap id="pone.0166694.t001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.t001</object-id>
<label>Table 1</label>
<caption>
<title>High-level description of the dataset of news events.</title>
</caption>
<alternatives>
<graphic id="pone.0166694.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.t001" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Event Collection Statistics</th>
<th align="left">Minimum</th>
<th align="left">Mean</th>
<th align="left">Median</th>
<th align="left">Maximum</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left"># of posts (per event)</td>
<td align="left">1,000</td>
<td align="left">8,254</td>
<td align="left">2,474</td>
<td align="left">510,920</td>
</tr>
<tr>
<td align="left"># of keywords (per tweet)</td>
<td align="left">2</td>
<td align="left">3.77</td>
<td align="left">3</td>
<td align="left">39</td>
</tr>
<tr>
<td align="left">Event duration (hours)</td>
<td align="left">0.12</td>
<td align="left">20.93</td>
<td align="left">7.46</td>
<td align="left">190.43</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>In <xref ref-type="fig" rid="pone.0166694.g002">Fig 2</xref> we characterize an example event from our dataset, by showing the set of keywords and a sample of tweets associated to the event. These keywords form a semantically meaningful event; they refer to the incident where soccer player Luis Suarez was charged for biting another player during the FIFA World Cup in 2014. This general collection process results in a set of social media posts associated to an event which can encompass several memes, viral tweets and pieces of information. Therefore, an event is composed of diverse information, addressing more heterogeneous content than prior work [<xref ref-type="bibr" rid="pone.0166694.ref002">2</xref>–<xref ref-type="bibr" rid="pone.0166694.ref004">4</xref>, <xref ref-type="bibr" rid="pone.0166694.ref006">6</xref>, <xref ref-type="bibr" rid="pone.0166694.ref007">7</xref>, <xref ref-type="bibr" rid="pone.0166694.ref026">26</xref>, <xref ref-type="bibr" rid="pone.0166694.ref027">27</xref>] which focus on single pieces of information (e.g., a particular meme, a viral tweet etc.).</p>
<fig id="pone.0166694.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g002</object-id>
<label>Fig 2</label>
<caption>
<title>An example event, collected on 06/25/2014 with keywords (left) and sample user posts (right) obtained from the Twitter Search API.</title>
<p>The tweets in the event contain at least a pair of descriptive keywords and were retrieved close to the time of the event.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g002" xlink:type="simple"/>
</fig>
<p>The collection of events is converted into their VQ-event model representation. Using this model, we can identify events that have produced similar levels of activity in the social network. In other words, events are considered to have similar activity if the interarrival times between their social media posts are similarly distributed, implying a very much alike collective reaction from users to the events within a group. In order to identify groups of similar events, we cluster the event models. We sort the resulting groups of events from highest to lowest activity, according to the concentration of social media posts in the bins that correspond to short interarrival times. We consider the events that fall in the top cluster to be high-activity events as most of their interarrival times are concentrated in the smallest interval of the VQ-event model. In our dataset, these correspond to roughly 8% of the events. We consider the next clusters in the sorted ranking to form medium-high activity events, and so on. Thus we end with four groups of events: high, medium-high, medium-low and low. <xref ref-type="fig" rid="pone.0166694.g003">Fig 3</xref> shows a heatmap of the interarrival relative frequency for each cluster. This classification of events based on activity intensity is independent of event size. More details of this methodology are provided in <xref ref-type="supplementary-material" rid="pone.0166694.s001">S1 Appendix</xref>.</p>
<fig id="pone.0166694.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Each row is the average representation of all the events in a cluster.</title>
<p>A darker cell represents a higher relative frequency value. The y-axis specifies the number of events in each cluster. Clusters are (top to bottom): high-activity, medium-high medium-low and low.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g003" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec003" sec-type="conclusions">
<title>Results and Discussion</title>
<p>Our main objective in this work is to analyze the characteristics of high-activity events which differentiate them from other types of events. In particular, we identify how early on in an event’s lifecycle can we determine if an event is going produce high activity in the on-line social network.</p>
<p>Tables <xref ref-type="table" rid="pone.0166694.t002">2</xref> and <xref ref-type="table" rid="pone.0166694.t003">3</xref> show examples of events from the high-activity category and low-activity category. We recall that the high-activity events are those which were in the top 8% of the ranking obtained by sorting the event clusters according to concentration of interarrival times of social media posts in the shortest interarrival time of the VQ-event model. <xref ref-type="table" rid="pone.0166694.t002">Table 2</xref> shows two events of different sizes (large and small) and different scopes (one global and the other of more local scope) categorized as high activity in our dataset. The first event, the death of Nelson Mandela, is one of the largest events in the dataset, with ≈ 134,000 tweets. The histogram representation of this event, shown in <xref ref-type="fig" rid="pone.0166694.g001">Fig 1</xref>, suggests that more than 80% of the activity of the event was produced in high-activity periods. This is an event of international, political, and social importance, that produced an overwhelming flood of messages on social media. Hence, it makes sense for such an example to be a high-activity event. The second event, on the other hand, about the 2013 Mumbai Gang Rape is of much smaller scale, with a total of ≈ 1,700 tweets. However, this event caused considerable amount of immediate reaction on social media, with close to 50% of its activity concentrated within high-activity periods. Despite its smaller size, in comparison to the previous event, this event displays a similar reaction to that of other high-activity events, but at a smaller scale.</p>
<table-wrap id="pone.0166694.t002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.t002</object-id>
<label>Table 2</label>
<caption>
<title>Examples of high-activity news events.</title> <p>The events shown were taken from the “high” category according to <xref ref-type="fig" rid="pone.0166694.g004">Fig 4</xref>.</p>
</caption>
<alternatives>
<graphic id="pone.0166694.t002g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.t002" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Event</th>
<th align="left">Sample Tweets</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left"><bold>Description:</bold><break/>Death of South African politician Nelson Mandela.</td>
<td align="left">@DaniellePeazer: RIP Nelson Mandela….. what a truly phenomenal and inspirational man xx</td>
</tr>
<tr>
<td align="left"><bold>Keywords:</bold><break/>[nelson, mandela]</td>
<td align="left">@iansomerhalder: Im in tears. The world has lost one of its greatest shepherds of peace. Thank you Mr.Mandela for the love you radiated. <ext-link ext-link-type="uri" xlink:href="http://t.co/u39MVVEKe8" xlink:type="simple">http://t.co/u39MVVEKe8</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Date:</bold><break/>2013-12-05</td>
<td align="left">@FootballFunnys: This is so true. RIP Nelson Mandela. <ext-link ext-link-type="uri" xlink:href="http://t.co/vF9xri8LdP" xlink:type="simple">http://t.co/vF9xri8LdP</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Size:</bold><break/>134,637 tweets</td>
<td align="left">@David_Cameron: I’ve spoken to the Speaker and there will be statements and tributes to Nelson Mandela in the House on Monday.</td>
</tr>
<tr>
<td align="left"><bold>Description:</bold><break/>2013 Mumbai Gang Rape</td>
<td align="left">@TheNewsRoundup: Mumbai gang-rape: Second accused confesses to crime: Mumbai Police—Daily News Analysis <ext-link ext-link-type="uri" xlink:href="http://t.co/KnabwhqH66" xlink:type="simple">http://t.co/KnabwhqH66</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Keywords:</bold><break/>[rape, mumbai]</td>
<td align="left">@vijayarumugam: An interesting take on the Mumbai rape: <ext-link ext-link-type="uri" xlink:href="http://t.co/ylBmW4l8sA" xlink:type="simple">http://t.co/ylBmW4l8sA</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Date:</bold><break/>2013-08-24</td>
<td align="left">@LondonStephanie: Two arrested over gang rape of Mumbai photojournalist that sparked renewed protests in India <ext-link ext-link-type="uri" xlink:href="http://t.co/McYfLNDvaE" xlink:type="simple">http://t.co/McYfLNDvaE</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Size:</bold><break/>1,705 tweets</td>
<td align="left">@GanapathyI: Most brutal rapist of Delhi gang-rape was 17. Most brutal rapist of Mumbai gang-rape is 18. Worst Young generation I have seen in my life.</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<table-wrap id="pone.0166694.t003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.t003</object-id>
<label>Table 3</label>
<caption>
<title>Examples of events with low activity.</title> <p>The events shown were taken from the “low” category according to <xref ref-type="fig" rid="pone.0166694.g004">Fig 4</xref>.</p>
</caption>
<alternatives>
<graphic id="pone.0166694.t003g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.t003" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Event</th>
<th align="left">Sample Tweets</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left"><bold>Description:</bold><break/>Teen survives hiding in a plane wheel.</td>
<td align="left">@ToniWoemmel: 16-year-old somehow survives flight from California to Hawaii stowed away in planes wheel well: <ext-link ext-link-type="uri" xlink:href="http://t.co/IGiJa60SiK" xlink:type="simple">http://t.co/IGiJa60SiK</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Keywords:</bold><break/>[teen, survives, old, well, skydivers, plane, wheel, flight]</td>
<td align="left">@iOver_think: 38,000 feet at -80F: Teen stowaway survives five-hour California-to-Hawaii flight in wheel well <ext-link ext-link-type="uri" xlink:href="http://t.co/ejXQH9VZyT" xlink:type="simple">http://t.co/ejXQH9VZyT</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Date:</bold><break/>2014-04-21</td>
<td align="left">@TruEntModels: GOD IS GOOD…runaway TEEN hid in plane’s wheel for 5 HOUR flight during FREEZING temps and survived <ext-link ext-link-type="uri" xlink:href="http://t.co/6g6Cqhs9Ib" xlink:type="simple">http://t.co/6g6Cqhs9Ib</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Size:</bold><break/>18,519</td>
<td align="left">@DvdVill: A 16-year-old kid, who was mad at his parents, hid inside a jet wheel and survived flight to Hawaii. <ext-link ext-link-type="uri" xlink:href="http://t.co/c82GbjrfUH" xlink:type="simple">http://t.co/c82GbjrfUH</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Description:</bold><break/>Surveying the damages of recent tornado in Canada.</td>
<td align="left">@Kathleen_Wynne: Visited #Angus today to survey the damage. Thankfully no fatalities or major injuries from recent tornado. <ext-link ext-link-type="uri" xlink:href="http://t.co/xRQyRWg5Vw" xlink:type="simple">http://t.co/xRQyRWg5Vw</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Keywords:</bold><break/>[canada, tornado]</td>
<td align="left">@SunNewsNetwork: PHOTOS &amp; VIDEO: Hundreds displaced after tornado hits Ontario town, destroying homes <ext-link ext-link-type="uri" xlink:href="http://t.co/L38rG6N1a6" xlink:type="simple">http://t.co/L38rG6N1a6</ext-link></td>
</tr>
<tr>
<td align="left"><bold>Date:</bold><break/>2014-06-21</td>
<td align="left">@CBCToronto: Kathleen Wynne is speaking at site of tornado damage in Angus, Ont. now. Watch live here: <ext-link ext-link-type="uri" xlink:href="http://t.co/EDKNUiZo0X" xlink:type="simple">http://t.co/EDKNUiZo0X</ext-link> #cbcto</td>
</tr>
<tr>
<td align="left"><bold>Size:</bold><break/>1,033</td>
<td align="left">@InsuranceBureau: @CTVBarrieNews: Insurance Bureau of Canada is setting up a mobile unit in #Angus today to help residents affected by #Tornado</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p><xref ref-type="table" rid="pone.0166694.t003">Table 3</xref> shows events that have been classified by our methodology in the category of low activity. The first event, about a teen surviving after hiding in the wheel of a airplane, had only a little more than 25% of its messages arriving with high-activity bursts although it had over 18,000 messages. The second event, about the damages caused by a tornado in Canada, did not garner much immediacy in attention of Twitter users, with only 7% of its messages produced with short interarrival times. Most of the messages of this event were well spaced out in time. Even though we cannot say whether or not this event had significant implications in the real-world, we can say that it did not have considerable impact on the Twitter network. The lack of interest could be due to several factors that are currently beyond the scope of this work, ranging from the lack of Twitter users in the locality of the real-world event, to it not being considered urgent by Twitter users. We intend to research the relation between the real-world impact of an event and the network reaction in future work.</p>
<p>
<xref ref-type="fig" rid="pone.0166694.g004">Fig 4</xref> shows the average histograms for events that belong to the high activity, medium-high activity, medium-low activity and low-activity clusters (displayed from left to right and top to bottom). All histograms show a quick decay in average relative frequency (resembling a distribution from the exponential family). In particular, the high-activity group concentrates most of its activity in the shortest interarrival rate, with lower activity groups mostly concentrating their activity in the second bin with slower decay. <xref ref-type="fig" rid="pone.0166694.g005">Fig 5</xref> further characterizes the differences in behavior of the high and low-activity groups, showing that high-activity events concentrate on average 70% fo their activity in the smallest bin (0 sec.), against 8% for low-activity events. In addition, <xref ref-type="fig" rid="pone.0166694.g006">Fig 6</xref> (left) shows the cumulative distribution function (CDF) for each group of events, and <xref ref-type="fig" rid="pone.0166694.g006">Fig 6</xref> (right) shows log(1−CDF). Visual inspection shows a clear difference in how interarrival rates are distributed within each group, however, these figures do not indicate a power-law distribution nor exponential distribution.</p>
<fig id="pone.0166694.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g004</object-id>
<label>Fig 4</label>
<caption>
<title>Average histograms of the high activity, medium-high activity, medium-low activity and low activity clusters in our dataset (from left to right and top to bottom).</title>
<p>All histograms include standard deviation bars and were cut-off at 60 second length for better visibility.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g004" xlink:type="simple"/>
</fig>
<fig id="pone.0166694.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g005</object-id>
<label>Fig 5</label>
<caption>
<title>Scatter plots of the average relative frequencies of interarrival times for the high-activity and low-activity clusters of events (i.e., scatter plots of the histograms in <xref ref-type="fig" rid="pone.0166694.g004">Fig 4</xref> in log-log scale).</title>
<p><italic>y</italic>-axis represents the average relative frequency of social media messages and <italic>x</italic>-axis the interarrival time.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g005" xlink:type="simple"/>
</fig>
<fig id="pone.0166694.g006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.g006</object-id>
<label>Fig 6</label>
<caption>
<title>(Left) Average cumulative distribution function (CDF) for the high activity, medium-high activity, medium-low activity and low activity clusters in our dataset.</title>
<p>(Right) log(1−CDF) for the same clusters.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.g006" xlink:type="simple"/>
</fig>
<p>Further analysis of the high-activity events shows significant differences to other events, in the following aspects: (i) how the information about these events is propagated, (ii) the characteristics of the conversations that they generate, and (iii) how focused users are on the news topic. In detail, high-activity events have a higher fraction of <italic>retweets</italic> (or shares) relative to their overall message volume. On average, a tweet from a high-activity event is retweeted 2.36 times more than a tweet from a low activity event. The most retweeted message in high-activity events is retweeted 7 times more than the most retweeted message in a medium or low activity event. We find that a small set of initial social media posts are propagated quickly and extensively through the network without any rephrasing by the user (just plain forwarding). Intuitively, this seems justified given general topic urgency of high-activity events. Events that are not high-activity did not exhibit these characteristics.</p>
<p>Our research also revealed that high-activity events tend to spark more conversation between users, 33.4% more than other events. This is reflected in the number of <italic>replies</italic> to social media posts. The number of different users that engage with high-activity events is 32.7% higher than in events that are not high-activity. Posts about high-activity events are much more topic focused than in other events. The vocabulary of unique words as well as <italic>hashtags</italic> used in high-activity events is much more narrow than for other events. Medium and low activity events have over 7 times more unique hashtags than high-activity events. This is intuitive, given that if a news item is sensational, people will seldom deviate from the main conversation topic.</p>
<p>In a real-world scenario, in order to predict if an early breaking news story will have a considerable impact in the social network, we will not have enough data to create its activity-based model, i.e., we will not yet know the distribution of the speed at which the social media posts will arrive for the event. For instance, an event can start slowly and later produce an explosive reaction, or start explosively and decay quickly to an overall slower message arrival rate. Still, reliable early prediction of very high-activity news is important in many aspects, from decisions of mass media information coverage, to natural disaster management, brand and political image monitoring, and so on.</p>
<p>For the task of early prediction of high-activity events we use features that are independent of our activity-based model such as the retweets, the sentiment of the posts about the event, etc. These features are computed on the early 5% of messages about the event. The results are an average from a 5-fold cross validation with randomly selected 60% training, 20% validation and 20% test splits. The high-activity events are identified with a precision of 82% using only the earliest 5% of the data of each event (<xref ref-type="table" rid="pone.0166694.t004">Table 4</xref>). Additionally, we were able to identify with high accuracy a considerable percentage of all high-activity events (≈ 46%) at an early stage, with very few false positives (Tables <xref ref-type="table" rid="pone.0166694.t004">4</xref> and <xref ref-type="table" rid="pone.0166694.t005">5</xref>).</p>
<table-wrap id="pone.0166694.t004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.t004</object-id>
<label>Table 4</label>
<caption>
<title>Classification of high-activity events.</title>
</caption>
<alternatives>
<graphic id="pone.0166694.t004g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.t004" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="center" colspan="4">Early 5% Tweets</th>
<th align="center" colspan="4">All Tweets</th>
</tr>
<tr>
<th align="left"/>
<th align="center">FP-Rate</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">ROC-area</th>
<th align="center">FP-Rate</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">ROC-area</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">high-activity</td>
<td align="char" char=".">0.009</td>
<td align="char" char=".">0.819</td>
<td align="char" char=".">0.455</td>
<td align="char" char=".">0.900</td>
<td align="char" char=".">0.01</td>
<td align="char" char=".">0.830</td>
<td align="char" char=".">0.540</td>
<td align="char" char=".">0.945</td>
</tr>
<tr>
<td align="left">non-high-activity</td>
<td align="char" char=".">0.545</td>
<td align="char" char=".">0.954</td>
<td align="char" char=".">0.991</td>
<td align="char" char=".">0.900</td>
<td align="char" char=".">0.460</td>
<td align="char" char=".">0.960</td>
<td align="char" char=".">0.990</td>
<td align="char" char=".">0.945</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<table-wrap id="pone.0166694.t005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0166694.t005</object-id>
<label>Table 5</label>
<caption>
<title>Confusion matrix for high-activity events prediction.</title>
</caption>
<alternatives>
<graphic id="pone.0166694.t005g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.t005" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="center" colspan="2">Early 5% Tweets 2c</th>
<th align="center" colspan="2">All Tweets</th>
</tr>
<tr>
<th align="left"/>
<th align="center">high-activity</th>
<th align="center">non-high-activity</th>
<th align="center">high-activity</th>
<th align="center">non-high-activity</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">high-activity</td>
<td align="center">194</td>
<td align="center">232</td>
<td align="center">230</td>
<td align="center">196</td>
</tr>
<tr>
<td align="left">non-high-activity</td>
<td align="center">43</td>
<td align="center">4,765</td>
<td align="center">47</td>
<td align="center">4,761</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>The precision using only the early tweets is almost as good as using all tweets in the event (0.819 to 0.830). This suggests that the social network somehow acts as a natural filter in separating out the high-activity events fairly early on. The recall goes from 0.455 to 0.540. This indicates that there are some high-activity events which require more data in order to determine what kind of activity they will produce, or events for which activity occurs due to random conditions. A detailed description of the features and different classification settings are provided in the supplementary material.</p>
</sec>
<sec id="sec004" sec-type="conclusions">
<title>Conclusion</title>
<p>We study the characteristics of the activity that real-world news produces in the Twitter social network. In particular, we propose to measure the impact of the real-world news event on the on-line social network by modeling the user activity related to the event using the distribution of their interarrival times between consecutive messages. In our research we observe that the activity triggered by real-world news events follows a similar pattern to that observed in other types of collective reactions to events. This is, by displaying periods of intense activity as well as long periods of inactivity. We further extend this analysis by identifying groups of events that produce much more concentration of high-activity than other events. We show that there are several specific properties that distinguish how high-activity events evolve in Twitter, when comparing them to other events. We design a model for events, based on the codebook approach, that allows us to do unambiguous classification of high-activity events based on the impact displayed by social network. Some notable characteristics of high-activity events are that they are forwarded more often by users, and generate a greater amount of conversation than other events. Social media posts from high-activity news events are much more focused on the news topic. Our experiments show that there are several properties that can suggest early on if an event will have high-activity on the on-line community. We can predict a high number of high-activity events <italic>before</italic> the network has shown any type of explosive reaction to them. This suggests that users are collectively quick at deciding whether an event should receive priority or not. However, there does exist a fraction of events which will create high activity, despite not presenting patterns of other high activity events during their early stages. These events are likely to be affected by other factors, such as random conditions found in the social network at the moment and require further investigation.</p>
</sec>
<sec id="sec005">
<title>Supporting Information</title>
<supplementary-material id="pone.0166694.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0166694.s001" xlink:type="simple">
<label>S1 Appendix</label>
<caption>
<title/>
<p>(PDF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ack>
<p>We thank Gonzalo Navarro (U Chile), Jeanna Matthews (Clarkson Univ.) and Vanessa Murdock (Microsoft) and the reviewers for their valuable feedback and comments.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0166694.ref001">
<label>1</label>
<mixed-citation publication-type="other" xlink:type="simple">
Kwak H, Lee C, Park H, Moon S. What is Twitter, a Social Network or a News Media? In: Proceedings of the 19th International Conference on World Wide Web. WWW’10. New York, NY, USA: ACM; 2010. p. 591–600. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/1772690.1772751" xlink:type="simple">http://doi.acm.org/10.1145/1772690.1772751</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref002">
<label>2</label>
<mixed-citation publication-type="other" xlink:type="simple">
Castillo C, El-Haddad M, Pfeffer J, Stempeck M. Characterizing the Life Cycle of Online News Stories Using Social Media Reactions. In: Proceedings of the 17th ACM Conference on Computer Supported Cooperative Work and Social Computing. CSCW’14. New York, NY, USA: ACM; 2014. p. 211–223. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/2531602.2531623" xlink:type="simple">http://doi.acm.org/10.1145/2531602.2531623</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref003">
<label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Szabo</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Huberman</surname> <given-names>BA</given-names></name>. <article-title>Predicting the Popularity of Online Content</article-title>. <source>Commun ACM</source>. <year>2010</year> <month>Aug</month>;<volume>53</volume>(<issue>8</issue>):<fpage>80</fpage>–<lpage>88</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/1787234.1787254" xlink:type="simple">http://doi.acm.org/10.1145/1787234.1787254</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref004">
<label>4</label>
<mixed-citation publication-type="other" xlink:type="simple">
Lerman K, Hogg T. Using a Model of Social Dynamics to Predict Popularity of News. In: Proceedings of the 19th International Conference on World Wide Web. WWW’10. New York, NY, USA: ACM; 2010. p. 621–630. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/1772690.1772754" xlink:type="simple">http://doi.acm.org/10.1145/1772690.1772754</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref005">
<label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Tatar</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>de Amorim</surname> <given-names>MD</given-names></name>, <name name-style="western"><surname>Fdida</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Antoniadis</surname> <given-names>P</given-names></name>. <article-title>A survey on predicting the popularity of web content</article-title>. <source>Journal of Internet Services and Applications</source>. <year>2014</year>;<volume>5</volume>(<issue>1</issue>):<fpage>1</fpage>–<lpage>20</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1186/s13174-014-0008-y" xlink:type="simple">http://dx.doi.org/10.1186/s13174-014-0008-y</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref006">
<label>6</label>
<mixed-citation publication-type="other" xlink:type="simple">
Pinto H, Almeida JM, Gonçalves MA. Using Early View Patterns to Predict the Popularity of Youtube Videos. In: Proceedings of the Sixth ACM International Conference on Web Search and Data Mining. WSDM’13. New York, NY, USA: ACM; 2013. p. 365–374. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/2433396.2433443" xlink:type="simple">http://doi.acm.org/10.1145/2433396.2433443</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref007">
<label>7</label>
<mixed-citation publication-type="other" xlink:type="simple">
Ahmed M, Spagna S, Huici F, Niccolini S. A Peek into the Future: Predicting the Evolution of Popularity in User Generated Content. In: Proceedings of the Sixth ACM International Conference on Web Search and Data Mining. WSDM’13. New York, NY, USA: ACM; 2013. p. 607–616. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/2433396.2433473" xlink:type="simple">http://doi.acm.org/10.1145/2433396.2433473</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref008">
<label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Li</surname> <given-names>CT</given-names></name>, <name name-style="western"><surname>Shan</surname> <given-names>MK</given-names></name>, <name name-style="western"><surname>Jheng</surname> <given-names>SH</given-names></name>, <name name-style="western"><surname>Chou</surname> <given-names>KC</given-names></name>. <article-title>Exploiting Concept Drift to Predict Popularity of Social Multimedia in Microblogs</article-title>. <source>Inf Sci</source>. <year>2016</year> <month>Apr</month>;<volume>339</volume>(<issue>C</issue>):<fpage>310</fpage>–<lpage>331</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1016/j.ins.2016.01.009" xlink:type="simple">http://dx.doi.org/10.1016/j.ins.2016.01.009</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref009">
<label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Liu</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Zhou</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Zhao</surname> <given-names>X</given-names></name>. <article-title>Understanding News 2.0</article-title>. <source>Inf Manage</source>. <year>2015</year> <month>Nov</month>;<volume>52</volume>(<issue>7</issue>):<fpage>764</fpage>–<lpage>776</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1016/j.im.2015.01.002" xlink:type="simple">http://dx.doi.org/10.1016/j.im.2015.01.002</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref010">
<label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Berger</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Milkman</surname> <given-names>KL</given-names></name>. <article-title>What makes online content viral?</article-title> <source>Journal of Marketing Research</source>. <year>2012</year>;<volume>49</volume>(<issue>2</issue>):<fpage>192</fpage>–<lpage>205</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1509/jmr.10.0353" xlink:type="simple">10.1509/jmr.10.0353</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0166694.ref011">
<label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Iribarren</surname> <given-names>JL</given-names></name>, <name name-style="western"><surname>Moro</surname> <given-names>E</given-names></name>. <article-title>Branching dynamics of viral information spreading</article-title>. <source>Physical Review E</source>. <year>2011</year>;<volume>84</volume>(<issue>4</issue>):<fpage>046116</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1103/PhysRevE.84.046116" xlink:type="simple">10.1103/PhysRevE.84.046116</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0166694.ref012">
<label>12</label>
<mixed-citation publication-type="other" xlink:type="simple">Guerini M, Strapparava C, Özbal G. Exploring Text Virality in Social Networks. In: ICWSM; 2011.</mixed-citation>
</ref>
<ref id="pone.0166694.ref013">
<label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Mills</surname> <given-names>AJ</given-names></name>. <article-title>Virality in social media: the SPIN framework</article-title>. <source>Journal of public affairs</source>. <year>2012</year>;<volume>12</volume>(<issue>2</issue>):<fpage>162</fpage>–<lpage>169</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1002/pa.1418" xlink:type="simple">10.1002/pa.1418</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0166694.ref014">
<label>14</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Gaugaz</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Siehndel</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Demartini</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Iofciu</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Georgescu</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Henze</surname> <given-names>N</given-names></name>. <chapter-title>Predicting the future impact of news events</chapter-title>. In: <source>Advances in Information Retrieval</source>. <publisher-name>Springer</publisher-name>; <year>2012</year>. p. <fpage>50</fpage>–<lpage>62</lpage>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref015">
<label>15</label>
<mixed-citation publication-type="other" xlink:type="simple">Telegraph. Chile News;. <ext-link ext-link-type="uri" xlink:href="http://www.telegraph.co.uk/news/worldnews/southamerica/chile/" xlink:type="simple">http://www.telegraph.co.uk/news/worldnews/southamerica/chile/</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref016">
<label>16</label>
<mixed-citation publication-type="other" xlink:type="simple">
dos Reis JC, Benevenuto F, de Melo POSV, Prates RO, Kwak H, An J. Breaking the News: First Impressions Matter on Online News. CoRR. 2015;abs/1503.07921. Available from: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1503.07921" xlink:type="simple">http://arxiv.org/abs/1503.07921</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref017">
<label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Barabasi</surname> <given-names>AL</given-names></name>. <article-title>The origin of bursts and heavy tails in human dynamics</article-title>. <source>Nature</source>. <year>2005</year>;<volume>435</volume>(<issue>7039</issue>):<fpage>207</fpage>–<lpage>211</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1038/nature03459" xlink:type="simple">10.1038/nature03459</ext-link></comment> <object-id pub-id-type="pmid">15889093</object-id></mixed-citation>
</ref>
<ref id="pone.0166694.ref018">
<label>18</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Karsai</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Kaski</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Barabási</surname> <given-names>AL</given-names></name>, <name name-style="western"><surname>Kertész</surname> <given-names>J</given-names></name>. <article-title>Universal features of correlated bursty behaviour</article-title>. <source>Scientific reports</source>. <year>2012</year>;<volume>2</volume>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1038/srep00397" xlink:type="simple">10.1038/srep00397</ext-link></comment> <object-id pub-id-type="pmid">22563526</object-id></mixed-citation>
</ref>
<ref id="pone.0166694.ref019">
<label>19</label>
<mixed-citation publication-type="other" xlink:type="simple">
Gao, L, Song, C, Gao, Z, Barabási, AL, Bagrow, JP, Wang, D. Quantifying information flow during emergencies. arXiv preprint arXiv:14011274. 2014;.</mixed-citation>
</ref>
<ref id="pone.0166694.ref020">
<label>20</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Yan</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Wu</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Liu</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Li</surname> <given-names>X</given-names></name>. <chapter-title>Information propagation in online social network based on human dynamics</chapter-title>. In: <source>Abstract and Applied Analysis. vol. 2013</source>. <publisher-name>Hindawi Publishing Corporation</publisher-name>; <year>2013</year>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref021">
<label>21</label>
<mixed-citation publication-type="other" xlink:type="simple">Fei-Fei L, Perona P. A Bayesian hierarchical model for learning natural scene categories. In: Computer Vision and Pattern Recognition, 2005. CVPR 2005. IEEE Computer Society Conference on. vol. 2; 2005. p. 524–531 vol. 2.</mixed-citation>
</ref>
<ref id="pone.0166694.ref022">
<label>22</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Vaizman</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>McFee</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Lanckriet</surname> <given-names>G</given-names></name>. <article-title>Codebook-based Audio Feature Representation for Music Information Retrieval</article-title>. <source>IEEE/ACM Trans Audio, Speech and Lang Proc</source>. <year>2014</year> <month>Oct</month>;<volume>22</volume>(<issue>10</issue>):<fpage>1483</fpage>–<lpage>1493</lpage>. Available from: <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1109/TASLP.2014.2337842" xlink:type="simple">http://dx.doi.org/10.1109/TASLP.2014.2337842</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref023">
<label>23</label>
<mixed-citation publication-type="other" xlink:type="simple">Twitter Inc;. <ext-link ext-link-type="uri" xlink:href="https://www.twitter.com" xlink:type="simple">https://www.twitter.com</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref024">
<label>24</label>
<mixed-citation publication-type="other" xlink:type="simple">Twitter API;. <ext-link ext-link-type="uri" xlink:href="https://dev.twitter.com" xlink:type="simple">https://dev.twitter.com</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref025">
<label>25</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Tan</surname> <given-names>PN</given-names></name>, <name name-style="western"><surname>Steinbach</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Kumar</surname> <given-names>V</given-names></name>. <source>Introduction to Data Mining</source>, (<edition>First Edition</edition>). <publisher-loc>Boston, MA, USA</publisher-loc>: <publisher-name>Addison-Wesley Longman Publishing Co., Inc.</publisher-name>; <year>2005</year>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref026">
<label>26</label>
<mixed-citation publication-type="other" xlink:type="simple">Tatar A, Leguay J, Antoniadis P, Limbourg A, de Amorim MD, Fdida S. Predicting the Popularity of Online Articles Based on User Comments. In: Proceedings of the International Conference on Web Intelligence, Mining and Semantics. WIMS’11. New York, NY, USA: ACM; 2011. p. 67:1–67:8. Available from: <ext-link ext-link-type="uri" xlink:href="http://doi.acm.org/10.1145/1988688.1988766" xlink:type="simple">http://doi.acm.org/10.1145/1988688.1988766</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0166694.ref027">
<label>27</label>
<mixed-citation publication-type="other" xlink:type="simple">Suh B, Hong L, Pirolli P, Chi EH. Want to be retweeted? large scale analytics on factors impacting retweet in twitter network. In: Social computing (socialcom), 2010 ieee second international conference on. IEEE; 2010. p. 177–184.</mixed-citation>
</ref>
</ref-list>
</back>
</article>