<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS ONE</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pone.0281773</article-id>
<article-id pub-id-type="publisher-id">PONE-D-22-14263</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Medical conditions</subject><subj-group><subject>Infectious diseases</subject><subj-group><subject>Viral diseases</subject><subj-group><subject>COVID 19</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Epidemiology</subject><subj-group><subject>Pandemics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Vaccination and immunization</subject><subj-group><subject>Vaccine development</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Vaccination and immunization</subject><subj-group><subject>Vaccine development</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Public and occupational health</subject><subj-group><subject>Preventive medicine</subject><subj-group><subject>Vaccination and immunization</subject><subj-group><subject>Vaccine development</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Medical conditions</subject><subj-group><subject>Infectious diseases</subject><subj-group><subject>Infectious disease control</subject><subj-group><subject>Vaccines</subject><subj-group><subject>Viral vaccines</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Microbiology</subject><subj-group><subject>Virology</subject><subj-group><subject>Viral vaccines</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Sociology</subject><subj-group><subject>Communications</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>People and places</subject><subj-group><subject>Population groupings</subject><subj-group><subject>Age groups</subject><subj-group><subject>Children</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>People and places</subject><subj-group><subject>Population groupings</subject><subj-group><subject>Families</subject><subj-group><subject>Children</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Vaccination and immunization</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Vaccination and immunization</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Public and occupational health</subject><subj-group><subject>Preventive medicine</subject><subj-group><subject>Vaccination and immunization</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Probability theory</subject><subj-group><subject>Probability distribution</subject></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Dear Pandemic: A topic modeling analysis of COVID-19 information needs among readers of an online science communication campaign</article-title>
<alt-title alt-title-type="running-head">Topic modeling analysis of COVID-19 information needs</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-5803-4164</contrib-id>
<name name-style="western">
<surname>Golos</surname>
<given-names>Aleksandra M.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Guntuku</surname>
<given-names>Sharath Chandra</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Piltch-Loeb</surname>
<given-names>Rachael</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff004"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff005"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Leininger</surname>
<given-names>Lindsey J.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff006"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Simanek</surname>
<given-names>Amanda M.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff007"><sup>7</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Kumar</surname>
<given-names>Aparna</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff008"><sup>8</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-3444-2192</contrib-id>
<name name-style="western">
<surname>Albrecht</surname>
<given-names>Sandra S.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff009"><sup>9</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Dowd</surname>
<given-names>Jennifer Beam</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff010"><sup>10</sup></xref>
<xref ref-type="aff" rid="aff011"><sup>11</sup></xref>
<xref ref-type="aff" rid="aff012"><sup>12</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Jones</surname>
<given-names>Malia</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff013"><sup>13</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Buttenheim</surname>
<given-names>Alison M.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff014"><sup>14</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001"><label>1</label> <addr-line>Department of Family and Community Health, School of Nursing, University of Pennsylvania, Philadelphia, PA, United States of America</addr-line></aff>
<aff id="aff002"><label>2</label> <addr-line>Department of Computer and Information Science, School of Engineering and Applied Science, University of Pennsylvania, Philadelphia, PA, United States of America</addr-line></aff>
<aff id="aff003"><label>3</label> <addr-line>Leonard Davis Institute of Health Economics, University of Pennsylvania, Philadelphia, PA, United States of America</addr-line></aff>
<aff id="aff004"><label>4</label> <addr-line>Department of Biostatistics, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA, United States of America</addr-line></aff>
<aff id="aff005"><label>5</label> <addr-line>Emergency Preparedness Research Evaluation and Practice Program, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA, United States of America</addr-line></aff>
<aff id="aff006"><label>6</label> <addr-line>Tuck School of Business, Dartmouth College, Hanover, NH, United States of America</addr-line></aff>
<aff id="aff007"><label>7</label> <addr-line>Joseph J. Zilber School of Public Health, University of Wisconsin-Milwaukee, Milwaukee, WI, United States of America</addr-line></aff>
<aff id="aff008"><label>8</label> <addr-line>College of Nursing, Thomas Jefferson University, Philadelphia, PA, United States of America</addr-line></aff>
<aff id="aff009"><label>9</label> <addr-line>Department of Epidemiology, Mailman School of Public Health, Columbia University, New York, NY, United States of America</addr-line></aff>
<aff id="aff010"><label>10</label> <addr-line>Leverhulme Centre for Demographic Science, University of Oxford, Oxford, United Kingdom</addr-line></aff>
<aff id="aff011"><label>11</label> <addr-line>Department of Sociology, University of Oxford, Oxford, United Kingdom</addr-line></aff>
<aff id="aff012"><label>12</label> <addr-line>Nuffield College, University of Oxford, Oxford, United Kingdom</addr-line></aff>
<aff id="aff013"><label>13</label> <addr-line>Applied Population Laboratory, Department of Community and Environmental Sociology, College of Agricultural and Life Sciences, University of Wisconsin-Madison, Madison, WI, United States of America</addr-line></aff>
<aff id="aff014"><label>14</label> <addr-line>Center for Health Incentives and Behavioral Economics, University of Pennsylvania, Philadelphia, PA, United States of America</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Taskin</surname>
<given-names>Nazim</given-names>
</name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1"><addr-line>Bogazici University, TURKEY</addr-line></aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">agolos@nursing.upenn.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>3</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>18</volume>
<issue>3</issue>
<elocation-id>e0281773</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>5</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>1</day>
<month>2</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-year>2023</copyright-year>
<copyright-holder>Golos et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0281773"/>
<abstract>
<sec id="sec001">
<title>Background</title>
<p>The COVID-19 pandemic was accompanied by an “infodemic”–an overwhelming excess of accurate, inaccurate, and uncertain information. The social media-based science communication campaign Dear Pandemic was established to address the COVID-19 infodemic, in part by soliciting submissions from readers to an online question box. Our study characterized the information needs of Dear Pandemic’s readers by identifying themes and longitudinal trends among question box submissions.</p>
</sec>
<sec id="sec002">
<title>Methods</title>
<p>We conducted a retrospective analysis of questions submitted from August 24, 2020, to August 24, 2021. We used Latent Dirichlet Allocation topic modeling to identify 25 topics among the submissions, then used thematic analysis to interpret the topics based on their top words and submissions. We used t-Distributed Stochastic Neighbor Embedding to visualize the relationship between topics, and we used generalized additive models to describe trends in topic prevalence over time.</p>
</sec>
<sec id="sec003">
<title>Results</title>
<p>We analyzed 3839 submissions, 90% from United States-based readers. We classified the 25 topics into 6 overarching themes: ‘Scientific and Medical Basis of COVID-19,’ ‘COVID-19 Vaccine,’ ‘COVID-19 Mitigation Strategies,’ ‘Society and Institutions,’ ‘Family and Personal Relationships,’ and ‘Navigating the COVID-19 Infodemic.’ Trends in topics about viral variants, vaccination, COVID-19 mitigation strategies, and children aligned with the news cycle and reflected the anticipation of future events. Over time, vaccine-related submissions became increasingly related to those surrounding social interaction.</p>
</sec>
<sec id="sec004">
<title>Conclusions</title>
<p>Question box submissions represented distinct themes that varied in prominence over time. Dear Pandemic’s readers sought information that would not only clarify novel scientific concepts, but would also be timely and practical to their personal lives. Our question box format and topic modeling approach offers science communicators a robust methodology for tracking, understanding, and responding to the information needs of online audiences.</p>
</sec>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution>National Institutes of Health, National Institute on Minority Health and Health Disparities</institution>
</funding-source>
<award-id>R01MD018340</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Guntuku</surname>
<given-names>Sharath Chandra</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="award002">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100000275</institution-id>
<institution>Leverhulme Trust</institution>
</institution-wrap>
</funding-source>
<principal-award-recipient>
<name name-style="western">
<surname>Dowd</surname>
<given-names>Jennifer Beam</given-names>
</name>
</principal-award-recipient>
</award-group>
<funding-statement>Research supported by this publication was partially supported by funding to author SCG from the National Institutes of Health, National Institute on Minority Health and Health Disparities R01MD018340. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health. This work has also been supported by funding to author JBD from the Leverhulme Trust (Centre grant). The funder had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="1"/>
<page-count count="15"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All relevant data are within the paper and its <xref ref-type="sec" rid="sec022">Supporting Information</xref> files.</meta-value>
</custom-meta>
<custom-meta id="outbreaks">
<meta-name>Outbreaks</meta-name>
<meta-value>COVID-19</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec005" sec-type="intro">
<title>Introduction</title>
<p>The COVID-19 pandemic spurred a global communication crisis that the World Health Organization (WHO) named an “infodemic.” Defined as a problem of too much information–including accurate information, rapidly evolving guidance, and misinformation–the COVID-19 infodemic eroded trust in science and undermined compliance with public health measures [<xref ref-type="bibr" rid="pone.0281773.ref001">1</xref>]. As seeking information became more complicated, it also became more consequential: exposure to credible COVID-19 information may have increased worry about disease susceptibility and severity, prompting people to engage in COVID-19 mitigation strategies [<xref ref-type="bibr" rid="pone.0281773.ref002">2</xref>]. Conversely, exposure to misinformation may have led people to heuristic processing and information avoidance, entrenching detrimental health beliefs [<xref ref-type="bibr" rid="pone.0281773.ref003">3</xref>]. The WHO developed an infodemic management framework to help science communicators address this crisis; it recommends conducting information surveillance, translating expert knowledge into accessible language, bolstering scientific literacy, and fact-checking information [<xref ref-type="bibr" rid="pone.0281773.ref004">4</xref>, <xref ref-type="bibr" rid="pone.0281773.ref005">5</xref>]. These efforts should happen through two-way, rather than one-way, engagement with the public [<xref ref-type="bibr" rid="pone.0281773.ref006">6</xref>].</p>
<p>The science communication campaign Dear Pandemic was established in March 2020 to address the COVID-19 infodemic via two-way engagement on Instagram, Facebook, and Twitter. Run by an interdisciplinary team of clinicians and researchers, Dear Pandemic is responsive to readers’ information needs, translating emerging news and science into actionable guidance. To more effectively engage with readers, Dear Pandemic opened an online “question box” in August 2020. Readers submit questions via a simple online webform, and Dear Pandemic contributors use the questions to inform upcoming content. This approach was highly successful; by late 2021, the campaign had 180,000 regular readers from over 60 countries, and its content reached over 1 million unique monthly views.</p>
<p>Dear Pandemic’s question box presents a unique opportunity to understand the information needs of an engaged online audience during the COVID-19 pandemic. Though other studies have analyzed online COVID-19 discourse, most relied on broad sets of keywords to identify relevant content or did not discern between different types of discourse [<xref ref-type="bibr" rid="pone.0281773.ref007">7</xref>–<xref ref-type="bibr" rid="pone.0281773.ref010">10</xref>]. Among the studies that focused on information-seeking, Mangono et al. and Chan &amp; Chua used Google search volume as a proxy for information needs during the onset of the pandemic [<xref ref-type="bibr" rid="pone.0281773.ref011">11</xref>, <xref ref-type="bibr" rid="pone.0281773.ref012">12</xref>]. Kim &amp; Oh used a dataset of COVID-19 related posts on the South Korean question-and-answer platform Naver Knowledge iN, allowing for a more direct analysis of information needs [<xref ref-type="bibr" rid="pone.0281773.ref013">13</xref>]. However, their dataset was limited to posts from February 1 to October 31, 2020, and questions were answered by the lay public rather than by experts with specialized, fact-checked knowledge. Little is therefore known about discourse written for the purpose of seeking credible information about COVID-19. Our study makes significant contributions to the literature on COVID-19 information needs by analyzing data through August 2021 (covering the development and rollout of the COVID-19 vaccine) and focusing on audiences who sought information from trusted science communicators.</p>
<p>Many of the aforementioned studies have used Latent Dirichlet Allocation (LDA) topic modeling to analyze online COVID-19 discourse. LDA is a quantitative method that groups words with high probabilities of co-occurrence into topics. Each text input to the model is described by a probability distribution of topics into which it falls, and each topic is described by a distribution of highly-associated words [<xref ref-type="bibr" rid="pone.0281773.ref014">14</xref>]. LDA has often been used to analyze large-scale datasets (e.g., tweets) [<xref ref-type="bibr" rid="pone.0281773.ref007">7</xref>, <xref ref-type="bibr" rid="pone.0281773.ref009">9</xref>, <xref ref-type="bibr" rid="pone.0281773.ref010">10</xref>], but it is also well-suited for short-text data and small corpora [<xref ref-type="bibr" rid="pone.0281773.ref015">15</xref>], making it appropriate for our dataset. Murakami et al. described the advantages of LDA topic modeling in the exploration of a specialized corpus, highlighting its utility in identifying different types of content, differentiating multiple senses of words, and examining chronological changes [<xref ref-type="bibr" rid="pone.0281773.ref016">16</xref>]. A further advantage of LDA is that it allows for researcher input in naming, interpreting, and categorizing topics, providing a deeper understanding of the results. We harnessed this by using established qualitative methods of thematic analysis to identify overarching patterns, or “themes,” among the topics [<xref ref-type="bibr" rid="pone.0281773.ref017">17</xref>].</p>
<p>In this study, we use LDA and thematic analysis to 1) identify topics among reader-submitted questions to Dear Pandemic’s question box; 2) visualize the association between topics; and 3) describe trends in topic prevalence over time. By characterizing the information needs of Dear Pandemic’s audience, we not only provide other science communicators with insights about information-seeking during COVID-19, but also illustrate how our question box format and topic modeling approach can be useful for infodemic management.</p>
</sec>
<sec id="sec006" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="sec007">
<title>Data collection and pre-processing</title>
<p><xref ref-type="supplementary-material" rid="pone.0281773.s005">S1 Fig</xref> displays the research process, from data collection to visualization. Our data consisted of submissions to Dear Pandemic’s question box from August 24, 2020, to August 24, 2021. We chose a year-long period in order to cover a wide range of the rapidly shifting COVID-19 information landscape, while also limiting the inclusion of extraneous questions that were submitted in late 2021 as Dear Pandemic broadened its scope to other health and science topics. To assemble our corpus, we retrospectively collected all 3947 questions that readers submitted via webform during the analysis period. The webform also included a field for readers to report their geographic location, which we categorized by US state and/or country. After removing 108 duplicate or non-English-language questions, we obtained a final corpus of 3839 submissions.</p>
<p>We pre-processed the text of each submission by converting it to lowercase and removing numbers, special characters, stop words (i.e., common words with little meaning), and URLs. Finally, we tokenized (i.e., split the submission text into individual words) and lemmatized (i.e., converted words into their root form) the text using the Python library spaCy [<xref ref-type="bibr" rid="pone.0281773.ref018">18</xref>]. The mean (SD) submission length was 66.7 (61.7) words originally and 31.5 (29.7) tokens after pre-processing.</p>
</sec>
<sec id="sec008">
<title>Topic modeling</title>
<p>To analyze submissions, we used an LDA topic modeling algorithm provided by the Differential Language Analysis ToolKit (DLATK) Mallet interface [<xref ref-type="bibr" rid="pone.0281773.ref019">19</xref>, <xref ref-type="bibr" rid="pone.0281773.ref020">20</xref>]. We adjusted one parameter (alpha = 0.3) to favor fewer topics per document (i.e., submission), reflecting the fact that our documents were relatively short in length (compared to documents such as book chapters or news articles) and were thus likely to contain fewer topics. All other LDA parameters were kept at their default.</p>
<p>The number of topics to generate, <italic>k</italic>, must be manually pre-specified in LDA models. This parameter can be optimized by comparing the results of models run with different values of <italic>k</italic>, using quantitative topic coherence metrics and qualitative measures of interpretability. We estimated a series of models with 3 &lt; <italic>k</italic> &lt; 50 topics and calculated coherence using the C<sub>v</sub> method [<xref ref-type="bibr" rid="pone.0281773.ref021">21</xref>]. We also reviewed the topics generated in each iteration for semantic validity (whether topic meanings could be clearly discerned by their associated words) and granularity (whether topics were broad or specific). While the 9-topic model had the highest C<sub>v</sub> score (<xref ref-type="supplementary-material" rid="pone.0281773.s006">S2 Fig</xref>), we judged the topics to be insufficiently granular. We reached the consensus that the 25-topic model was optimal based on its semantic validity, granularity, and C<sub>v</sub> score.</p>
<p>After the model was finalized, 5 of the authors with expertise in qualitative research conducted a thematic analysis of the topics, following the process described by Braun &amp; Clarke [<xref ref-type="bibr" rid="pone.0281773.ref017">17</xref>]. To integrate LDA and thematic analysis, we followed the best practice of identifying the words and documents that were most highly associated with each topic [<xref ref-type="bibr" rid="pone.0281773.ref022">22</xref>]. We calculated log-likelihood word frequencies to identify the top 50 words in each topic, and we used submission-level probability distributions to identify the top 10 submissions in each topic. The 5 authors independently familiarized themselves with the data; generated names and descriptions for each of the 25 topics using the top words and submissions; and grouped the 25 topics into candidate themes based on common features and meanings. The authors identified 11 candidate themes, which were compared, mapped, and consolidated until 6 final themes were generated. The names and interpretations of the topics and themes were refined until a consensus among all authors was reached.</p>
<p>Finally, we visualized the submissions using a t-distributed stochastic neighbor embedding (t-SNE) algorithm, which represented each submission’s 25-dimensional topic probability distribution in a 2-dimensional space. To examine trends over time, we computed each topic’s average probability by day and fit generalized additive models to the data. We visualized the data and conducted trend analyses using R version 4.0.3. Our study was determined to not meet the definition of human subjects research by the University of Pennsylvania Institutional Review Board, waiving the requirement for informed consent.</p>
</sec>
</sec>
<sec id="sec009" sec-type="results">
<title>Results</title>
<sec id="sec010">
<title>Characteristics of submissions to the Dear Pandemic question box</title>
<p>The topic model included 3839 question box submissions. The number of daily submissions ranged from 1 to 32, and the most active month was March 2021, with 592 submissions (<xref ref-type="supplementary-material" rid="pone.0281773.s007">S3 Fig</xref>). 90.0% of submissions came from US-based readers; the 5 states with the most submissions were Wisconsin (417), Texas (333), Pennsylvania (265), New York (244), and California (234) (<xref ref-type="supplementary-material" rid="pone.0281773.s002">S1 Data</xref>). Other countries with 5 or more submissions were Canada (96), the United Kingdom (67), Australia (12), Mexico (6), South Africa (5), India (5), and Germany (5). These frequencies generally map to the location of Dear Pandemic readers on social media. As Dear Pandemic periodically promoted the question box during media interviews and events, submission volume and geographic distribution may not have been solely organic.</p>
</sec>
<sec id="sec011">
<title>Themes and topics identified among submissions to the Dear Pandemic question box</title>
<p><xref ref-type="table" rid="pone.0281773.t001">Table 1</xref> provides an overview of the 25 topics, including each topic’s 20 most frequently associated words, an example submission, the number of submissions for which each topic was the highest-probability topic, and each topic’s mean probability. Topic probabilities were low on average, highly variable, and highly right-skewed; relatively few submissions had high probabilities of being represented by a single topic (<xref ref-type="supplementary-material" rid="pone.0281773.s003">S2 Data</xref>). The following results describe our qualitative interpretation of the topics and themes.</p>
<table-wrap id="pone.0281773.t001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0281773.t001</object-id>
<label>Table 1</label> <caption><title>Overview of themes and topics among submissions to the Dear Pandemic question box.</title></caption>
<alternatives>
<graphic id="pone.0281773.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.t001" xlink:type="simple"/>
<table>
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<thead>
<tr>
<th align="left">Topic name</th>
<th align="left">Most frequently associated words</th>
<th align="left">Example submission</th>
<th align="left">Number (%) of submissions<xref ref-type="table-fn" rid="t001fn001"><sup>a</sup></xref></th>
<th align="left">Mean (SD) topic probability<xref ref-type="table-fn" rid="t001fn002"><sup>b</sup></xref></th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="5"><bold>Theme 1: Scientific and medical basis of COVID-19</bold></td>
</tr>
<tr>
<td align="left">Immunology</td>
<td align="left">immune, antibody, disease, system, vaccination, response, cell, spike, blood, treatment, protein, autoimmune, body, mrna, produce, medication, patient, treat, specifically, develop</td>
<td align="left">Can the Pfizer vaccine alter or weaken my existing cross-reactive antibodies / immune system?</td>
<td align="left">209 (5.34)</td>
<td align="left">4.01 (6.70)</td>
</tr>
<tr>
<td align="left">Disease process</td>
<td align="left">covid, long, term, infection, symptom, mild, severe, concern, case, effect, prevent, issue, health, hear, may, heart, damage, disease, serious, illness</td>
<td align="left">How do you know if you have heart damage post covid?</td>
<td align="left">176 (4.58)</td>
<td align="left">4.40 (5.33)</td>
</tr>
<tr>
<td align="left">Immunity and transmission</td>
<td align="left">virus, people, immunity, still, person, vaccinate, infection, infect, someone, could, even, spread, transmit, other, herd, understand, contract, asymptomatic, viral, likely</td>
<td align="left">Why is there little to no talk about natural immunity from exposure or infection in terms of reaching herd-immunity?</td>
<td align="left">173 (4.51)</td>
<td align="left">4.65 (5.23)</td>
</tr>
<tr>
<td align="left">Viral variants</td>
<td align="left">variant, new, delta, virus, concern, transmission, spread, come, strain, sar, evidence, mean, possible, base, show, cov, protect, coronavirus, uk, infect</td>
<td align="left">Is the delta variant more severe for children than the other variants?</td>
<td align="left">142 (3.70)</td>
<td align="left">3.41 (5.80)</td>
</tr>
<tr>
<td align="left">Other medical conditions</td>
<td align="left">get, vaccine, covid, flu, tell, sick, able, wait, shoot, try, already, mom, never, type, yet, thing, eligible, chance, turn, stop</td>
<td align="left">Should we get the flu shot earlier this year? We usually get ours in October, but I’m nervous about the double whammy of seasonal flu and Covid in schools.</td>
<td align="left">89 (2.32)</td>
<td align="left">3.70 (4.18)</td>
</tr>
<tr>
<td align="left" colspan="5"><bold>Theme 2: COVID-19 vaccine</bold></td>
</tr>
<tr>
<td align="left">Dosage and timing</td>
<td align="left">dose, get, first, second, shot, pfizer, week, moderna, receive, one, two, nd, booster, shoot, wait, month, arm, effective, protection, schedule</td>
<td align="left">The vaccine is supposed to be effective for six months. Is that six months from the first shot, the second shot, or from the two weeks after the second shot?</td>
<td align="left">281 (7.32)</td>
<td align="left">5.14 (7.60)</td>
</tr>
<tr>
<td align="left">Development and rollout</td>
<td align="left">vaccine, trial, study, available, pfizer, efficacy, mrna, johnson, approve, moderna, safety, receive, datum, pregnant, effective, currently, approval, fda, develop, woman</td>
<td align="left">Dear Pandemic, what is the difference between emergency authorization for a vaccine, and "regular" authorization?</td>
<td align="left">212 (5.52)</td>
<td align="left">4.68 (6.27)</td>
</tr>
<tr>
<td align="left">Behavior surrounding vaccination</td>
<td align="left">vaccinated, vaccinate, fully, people, unvaccinated, indoor, child, household, cdc, person, guideline, adult, guidance, visit, without, different, husband, unmasked, grandparent, change</td>
<td align="left">Can fully vaccinated people socialize with other fully vaccinated people? With or without masks? How about eating in a restaurant?</td>
<td align="left">197 (5.13)</td>
<td align="left">4.64 (5.72)</td>
</tr>
<tr>
<td align="left">Safety and side effects</td>
<td align="left">vaccine, effect, side, reaction, take, report, cause, receive, response, woman, allergy, concern, might, allergic, affect, bad, experience, body, pain, immune</td>
<td align="left">Are the vaccine doses for women smaller than for men? I know more women that have adverse reactions to the vaccines than men.</td>
<td align="left">182 (4.74)</td>
<td align="left">3.93 (6.30)</td>
</tr>
<tr>
<td align="left" colspan="5"><bold>Theme 3: COVID-19 mitigation strategies</bold></td>
</tr>
<tr>
<td align="left">Socializing safely</td>
<td align="left">mask, wear, distance, outside, kid, safe, outdoor, indoor, outdoors, social, play, distancing, even, foot, other, etc, summer, friend, apart, everyone</td>
<td align="left">I take brisk walks outside with a friend. We stay approximately six feet apart but do not wear masks. How safe is this?</td>
<td align="left">232 (6.04)</td>
<td align="left">4.60 (6.60)</td>
</tr>
<tr>
<td align="left">Testing and isolation</td>
<td align="left">test, covid, positive, day, quarantine, symptom, negative, testing, expose, antibody, result, week, exposure, contact, someone, pcr, month, rapid, tell, recently</td>
<td align="left">I am vaccinated and recently tested positive. My only symptom is no taste/smell. Do I still have to quarantine for 10 days?</td>
<td align="left">229 (5.97)</td>
<td align="left">4.57 (6.41)</td>
</tr>
<tr>
<td align="left">Ventilation</td>
<td align="left">air, office, room, open, work, window, hour, can, not, indoor, small, someone, space, door, building, share, close, house, ventilation, line</td>
<td align="left">I live in an apartment building. Is it true that I can catch covid from air that is in other apartments around me that comes through my air ducts? Should I install air filters?</td>
<td align="left">128 (3.33)</td>
<td align="left">3.22 (4.90)</td>
</tr>
<tr>
<td align="left">Sanitation and hygiene</td>
<td align="left">hand, need, grocery, eat, surface, use, food, store, cold, touch, wash, really, every, transmission, leave, place, avoid, fomite, restaurant, thing</td>
<td align="left">What’s the latest on fomite transmission? Do we need to wash our hands when we get the mail?</td>
<td align="left">117 (3.05)</td>
<td align="left">2.98 (5.61)</td>
</tr>
<tr>
<td align="left">Masks</td>
<td align="left">mask, wear, well, protect, protection, face, public, good, double, recommend, course, cloth, effective, recommendation, require, clear, way, kn, provide, talk</td>
<td align="left">Which masks provide the most protection for kids in school? (Filter and cloth, medical or KN95).</td>
<td align="left">91 (2.37)</td>
<td align="left">3.01 (4.88)</td>
</tr>
<tr>
<td align="left" colspan="5"><bold>Theme 4: Society and institutions</bold></td>
</tr>
<tr>
<td align="left">Education</td>
<td align="left">school, student, kid, teacher, person, back, classroom, require, decision, class, district, open, return, reopen, send, full, learn, fall, guidance, elementary</td>
<td align="left">What mitigating steps can school leaders / individual teachers put in place to reduce spread when schools reopen to all pupils?</td>
<td align="left">163 (4.25)</td>
<td align="left">3.54 (5.23)</td>
</tr>
<tr>
<td align="left">Healthcare and policy</td>
<td align="left">health, community, care, worker, county, state, public, medical, follow, guidance, rate, area, healthcare, local, group, right, patient, hospital, continue, number</td>
<td align="left">Are health care providers still getting covid? Is their PPE really working? Do all health care providers have sufficient PPE now?</td>
<td align="left">87 (2.27)</td>
<td align="left">2.91 (4.09)</td>
</tr>
<tr>
<td align="left">Social participation</td>
<td align="left">travel, vaccination, state, month, post, early, return, last, country, summer, may, event, end, soon, plan, late, expect, safely, fall, look</td>
<td align="left">When will the US / Canadian border be open and allow for travel without quarantine restrictions?</td>
<td align="left">73 (1.90)</td>
<td align="left">3.07 (4.30)</td>
</tr>
<tr>
<td align="left" colspan="5"><bold>Theme 5: Family and personal relationships</bold></td>
</tr>
<tr>
<td align="left">Interacting with family</td>
<td align="left">family, would, visit, safe, see, home, stay, parent, we, member, quarantine, plan, precaution, live, travel, together, come, house, husband, possible</td>
<td align="left">Can I visit my family for Christmas if we’ve all been taking COVID seriously?</td>
<td align="left">205 (5.34)</td>
<td align="left">4.48 (5.50)</td>
</tr>
<tr>
<td align="left">Children and parenting</td>
<td align="left">kid, child, year, old, young, adult, age, parent, son, healthy, small, able, daughter, great, start, group, fall, toddler, eligible, yo</td>
<td align="left">What are parents of young children supposed to do with the new CDC guidance about masklessness?</td>
<td align="left">175 (4.56)</td>
<td align="left">4.36 (5.29)</td>
</tr>
<tr>
<td align="left">Personal relationships</td>
<td align="left">home, go, we, daughter, want, husband, stay, also, friend, day, back, since, son, come, live, two, start, around, couple, march</td>
<td align="left">Can I let my neighbors (children) who go to in-person school and play with friends pet my dog?</td>
<td align="left">87 (2.27)</td>
<td align="left">3.69 (3.92)</td>
</tr>
<tr>
<td align="left" colspan="5"><bold>Theme 6: Navigating the COVID-19 infodemic</bold></td>
</tr>
<tr>
<td align="left">Verifying information</td>
<td align="left">say, see, article, please, make, post, study, read, something, find, explain, claim, dr, hear, true, information, link, ask, news, cdc</td>
<td align="left">Can you comment on the science behind the new claims from Geert Vanden Bossche?</td>
<td align="left">154 (4.01)</td>
<td align="left">4.26 (5.89)</td>
</tr>
<tr>
<td align="left">Feedback</td>
<td align="left">thank, question, love, much, post, answer, wonder, think, appreciate, good, pandemic, information, share, info, girl, thought, help, address, hi, provide</td>
<td align="left">This is just a thank you for your answers and you really helpful job! Thank you!</td>
<td align="left">144 (3.75)</td>
<td align="left">4.22 (5.15)</td>
</tr>
<tr>
<td align="left">Data and statistics</td>
<td align="left">case, death, rate, number, datum, seem, infection, spread, report, less, percentage, low, die, read, increase, age, example, hospital, information, among</td>
<td align="left">Should we paying attention to case count per 100,000 or positivity rates to gauge community spread? What does it mean when positivity is low but case count high?</td>
<td align="left">137 (3.57)</td>
<td align="left">3.92 (4.89)</td>
</tr>
<tr>
<td align="left">Sense-making</td>
<td align="left">people, like, feel, seem, many, thing, even, normal, really, want, keep, life, right, pandemic, lot, maybe, less, happen, social, little</td>
<td align="left">Honest question—how do you deal with vocal vaccine deniers? I don’t want them polluting the vast majority of the population into not vaccinating, and it is honestly scary. Thanks from the UK.</td>
<td align="left">103 (2.68)</td>
<td align="left">4.37 (3.91)</td>
</tr>
<tr>
<td align="left">Risk assessment</td>
<td align="left">risk, high, low, seem, make, level, reduce, transmission, good, understand, factor, consider, tell, try, study, way, exposure, everyone, without, whether, follow, big, kind, show, regard</td>
<td align="left">I was wondering if you could create a chart where one side lists the risks associated with the vaccine and the other side lists the risks associated with the virus.</td>
<td align="left">53 (1.38)</td>
<td align="left">3.66 (3.66)</td>
</tr>
</tbody>
</table>
</alternatives>
<table-wrap-foot>
<fn id="t001fn001"><p><sup>a</sup> Determined based on the number of submissions for which each topic was the highest-probability topic in the distribution.</p></fn>
<fn id="t001fn002"><p><sup>b</sup> Calculated using the full topic probability distribution for each submission.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec012">
<title>Theme 1: Scientific and medical basis of COVID-19</title>
<p>This theme contained 5 topics related to scientific and medical aspects of the pandemic. ‘Immunology’ covered questions about the immune system, the SARS-CoV-2 virus, and vaccines. ‘Disease Process’ was related to the symptoms and sequelae of COVID-19, distinguishing between “mild,” “severe,” and “long” COVID. ‘Immunity and Transmission’ reflected uncertainties about viral transmission and infection- or vaccine-induced immunity. ‘Viral Variants’ mainly referenced the Delta variant, though the “UK” (i.e., Alpha) variant was also mentioned. Finally, submissions in ‘Other Medical Conditions’ raised concerns about comorbid conditions and infection or vaccination. “Flu” was the most frequently mentioned disease or condition, though others among the top 50 included “pneumonia,” “asthma,” “obese,” “arthritis,” and “diabetes.” Other frequent words reflected uncertainty about timing (e.g., “able,” “wait,” “eligible”).</p>
</sec>
<sec id="sec013">
<title>Theme 2: COVID-19 vaccine</title>
<p>This theme contained 4 topics specifically related to the COVID-19 vaccine. ‘Dosage and Timing’ captured general questions about the vaccination schedule, as well as words that provided context about readers’ vaccination status and time since vaccination (e.g., “first,” “Moderna,” “month”). Submissions in ‘Development and Rollout’ were related to the vaccine’s research and development process, its rollout, and regulatory agencies. Words indicating objective metrics (e.g., “datum [data],” “efficacy,” “effective”) were frequent. ‘Behavior Surrounding Vaccination’ captured considerations about socializing given vaccination status and official guidelines. Finally, ‘Safety and Side Effects’ reflected fears about adverse effects (e.g., “report,” “cause,” “concern”). Safety concerns were sometimes gender-specific; the word “woman” was frequently mentioned, and “fertility,” “menstrual,” and “cycle” were also among the top 50 words.</p>
</sec>
<sec id="sec014">
<title>Theme 3: COVID-19 mitigation strategies</title>
<p>This theme contained 5 topics related to COVID-19 mitigation strategies aside from vaccination. ‘Socializing Safely’ contained situation-specific questions about mitigating personal risk via non-pharmaceutical interventions (e.g., “mask,” “outdoor,” “distancing”). ‘Testing and Isolation’ was related to best practices following exposure or infection. Though “asymptomatic” was among the top 50 words associated with this topic, no words were explicitly related to vaccination. ‘Ventilation’ was specific to mitigating airborne transmission, including questions about airflow in homes, schools, and workplaces. ‘Sanitation and Hygiene’ was specific to mitigating surface transmission. Activities surrounding food (e.g., “grocery,” “food,” “restaurant”) were frequently referenced. Finally, ‘Masks’ reflected the desire to optimize face mask usage (e.g., “protect,” “double,” “effective”). References to mask “recommend[ations]” or “require[ments]” were also common.</p>
</sec>
<sec id="sec015">
<title>Theme 4: Society and institutions</title>
<p>This theme contained 3 topics related to social institutions. The most frequent words in ‘Education’ (e.g., “decision,” “reopen,” “guidance”) reflected uncertainty surrounding the return to in-person schooling, especially for “elementary” school-aged children. ‘Healthcare and Policy’ was a broad topic covering the healthcare system, public policy, official COVID-19 guidelines, and occupations. Many submissions referenced specific locations. Finally, ‘Social Participation’ captured participation in public life, most notably “travel.” Guidelines or requirements (e.g., “vaccination”), locations (e.g., “state,” “country”), and timing or anticipation (e.g., “month,” “summer,” “plan”) were frequently referenced.</p>
</sec>
<sec id="sec016">
<title>Theme 5: Family and personal relationships</title>
<p>This theme contained 3 topics that mostly captured submissions seeking situation-specific advice. ‘Interacting with Family’ covered interactions within and between households, such as living arrangements and travel. Safety concerns (e.g., “safe,” “quarantine,” “precaution”) and modes of travel were more common here than in the ‘Social Participation’ topic. ‘Children and Parenting’ mostly pertained to younger children. “Eligible” was among the most frequent words, reflecting anticipation of vaccines for this demographic. Finally, ‘Personal Relationships’ referenced specific relationships (e.g., “daughter,” “husband,” “friend”) and arrangements or plans (e.g., “stay,” “come,” “live”).</p>
</sec>
<sec id="sec017">
<title>Theme 6: Navigating the COVID-19 infodemic</title>
<p>This theme contained 5 topics related to interacting with COVID-19 information. Many submissions in ‘Verifying Information’ referenced academic or online sources (e.g., “article,” “post,” “study”) and asked the team to fact-check information. Among the top 50 words in this topic, only “ivermectin” referenced a specific news item, in contrast to general references such as “article” or “news.” ‘Feedback’ captured salutations, post requests, and expressions of gratitude. ‘Data and Statistics’ referenced metrics of COVID-19 burden (e.g., “case,” “death,” “infection”) and reflected confusion about interpreting them. ‘Sense-making’ captured uncertainty and reflected the desire to cope with the pandemic (e.g., “normal,” “want,” “life”). Of note, the word “pandemic” (but not “covid” or “coronavirus”) was among the top 50 words. Finally, ‘Risk Assessment’ captured complex risk decisions (e.g., “understand,” “factor,” “consider”) and reflected desires to simplify them (e.g., “high,” “low,” “reduce”).</p>
</sec>
<sec id="sec018">
<title>Visualization of question box submissions</title>
<p><xref ref-type="fig" rid="pone.0281773.g001">Fig 1</xref> displays the t-SNE visualization of submissions and topics. Data points (i.e., submissions) are differentiated by colors and symbols according to their theme and highest-probability topic. Points that are closer in proximity represent submissions with relatively similar topic probability distributions. Overall, the classification of submissions according to their highest-probability topic resulted in well-defined t-SNE clusters. The algorithmic clustering of topics was generally consistent with the themes that we generated using thematic analysis, with three notable exceptions: ‘Testing and Isolation,’ ‘Behavior Surrounding Vaccination,’ and ‘Viral Variants.’ For example, we placed the topic ‘Viral Variants’ under the ‘Scientific and Medical Basis of COVID-19’ theme, but the algorithm clustered submissions in this topic alongside topics in ‘COVID-19 Mitigation Strategies’ and ‘Navigating the COVID-19 Infodemic.’ This demonstrates the interconnectedness among topics and reflects that different thematic classifications may be possible.</p>
<fig id="pone.0281773.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0281773.g001</object-id>
<label>Fig 1</label>
<caption>
<title>T-SNE visualization of submissions to the Dear Pandemic question box.</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.g001" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec019">
<title>Trend analysis of topic prevalence</title>
<p><xref ref-type="fig" rid="pone.0281773.g002">Fig 2</xref> displays each topic’s prevalence over time, as determined by its mean daily probability. A number of topics showed distinct trends. ‘Viral Variants’ peaked around January and July 2021, corresponding to increased news attention on the Alpha and Delta variants. Topics in the ‘COVID-19 Vaccine’ theme showed a sequential increase in prevalence over time: first ‘Development and Rollout,’ then ‘Safety and Side Effects,’ then ‘Dosage and Timing,’ then finally ‘Behavior Surrounding Vaccination.’ Conversely, topics in the ‘COVID-19 Mitigation Strategies’ theme decreased in prevalence over time. Finally, topics related to family and children, such as ‘Education,’ ‘Interacting with Family,’ ‘Personal Relationships,’ and ‘Children and Parenting’ peaked around the winter holiday and summer seasons.</p>
<fig id="pone.0281773.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0281773.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Trends in topic prevalence from August 24, 2020 to August 24, 2021.</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.g002" xlink:type="simple"/>
</fig>
</sec>
</sec>
<sec id="sec020" sec-type="conclusions">
<title>Discussion</title>
<p>In this study, we analyzed questions that readers submitted to Dear Pandemic, a prominent science communication campaign, during the height of the COVID-19 pandemic. We used LDA topic modeling to discover latent topics among the submissions, and we used thematic analysis to interpret the results. We generated six themes that characterized readers’ information needs: ‘Scientific and Medical Basis of COVID-19,’ ‘COVID-19 Vaccine,’ ‘COVID-19 Mitigation Strategies,’ ‘Society and Institutions,’ ‘Family and Personal Relationships,’ and ‘Navigating the COVID-19 Infodemic.’</p>
<p>The topics and themes we identified highlight that Dear Pandemic’s readers sought information not only to explain novel scientific concepts, but also to understand how they applied to their personal lives. For example, the topic ‘Immunology’ included technical words like “antibody,” “autoimmune,” and “cell,” but the questions themselves revealed pragmatic concerns (e.g., “I have 2 autoimmune illnesses, celiac disease &amp; Hashimoto’s thyroiditis. Does that mean I am immunocompromised?”). In their analysis of COVID-19 information seeking on Naver Knowledge iN, Kim &amp; Oh similarly identified topics related to the “cognitive context” (e.g., symptoms and masks) as well as the “situational context” (e.g., financial support, study, and work) [<xref ref-type="bibr" rid="pone.0281773.ref013">13</xref>]. Compared to their study, as well as other studies of general social media discourse [<xref ref-type="bibr" rid="pone.0281773.ref008">8</xref>, <xref ref-type="bibr" rid="pone.0281773.ref009">9</xref>], we found a higher prevalence of technical terms among the top words (<xref ref-type="table" rid="pone.0281773.t001">Table 1</xref>). This could suggest that the scientific literacy of Dear Pandemic’s readership was higher, or that readers used more technical language to engage with expert science communicators than they would use to engage with the lay public.</p>
<p>Our t-SNE visualization (<xref ref-type="fig" rid="pone.0281773.g001">Fig 1</xref>) also suggests ways in which topics were interconnected. As illustrated by the difference between our thematic categorization of certain topics (‘Behavior Surrounding Vaccination,’ ‘Testing and Isolation,’ and ‘Viral Variants’) and the clustering provided by the t-SNE algorithm, quantitative methods can reveal novel insights. For example, submissions related to viral variants were more similar to those surrounding mitigation strategies and the infodemic, rather than the scientific basis of COVID-19. Readers may have been more immediately concerned with, for instance, assessing and mitigating risks due to the Delta variant than with understanding the science behind its emergence. This reinforces the need for science communicators to offer practical, timely guidance–not just science lessons. In their study of psychosocial stressors during the pandemic, Leung &amp; Khalvati similarly used t-SNE to visualize the association between latent topics in Reddit posts [<xref ref-type="bibr" rid="pone.0281773.ref023">23</xref>]. They found that family-related topics clustered with topics about the fear of COVID-19. In our analysis, family-related topics clustered with topics about personal behavior (e.g., ‘Behavior Surrounding Vaccination,’ ‘Socializing Safely,’ and ‘Testing and Isolation’). Based on these findings, science communicators may hypothesize that family-related worries about COVID-19 prompt online audiences to seek information about how their family members can modify their behavior, and tailor their content accordingly.</p>
<p>The prevalence of topics related to family and children in our results is unsurprising in light of Dear Pandemic’s modal readers: women aged 35–54 years old. Women generally seek online health information more often than men [<xref ref-type="bibr" rid="pone.0281773.ref024">24</xref>]. In an analysis of COVID-19 related worries, van der Vegt &amp; Kleinberg found that women disproportionately worried about family members and severe health consequences, whereas men disproportionately worried about the economy and society [<xref ref-type="bibr" rid="pone.0281773.ref025">25</xref>]. Interestingly, economic concerns did not emerge as a distinct topic in our analysis, which may reflect this demographic trend. In contrast to other analyses conducted using Facebook posts or Tweets [<xref ref-type="bibr" rid="pone.0281773.ref009">9</xref>, <xref ref-type="bibr" rid="pone.0281773.ref026">26</xref>], we also did not identify any topics related to politics, pseudoscience, or conspiracy theories. This could be due to differences in audience characteristics (e.g., scientific literacy, partisanship, or preferred information sources). It is important to counter misinformation through engagement with skeptics, but it is also critical to recognize that even those who are highly engaged with trusted science communicators and attuned to evidence-based guidance have unmet information needs.</p>
<p>As revealed in our trend analysis (<xref ref-type="fig" rid="pone.0281773.g002">Fig 2</xref>), Dear Pandemic readers sought information about certain issues, such as the emergence of viral variants, in line with news media attention. Readers also sought information in anticipation of upcoming events in their lives, as indicated by peaks in topics related to children or family prior to the start of the school year or the winter holidays. Finally, the sequential rise of topics within the ‘COVID-19 Vaccine’ theme–from ‘Development and Rollout,’ to ‘Safety and Side Effects,’ to ‘Dosage and Timing,’ to ‘Behavior Surrounding Vaccination’–mirrored the real-world progression of vaccine-related events. Kim &amp; Oh similarly found that COVID-19 information needs related to the “social context” rose in prevalence later than information needs related to the “cognitive” and “situational” contexts [<xref ref-type="bibr" rid="pone.0281773.ref013">13</xref>]. Science communicators should thus be mindful of these temporal factors when fulfilling information needs.</p>
<p>Our results offer significant contributions to the literature on online COVID-19 discourse. To the best of our knowledge, we are the first to analyze text written for the specific purpose of seeking information from trusted experts. Most other studies used pre-determined hashtags to extract COVID-19 discourse from platforms such as Twitter [<xref ref-type="bibr" rid="pone.0281773.ref007">7</xref>, <xref ref-type="bibr" rid="pone.0281773.ref009">9</xref>, <xref ref-type="bibr" rid="pone.0281773.ref010">10</xref>], or used search trends as a proxy for information needs [<xref ref-type="bibr" rid="pone.0281773.ref011">11</xref>]. These studies were able to capture a wide breadth of discourse, but they lacked the ability to discern the purpose for which it was written. For example, Abd-Alrazaq et al. identified topics such as deaths caused by COVID-19, economic losses, and mask usage in a seminal infodemic surveillance study of Twitter discourse [<xref ref-type="bibr" rid="pone.0281773.ref007">7</xref>]. While they were able to compute prevalence statistics, sentiment scores, and engagement metrics for each topic, we note that their use of a non-specific dataset meant that their results may not have necessarily reflected information needs. Science communicators may therefore have difficulty translating the results of such studies into practice. In contrast, Kim &amp; Oh were able to more plausibly conclude that the topics they identified among posts to the question-and-answer platform Naver Knowledge iN reflected users’ information needs [<xref ref-type="bibr" rid="pone.0281773.ref013">13</xref>]. Since our study explicitly focused on readers who contacted science communicators about their information needs, our insights may be even more readily translated into practice.</p>
<p>More broadly, we demonstrate that Dear Pandemic’s question box format and our analytical approach can be adopted by other science communicators and researchers. These tools align with the WHO’s infodemic management framework, which recommends mixed-methods protocols for analyzing information flows [<xref ref-type="bibr" rid="pone.0281773.ref005">5</xref>]. The question box format, which facilitates two-way engagement with readers, generates rich text data that can be used in LDA topic models. One significant advantage of using such quantitative methods to analyze the data is that they are more time- and resource-efficient. If researchers want to glean deeper insights, however, we show that they can certainly integrate topic modeling with qualitative methods such as thematic analysis. Our longitudinal analysis also shows that topic modeling can be useful in real time, as we found that trends in topic prevalence co-occurred with developments in the COVID-19 information landscape. We note that dynamic topic models may be better-suited to identify real-time information needs in practice, as they can quickly track the evolution of individual topics, but these methods require larger datasets than we used for our study [<xref ref-type="bibr" rid="pone.0281773.ref008">8</xref>].</p>
<p>We also acknowledge our study’s limitations. Regarding our data, limiting our corpus to English-language webform submissions excluded attempts that readers made to seek information in other languages and through other channels. In order to protect anonymity and promote trust from readers, the webform did not request any demographic information beyond location. Information-seeking habits on social media are associated with demographic and personality characteristics [<xref ref-type="bibr" rid="pone.0281773.ref027">27</xref>], so Dear Pandemic readers who submitted questions likely differed from those who did not. Regarding our methods, our decision to use a 25-topic model was partially based on our subjective evaluation of semantic validity and granularity, in addition to C<sub>v</sub> score. A 9-topic model would have yielded the highest C<sub>v</sub> score, but we judged the topics to be insufficiently granular; selecting a different number of topics may have yielded different conclusions. Though we subsequently followed an established process of thematic analysis to interpret the results of our LDA topic model, it has known limitations, including its inherently subjective nature and inability to allow for much interpretation beyond description [<xref ref-type="bibr" rid="pone.0281773.ref017">17</xref>, <xref ref-type="bibr" rid="pone.0281773.ref022">22</xref>]. Finally, we recognize that the social media landscape is fragmented and heterogeneous, so our findings may not be generalizable to other platforms.</p>
</sec>
<sec id="sec021" sec-type="conclusions">
<title>Conclusions</title>
<p>In our continued efforts to manage the COVID-19 pandemic, we should also strive to manage the parallel infodemic. Our analysis of submissions to the Dear Pandemic question box adds to the literature on online COVID-19 discourse, focusing on the information needs of online audiences who engage with trusted science communicators. Our topic model and thematic analysis demonstrates that such audiences seek information that is scientifically rigorous, timely, and practical to their daily decisions and personal circumstances. This mixed-methods approach offers a model for other science communicators who aim to establish a mechanism for tracking, understanding, and responding to their readers’ information needs. Future studies should continue to characterize the information needs of different online audiences during the COVID-19 pandemic, as well as the ways in which science communicators and other entities have attempted to fulfill them.</p>
</sec>
<sec id="sec022" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pone.0281773.s001" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s001" xlink:type="simple">
<label>S1 File</label>
<caption>
<title>R analysis code.</title>
<p>(R)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s002" mimetype="text/csv" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s002" xlink:type="simple">
<label>S1 Data</label>
<caption>
<title>Geographic location of Dear Pandemic question box submissions.</title>
<p>(CSV)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s003" mimetype="text/csv" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s003" xlink:type="simple">
<label>S2 Data</label>
<caption>
<title>Full topic probability distribution for Dear Pandemic question box submissions.</title>
<p>(CSV)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s004" mimetype="text/csv" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s004" xlink:type="simple">
<label>S3 Data</label>
<caption>
<title>Topics, themes, and top words.</title>
<p>(CSV)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s005" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s005" xlink:type="simple">
<label>S1 Fig</label>
<caption>
<title>Overview of research process.</title>
<p>(TIF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s006" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s006" xlink:type="simple">
<label>S2 Fig</label>
<caption>
<title>C<sub>v</sub> coherence scores for LDA models with 3 to 50 topics.</title>
<p>(TIFF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0281773.s007" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0281773.s007" xlink:type="simple">
<label>S3 Fig</label>
<caption>
<title>Daily submissions to the Dear Pandemic question box, August 24, 2020 to August 24, 2021.</title>
<p>(TIFF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ack>
<p>The authors thank Garrick Sherman at the World Well-Being Project, the Dear Pandemic team (including Alejandra Silva Hernández, Ashley Ritter, Chana Davis, Chloe Gibbs, Christine Whelan, Daisey Velazquez, Dena Jennings, Gretchen Peterson, Jenny Leininger, Jessica Williams-Nguyen, Joanna Dreifus, Lauren Hale, Mahima Bhattar, Mary-Jo Valentino, Maya Clark-Cutaia, Megan Osvath, Rebecca Doyle, Sarah Coles, Shoshana Aronowitz, Sol Vidal Almela. Vijaya Knight, and Tita Smyth Escobedo), and Dear Pandemic’s readers.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0281773.ref001"><label>1</label><mixed-citation publication-type="other" xlink:type="simple">Infodemic. World Health Organization; 2020 [cited 2021 Nov 18]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/health-topics/infodemic#tab=tab_1" xlink:type="simple">https://www.who.int/health-topics/infodemic#tab=tab_1</ext-link>.</mixed-citation></ref>
<ref id="pone.0281773.ref002"><label>2</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Liu</surname> <given-names>PL</given-names></name>. <article-title>COVID-19 Information Seeking on Digital Media and Preventive Behaviors: The Mediation Role of Worry</article-title>. <source>Cyberpsychol, Behav, Soc Netw</source>. <year>2020</year>;<volume>23</volume>(<issue>10</issue>):<fpage>677</fpage>–<lpage>82</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1089/cyber.2020.0250" xlink:type="simple">10.1089/cyber.2020.0250</ext-link></comment> <object-id pub-id-type="pmid">32498549</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref003"><label>3</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kim</surname> <given-names>HK</given-names></name>, <name name-style="western"><surname>Ahn</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Atkinson</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Kahlor</surname> <given-names>LA</given-names></name>. <article-title>Effects of COVID-19 Misinformation on Information Seeking, Avoidance, and Processing: A Multicountry Comparative Study</article-title>. <source>Sci Commun</source>. <year>2020</year>;<volume>42</volume>(<issue>5</issue>):<fpage>586</fpage>–<lpage>615</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Eysenbach</surname> <given-names>G.</given-names></name> <article-title>How to Fight an Infodemic: The Four Pillars of Infodemic Management.</article-title> <source>J Med Internet Res</source>. <year>2020</year>;<volume>22</volume>(<issue>6</issue>):<fpage>e21820</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/21820" xlink:type="simple">10.2196/21820</ext-link></comment> <object-id pub-id-type="pmid">32589589</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Tangcharoensathien</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Calleja</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Nguyen</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Purnat</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>D’Agostino</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Garcia-Saiso</surname> <given-names>S</given-names></name>, <etal>et al</etal>. <article-title>Framework for Managing the COVID-19 Infodemic: Methods and Results of an Online, Crowdsourced WHO Technical Consultation.</article-title> <source>J Med Internet Res</source>. <year>2020</year>;<volume>22</volume>(<issue>6</issue>):<fpage>e19659</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/19659" xlink:type="simple">10.2196/19659</ext-link></comment> <object-id pub-id-type="pmid">32558655</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Fontaine</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Maheu-Cadotte</surname> <given-names>M-A</given-names></name>, <name name-style="western"><surname>Lavallée</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Mailhot</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Rouleau</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Bouix-Picasso</surname> <given-names>J</given-names></name>, <etal>et al</etal>. <article-title>Communicating Science in the Digital and Social Media Ecosystem: Scoping Review and Typology of Strategies Used by Health Scientists.</article-title> <source>JMIR Public Health Surveill</source>. <year>2019</year>;<volume>5</volume>(<issue>3</issue>):<fpage>e14447</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/14447" xlink:type="simple">10.2196/14447</ext-link></comment> <object-id pub-id-type="pmid">31482854</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref007"><label>7</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Abd-Alrazaq</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Alhuwail</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Househ</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Hamdi</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Shah</surname> <given-names>Z</given-names></name>. <article-title>Top Concerns of Tweeters During the COVID-19 Pandemic: Infoveillance Study.</article-title> <source>J Med Internet Res</source>. <year>2020</year>;<volume>22</volume>(<issue>4</issue>):<fpage>e19016</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/19016" xlink:type="simple">10.2196/19016</ext-link></comment> <object-id pub-id-type="pmid">32287039</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref008"><label>8</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zamani</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Schwartz</surname> <given-names>HA</given-names></name>, <name name-style="western"><surname>Eichstaedt</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Guntuku</surname> <given-names>SC</given-names></name>, <name name-style="western"><surname>Ganesan</surname> <given-names>AV</given-names></name>, <name name-style="western"><surname>Clouston</surname> <given-names>S</given-names></name>, <etal>et al</etal>. <article-title>Understanding Weekly COVID-19 Concerns through Dynamic Content-Specific LDA Topic Modeling</article-title>. <source>Proc Conf Empir Methods Nat Lang Process</source>. <year>2020</year>;<volume>2020</volume>:<fpage>193</fpage>–<lpage>8</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.18653/v1/2020.nlpcss-1.21" xlink:type="simple">10.18653/v1/2020.nlpcss-1.21</ext-link></comment> <object-id pub-id-type="pmid">34095902</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref009"><label>9</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Guntuku</surname> <given-names>SC</given-names></name>, <name name-style="western"><surname>Buttenheim</surname> <given-names>AM</given-names></name>, <name name-style="western"><surname>Sherman</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Merchant</surname> <given-names>RM</given-names></name>. <article-title>Twitter discourse reveals geographical and temporal variation in concerns about COVID-19 vaccines in the United States</article-title>. <source>Vaccine</source>. <year>2021</year>;<volume>39</volume>(<issue>30</issue>):<fpage>4034</fpage>–<lpage>8</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.vaccine.2021.06.014" xlink:type="simple">10.1016/j.vaccine.2021.06.014</ext-link></comment> <object-id pub-id-type="pmid">34140171</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref010"><label>10</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Valdez</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>ten Thij</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Bathina</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Rutter</surname> <given-names>LA</given-names></name>, <name name-style="western"><surname>Bollen</surname> <given-names>J</given-names></name>. <article-title>Social Media Insights Into US Mental Health During the COVID-19 Pandemic: Longitudinal Analysis of Twitter Data.</article-title> <source>J Med Internet Res</source>. <year>2020</year>;<volume>22</volume>(<issue>12</issue>):<fpage>e21418</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/21418" xlink:type="simple">10.2196/21418</ext-link></comment> <object-id pub-id-type="pmid">33284783</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mangono</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Smittenaar</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Caplan</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Huang</surname> <given-names>VS</given-names></name>, <name name-style="western"><surname>Sutermaster</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Kemp</surname> <given-names>H</given-names></name>, <etal>et al</etal>. <article-title>Information-Seeking Patterns During the COVID-19 Pandemic Across the United States: Longitudinal Analysis of Google Trends Data.</article-title> <source>J Med Internet Res</source>. <year>2021</year>;<volume>23</volume>(<issue>5</issue>):<fpage>e22933</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/22933" xlink:type="simple">10.2196/22933</ext-link></comment> <object-id pub-id-type="pmid">33878015</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Chan</surname> <given-names>WW</given-names></name>, <name name-style="western"><surname>Chua</surname> <given-names>HN</given-names></name>. <article-title>Using Word2Vec-LDA-Word Mover Distance for Comparing the Patterns of Information Seeking and Sharing during the COVID-19 Pandemic</article-title>. <year>2022</year> <source>IEEE 7th International conference for Convergence in Technology (I2CT)</source>; <volume>2022</volume>: <fpage>1</fpage>–<lpage>8</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref013"><label>13</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kim</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Oh</surname> <given-names>S</given-names></name>. <article-title>Everyday life information seeking in South Korea during the COVID-19 pandemic: daily topics of information needs in social Q&amp;A</article-title>. <source>Online Inf Rev</source>. <year>2022</year>.</mixed-citation></ref>
<ref id="pone.0281773.ref014"><label>14</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Blei</surname> <given-names>DM</given-names></name>, <name name-style="western"><surname>Ng</surname> <given-names>AY</given-names></name>, <name name-style="western"><surname>Jordan</surname> <given-names>MI</given-names></name>. <article-title>Latent dirichlet allocation</article-title>. <source>J Mach Learn Res</source>. <year>2003</year>;<volume>3</volume>:<fpage>993</fpage>–<lpage>1022</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref015"><label>15</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Albalawi</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Yeap</surname> <given-names>TH</given-names></name>, <name name-style="western"><surname>Benyoucef</surname> <given-names>M</given-names></name>. <article-title>Using Topic Modeling Methods for Short-Text Data: A Comparative Analysis.</article-title> <source>Front Artif Intell</source>. <year>2020</year>;<volume>3</volume>:<fpage>42</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frai.2020.00042" xlink:type="simple">10.3389/frai.2020.00042</ext-link></comment> <object-id pub-id-type="pmid">33733159</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref016"><label>16</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Murakami</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Thompson</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Hunston</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Vajn</surname> <given-names>D</given-names></name>. ‘<article-title>What is this corpus about?’: using topic modelling to explore a specialised corpus</article-title>. <source>Corpora</source>. <year>2017</year>;<volume>12</volume>(<issue>2</issue>):<fpage>243</fpage>–<lpage>77</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref017"><label>17</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Braun</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Clarke</surname> <given-names>V</given-names></name>. <article-title>Using thematic analysis in psychology</article-title>. <source>Qual Res Psychol</source>. <year>2006</year>;<volume>3</volume>:<fpage>77</fpage>–<lpage>101</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref018"><label>18</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Honnibal</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Montani</surname> <given-names>I</given-names></name>. <article-title>spaCy Usage Documentation [Internet].</article-title> <year>2021</year> [cited 2021 Nov 16]. Available: <ext-link ext-link-type="uri" xlink:href="https://spacy.io/usage" xlink:type="simple">https://spacy.io/usage</ext-link>.</mixed-citation></ref>
<ref id="pone.0281773.ref019"><label>19</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schwartz</surname> <given-names>HA</given-names></name>, <name name-style="western"><surname>Giorgi</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Sap</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Crutchley</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Ungar</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Eichstaedt</surname> <given-names>J</given-names></name>, editors. <article-title>Dlatk: Differential language analysis toolkit.</article-title> <source>Proceedings of the 2017 conference on empirical methods in natural language processing: System demonstrations</source>; <year>2017</year>.</mixed-citation></ref>
<ref id="pone.0281773.ref020"><label>20</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>McCallum</surname> <given-names>AK</given-names></name>. <article-title>MALLET:A Machine Learning for Language Toolkit [Internet].</article-title> <year>2002</year> [cited 2021 Nov 16]. Available: <ext-link ext-link-type="uri" xlink:href="http://mallet.cs.umass.edu" xlink:type="simple">http://mallet.cs.umass.edu</ext-link>.</mixed-citation></ref>
<ref id="pone.0281773.ref021"><label>21</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Röder</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Both</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Hinneburg</surname> <given-names>A</given-names></name>. <article-title>Exploring the Space of Topic Coherence Measures.</article-title> <source>Proceedings of the Eighth ACM International Conference on Web Search and Data Mining; Shanghai, China: Association for Computing Machinery</source>; <year>2015</year>:<fpage>399</fpage>–<lpage>408</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref022"><label>22</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Isoaho</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Gritsenko</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Mäkelä</surname> <given-names>E</given-names></name>. <article-title>Topic Modeling and Text Analysis for Qualitative Policy Research</article-title>. <source>Policy Stud J</source>. <year>2021</year>;<volume>49</volume>(<issue>1</issue>):<fpage>300</fpage>–<lpage>24</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref023"><label>23</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Leung</surname> <given-names>YT</given-names></name>, <name name-style="western"><surname>Khalvati</surname> <given-names>F</given-names></name>. <article-title>Exploring COVID-19–Related Stressors: Topic Modeling Study.</article-title> <source>J Med Internet Res</source>. <year>2022</year>;<volume>24</volume>(<issue>7</issue>):<fpage>e37142</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2196/37142" xlink:type="simple">10.2196/37142</ext-link></comment> <object-id pub-id-type="pmid">35731966</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref024"><label>24</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Manierre</surname> <given-names>MJ</given-names></name>. <article-title>Gaps in knowledge: Tracking and explaining gender differences in health information seeking</article-title>. <source>Soc Sci Med</source>. <year>2015</year>;<volume>128</volume>:<fpage>151</fpage>–<lpage>8</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.socscimed.2015.01.028" xlink:type="simple">10.1016/j.socscimed.2015.01.028</ext-link></comment> <object-id pub-id-type="pmid">25618604</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref025"><label>25</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>van der Vegt</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Kleinberg</surname> <given-names>B</given-names></name>, editors. <article-title>Women Worry About Family, Men About the Economy: Gender Differences in Emotional Responses to COVID-19</article-title>. <source>Social Informatics: 12th International Conference, SocInfo</source> <year>2020</year>; Pisa, Italy; <volume>2020</volume>: <fpage>397</fpage>–<lpage>409</lpage>.</mixed-citation></ref>
<ref id="pone.0281773.ref026"><label>26</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Sear</surname> <given-names>RF</given-names></name>, <name name-style="western"><surname>Velásquez</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Leahy</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Restrepo</surname> <given-names>NJ</given-names></name>, <name name-style="western"><surname>Oud</surname> <given-names>SE</given-names></name>, <name name-style="western"><surname>Gabriel</surname> <given-names>N</given-names></name>, <etal>et al</etal>. <article-title>Quantifying COVID-19 Content in the Online Health Opinion War Using Machine Learning</article-title>. <source>IEEE Access</source>. <year>2020</year>;<volume>8</volume>:<fpage>91886</fpage>–<lpage>93</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/ACCESS.2020.2993967" xlink:type="simple">10.1109/ACCESS.2020.2993967</ext-link></comment> <object-id pub-id-type="pmid">34192099</object-id></mixed-citation></ref>
<ref id="pone.0281773.ref027"><label>27</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kim</surname> <given-names>K-S</given-names></name>, <name name-style="western"><surname>Sin</surname> <given-names>S-CJ</given-names></name>, <name name-style="western"><surname>Tsai</surname> <given-names>T-I</given-names></name>. <article-title>Individual Differences in Social Media Use for Information Seeking</article-title>. <source>J Acad Libr</source>. <year>2014</year>;<volume>40</volume>(<issue>2</issue>):<fpage>171</fpage>–<lpage>8</lpage>.</mixed-citation></ref>
</ref-list>
</back>
</article>