<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id>
<journal-title-group>
<journal-title>PLOS Computational Biology</journal-title>
</journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1011676</article-id>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-23-00967</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Software</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Research assessment</subject><subj-group><subject>Reproducibility</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Database and informatics methods</subject><subj-group><subject>Bioinformatics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Computer software</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Engineering and technology</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Computer software</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Information theory</subject><subj-group><subject>Graph theory</subject><subj-group><subject>Directed graphs</subject><subj-group><subject>Directed acyclic graphs</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Graph theory</subject><subj-group><subject>Directed graphs</subject><subj-group><subject>Directed acyclic graphs</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Computational biology</subject><subj-group><subject>Genome analysis</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject><subj-group><subject>Genome analysis</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Software tools</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Engineering and technology</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Software tools</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Research design</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Computer software</subject><subj-group><subject>Open source software</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Engineering and technology</subject><subj-group><subject>Software engineering</subject><subj-group><subject>Computer software</subject><subj-group><subject>Open source software</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Science policy</subject><subj-group><subject>Open science</subject><subj-group><subject>Open source software</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Facilitating bioinformatics reproducibility with QIIME 2 Provenance Replay</article-title>
<alt-title alt-title-type="running-head">Facilitating bioinformatics reproducibility</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-4289-0991</contrib-id>
<name name-style="western">
<surname>Keefe</surname>
<given-names>Christopher R.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Dillon</surname>
<given-names>Matthew R.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Gehret</surname>
<given-names>Elizabeth</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Herman</surname>
<given-names>Chloe</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/validation/">Validation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Jewell</surname>
<given-names>Mary</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Wood</surname>
<given-names>Colin V.</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Bolyen</surname>
<given-names>Evan</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8865-1670</contrib-id>
<name name-style="western">
<surname>Caporaso</surname>
<given-names>J. Gregory</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
</contrib-group>
<aff id="aff001"><label>1</label> <addr-line>Center for Applied Microbiome Science, Pathogen and Microbiome Institute, Northern Arizona University, Flagstaff, Arizona, United States of America</addr-line></aff>
<aff id="aff002"><label>2</label> <addr-line>School of Informatics, Computing and Cyber Systems, Northern Arizona University, Flagstaff, Arizona, United States of America</addr-line></aff>
<aff id="aff003"><label>3</label> <addr-line>Department of Epidemiology, University of Washington, Seattle, Washington, United States of America</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Kreft</surname>
<given-names>Jan-Ulrich</given-names>
</name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1"><addr-line>University of Birmingham, UNITED KINGDOM</addr-line></aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">greg.caporaso@nau.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>11</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<month>11</month>
<year>2023</year>
</pub-date>
<volume>19</volume>
<issue>11</issue>
<elocation-id>e1011676</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>6</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>11</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-year>2023</copyright-year>
<copyright-holder>Keefe et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pcbi.1011676"/>
<abstract>
<p>Study reproducibility is essential to corroborate, build on, and learn from the results of scientific research but is notoriously challenging in bioinformatics, which often involves large data sets and complex analytic workflows involving many different tools. Additionally, many biologists are not trained in how to effectively record their bioinformatics analysis steps to ensure reproducibility, so critical information is often missing. Software tools used in bioinformatics can automate provenance tracking of the results they generate, removing most barriers to bioinformatics reproducibility. Here we present an implementation of that idea, Provenance Replay, a tool for generating new executable code from results generated with the QIIME 2 bioinformatics platform, and discuss considerations for bioinformatics developers who wish to implement similar functionality in their software.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100000054</institution-id>
<institution>National Cancer Institute</institution>
</institution-wrap>
</funding-source>
<award-id>1U24CA248454-01</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8865-1670</contrib-id>
<name name-style="western">
<surname>Caporaso</surname>
<given-names>J. Gregory</given-names>
</name>
</principal-award-recipient>
</award-group>
<funding-statement>This work was funded by NCI ITCR award 1U24CA248454-01 to JGC. The funders had no role in the design, implementation or presentation of the work shared here.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<page-count count="9"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>PLOS Publication Stage</meta-name>
<meta-value>vor-update-to-uncorrected-proof</meta-value>
</custom-meta>
<custom-meta>
<meta-name>Publication Update</meta-name>
<meta-value>2023-12-07</meta-value>
</custom-meta>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>Provenance Replay is open source and free for all use (BSD 3-clause license). The original stand-alone version is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/qiime2/provenance-lib" xlink:type="simple">https://github.com/qiime2/provenance-lib</ext-link>. As of QIIME 2 2023.9 (released 11 October 2023) Provenance Replay is included in the QIIME 2 framework, available at <ext-link ext-link-type="uri" xlink:href="https://github.com/qiime2/qiime2" xlink:type="simple">https://github.com/qiime2/qiime2</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Reproducibility, the ability of a researcher to duplicate the results of a study, is a necessary condition for scientific research to be considered informative and credible [<xref ref-type="bibr" rid="pcbi.1011676.ref001">1</xref>]. Peer review relies on study documentation to maintain the trustworthiness of scientific research [<xref ref-type="bibr" rid="pcbi.1011676.ref002">2</xref>–<xref ref-type="bibr" rid="pcbi.1011676.ref004">4</xref>]. Without comprehensive documentation, reviewers may be unable to verify a study’s validity and merit, and other researchers will be unable to interrogate the results or learn from the researchers’ approach, limiting the study’s value.</p>
<p>The biomedical research community has recently been concerned with a “reproducibility crisis,” and several high-profile publications have shown researchers unable to confirm findings of original studies [<xref ref-type="bibr" rid="pcbi.1011676.ref005">5</xref>,<xref ref-type="bibr" rid="pcbi.1011676.ref006">6</xref>]. This discussion generally focuses on one type of reproducibility failure: an inability to corroborate a study’s results. However, this literature neglects a deeper issue: many studies fail to provide even the minimum necessary documentation to reproduce a study’s methodology.</p>
<p>Although there is no standard nomenclature for reproducibility in the literature, existing definitions illustrate the goals of different types of reproducibility. For example, the Turing Way defines research as “Reproducible,” “Replicable,” “Robust,” or “Generalizable” based on whether a study’s results can be repeated using methods and data that are the same as, or different from, the original study [<xref ref-type="bibr" rid="pcbi.1011676.ref007">7</xref>]. Gundersen and Kjensmo create similar categories in their work on reproducibility, but they define a hierarchy based on the degree of generality [<xref ref-type="bibr" rid="pcbi.1011676.ref008">8</xref>].</p>
<p>We incorporated these ideas into a hierarchy of reproducible research, where different levels of reproducibility are represented by the classes “Reproducible,” “Replicable,” “Robust,” or “Generalizable” (<xref ref-type="fig" rid="pcbi.1011676.g001">Fig 1</xref>). Under this hierarchy, <italic>generalizable</italic> studies produce findings that are corroborated in other contexts, providing the building blocks for advancing scientific knowledge. Lower degrees of generality allow researchers to validate studies and expand focused work toward generalized conclusions [<xref ref-type="bibr" rid="pcbi.1011676.ref009">9</xref>].</p>
<fig id="pcbi.1011676.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1011676.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Turing Way (TW) reproducibility classes [<xref ref-type="bibr" rid="pcbi.1011676.ref007">7</xref>] ordered in a hierarchy based on generality, similar to Gundersen and Kjensmo [<xref ref-type="bibr" rid="pcbi.1011676.ref008">8</xref>].</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1011676.g001" xlink:type="simple"/>
</fig>
<p>High quality research documentation is essential to all levels of reproducibility. In bioinformatics, research typically involves large datasets, complex computer software, and analytical procedures with many distinct steps. In order to reproduce such a study, one needs both prospective provenance, the analytic workflow specified as a recipe for data creation, and retrospective provenance, the details of the runtime environment and the resources used for analysis [<xref ref-type="bibr" rid="pcbi.1011676.ref010">10</xref>].</p>
<p>Prospective provenance is most frequently realized using a document written by the data analyst, for example using Jupyter Notebooks, Snakemake, or RMarkdown documents that include executable code and in some cases analysis notes. Great care must be taken to link revisions of these types of research documentation with the results that they generated, as these documents tend to evolve over time. And even if specific revisions are conclusively linked to specific research results, problems can arise if it’s not clear how code was executed in an interactive environment to generate a specific research result. For example, it is not uncommon for novice data scientists to execute cells out of order in a large Jupyter Notebook or RMarkdown file, in which case a linear interpretation of the document would not accurately describe how it was used to generate a result. Retrospective provenance can be realized by capturing information about the analysis as it is run, including hardware and software environments, resource use, and the data and metadata involved. Similar to prospective provenance, this is most frequently captured by a data analyst in the form of written notes. Most published research does not meet these reproducibility needs, as researchers must balance competing demands on their time and grapple with publication structures that incentivize producing new work over documenting for reproducibility [<xref ref-type="bibr" rid="pcbi.1011676.ref009">9</xref>,<xref ref-type="bibr" rid="pcbi.1011676.ref011">11</xref>].</p>
<p>Even if researchers know what needs to be tracked and are diligent about tracking that information, recording all of the prospective and retrospective provenance required to reproduce a computational analysis is tedious and error-prone for humans. In our opinion, provenance tracking is a task better left to computer software. Analytic software tools that automatically produce research documentation have the potential to reduce the risk of paper retraction; facilitate collaboration, review, and debugging; and improve the continuity and impact of scientific research [<xref ref-type="bibr" rid="pcbi.1011676.ref007">7</xref>].</p>
<p>Engineering bioinformatics software to facilitate aspects of analysis reproducibility is a topic of contemporary interest in bioinformatics software literature [<xref ref-type="bibr" rid="pcbi.1011676.ref012">12</xref>]. For example, Snakemake [<xref ref-type="bibr" rid="pcbi.1011676.ref013">13</xref>] can automatically generate workflow graphs, using its <monospace specific-use="no-wrap">–-dag</monospace> option, providing an automated workflow visualization tool. Love et al [<xref ref-type="bibr" rid="pcbi.1011676.ref014">14</xref>] present tximeta, an approach for linking reference data checksums to RNA-seq analysis results that rely on that reference data, ensuring that relevant references can be uniquely identified. They also briefly review work highlighting the need for improved provenance tracking in bioinformatics, as well as scientific computing tools that aim to facilitate provenance tracking. Of the existing tools that we are aware of, CWLProv [<xref ref-type="bibr" rid="pcbi.1011676.ref015">15</xref>] and Research Objects [<xref ref-type="bibr" rid="pcbi.1011676.ref016">16</xref>] serve the broadest purpose of facilitating reproducibility of entire computational workflows, and are most similar to the work presented here. CWLProv provides a layer between the Common Workflow Language (CWL) and the W3C PROV model, to document retrospective provenance of arbitrary CWL workflows. Research Objects are a concept designed for value-added publication of research products, including research data that is discoverable for other work, and which includes data provenance to enable users to understand how the data was generated.</p>
<p>QIIME 2 is a biological data science platform that was initially built to facilitate microbiome amplicon analysis [<xref ref-type="bibr" rid="pcbi.1011676.ref017">17</xref>], but has been expanding into new domains, including analysis of highly-multiplexed serology assays [<xref ref-type="bibr" rid="pcbi.1011676.ref018">18</xref>], pathogen genomics [<xref ref-type="bibr" rid="pcbi.1011676.ref019">19</xref>], and microbiome shotgun metagenomics, through alternative distributions (i.e., bundles of QIIME 2 plugins, where plugins serve as Python 3 wrappers for arbitrary analytic software, including software written in languages other than Python). QIIME 2 has a built-in system that automatically tracks prospective and retrospective data provenance for users as they run their analyses, and the popularity of this feature is in part responsible for its adoption in other domains. In QIIME 2, users conduct analysis using Actions that each produce one or more Results. The prospective and retrospective provenance of all preceding analysis steps are automatically stored in each Result, allowing users to determine how a Result was generated and an analysis was conducted (<xref ref-type="fig" rid="pcbi.1011676.g002">Fig 2</xref>), even if scripts or notes were not recorded (or were misplaced) by the user, or if revision identifiers of those documents were not linked to research results. Additionally, QIIME 2 assigns universally unique identifiers (UUIDs) to every Result it creates, enabling data to be conclusively identified. Taken together, QIIME 2 therefore automatically supports <italic>reproducible</italic>, <italic>replicable</italic>, and <italic>robust</italic> bioinformatics analyses, without any effort on the part of its users.</p>
<fig id="pcbi.1011676.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1011676.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Schematic diagram of the provenance of a QIIME 2 Visualization.</title>
<p>A: An example of a QIIME 2 visualization illustrating the taxonomic composition of several samples. B: The directed, acyclic graph (DAG) tracing the history of the panel A visualization from initial data import into QIIME 2 through the creation of the visualization (denoted with an asterisk). This DAG can be used for analytical interpretation or publication, and serves as the input to Provenance Replay. Additional detail is provided in panel C on the nodes included in the dashed box. C: A DAG describing the inputs and outputs of the ‘action’ node highlighted in gray. D: “Action details,” captured during the execution of the node highlighted in gray. Some data collected as action details, such as information about the computational environment where the action was run, is not presented in this schematic for readability, but can be observed in the provenance of either of the two QIIME 2 results included in Supporting Information S1. The node selected for panel D was arbitrarily chosen.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1011676.g002" xlink:type="simple"/>
</fig>
<p>The work presented here attempts to improve computational methods reproducibility in bioinformatics by reducing the practical overhead of creating reproducibility documentation. Built around the automated, decentralized provenance capture implemented in QIIME 2, we present Provenance Replay, a tool that validates the integrity of QIIME 2 Results, parses the provenance data they contain, and programmatically generates executable scripts that allow for the reproduction, study, and extension of the source analysis.</p>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Design and implementation</title>
<p>Provenance Replay is written in Python 3 [<xref ref-type="bibr" rid="pcbi.1011676.ref020">20</xref>], and depends heavily on the Python standard library, NetworkX [<xref ref-type="bibr" rid="pcbi.1011676.ref021">21</xref>], pyyaml [<xref ref-type="bibr" rid="pcbi.1011676.ref022">22</xref>], and QIIME 2 itself, from which it takes advantage especially of the PluginManager and the Usage API. It ingests one or more QIIME 2 results and parses their provenance data into a Directed Acyclic Graph (DAG), implemented as a NetworkX DiGraph. It produces outputs by subsetting and manipulating this DiGraph and its contents. Outputs include BibTeX-formatted [<xref ref-type="bibr" rid="pcbi.1011676.ref023">23</xref>] citations for all Actions and Plugins used in a computational analysis, and executable scripts targeting the user’s preferred QIIME 2 interface. Users interact with the software through a command-line interface implemented with Click [<xref ref-type="bibr" rid="pcbi.1011676.ref024">24</xref>], or using its Python 3 API.</p>
<p>The initial software design was based on literature review, existing API targets, and discussion with QIIME 2 developers, as well as an initial requirements engineering process. The requirements engineering process consisted of requirements elicitation by focus group and requirements validation using the Technology Acceptance Model (TAM) [<xref ref-type="bibr" rid="pcbi.1011676.ref025">25</xref>]. Focus group participants were recruited through posts on the QIIME 2 community forum and participated in one-hour focus group sessions, which included a software demonstration and discussion. Discussion questions were intended to elicit open-ended feedback and exploration of possible features of value. When asked how likely they were to recommend Provenance Replay to a colleague who uses QIIME 2, 79% of focus group participants (15/19) were classified as promoters (scores of 9–10 out of 10), 21% (4/19) as passive (scores 7–8 out of 10) and none as detractors, resulting in a net promoter score of +79%. Provenance Replay also scored well on the TAM instruments, with respondents rating both its Perceived Ease of Use as “high,” (overall mean 5.8, on a scale of 1–7) and Perceived Usefulness as “high” (overall mean 6.0, on a scale of 1–7). Additional detail on this process is provided in [<xref ref-type="bibr" rid="pcbi.1011676.ref026">26</xref>].</p>
<p>Provenance Replay is supported by QIIME 2 versions 2021.11 and newer, and can parse data provenance from Results generated with any version of QIIME 2. Provenance Replay is capable of replaying a single QIIME 2 Result in a few seconds, and a very large analysis (450 results) in 8–10 minutes on a contemporary small-business laptop (Intel Core i7-8565U CPU @ 1.8GHz, 16 GB RAM, OpenSUSE Tumbleweed running on an M.2 SSD). As such, most users will not need to work in a cluster environment, and native installation is recommended (and supported on Linux, macOS, and Windows via Windows Subsystem for Linux 2 (WSL2)).</p>
</sec>
<sec id="sec003" sec-type="results">
<title>Results</title>
<p>Provenance Replay is software for the documentation and enactment of <italic>in silico</italic> reproducibility in QIIME 2, which can produce command-line (bash) and Python 3 scripts directly from a QIIME 2 Result. Provenance Replay outputs are self-documenting, using UUIDs to identify them as products of specific QIIME 2 Results, and they include step-by-step instructions for users to execute the scripts produced. Provenance Replay also implements MD5 checksum-based validation of Result provenance, which can alert if the Results were altered since they were generated, in which case the data provenance would no longer be reliable.</p>
<p>Provenance Replay has many features that we consider good general targets for tools that automate reproducibility documentation:</p>
<list list-type="bullet">
<list-item><p>Completeness: Provenance Replay provides comprehensive access to captured provenance data.</p></list-item>
<list-item><p>Ease of Documentation: Users can generate a complete “reproducibility supplement,” including replay scripts and citation information, with a single command through different user interface types.</p></list-item>
<list-item><p>Ease of Application: Replay scripts are executable with minimal modification, target a variety of interfaces, and are self-documenting.</p></list-item>
<list-item><p>Accessibility: Replay documents are designed for human readability and include their own usage instructions. Additionally, by providing multiple user interfaces for running Provenance Replay, as well as multiple target interfaces for its outputs, users with varying degrees of computational experience can interpret its results.</p></list-item>
</list>
<p>Provenance Replay automatically removes most barriers to <italic>in silico</italic> methods reproducibility in QIIME 2 (with some exceptions discussed below). This simplifies the process of documenting research, and it has already been used to generate <italic>reproducibility supplements</italic> for scientific publications [<xref ref-type="bibr" rid="pcbi.1011676.ref027">27</xref>,<xref ref-type="bibr" rid="pcbi.1011676.ref028">28</xref>].</p>
</sec>
<sec id="sec004">
<title>Availability and future directions</title>
<p>The Provenance Replay software is open source and free for all use (BSD 3-clause license). As of QIIME 2 2023.5 (released 24 May 2023), the software is included in the QIIME 2 “core distribution” (such that it is installed with QIIME 2), and as of QIIME 2 2023.9 (released 11 October 2023) it is included in the QIIME 2 framework itself, ensuring that it will stay current as QIIME 2 continues to evolve and will be available in all QIIME 2 distributions, including any developed by third-parties.</p>
<p><italic>Reproducible</italic> and <italic>robust</italic> bioinformatics (<xref ref-type="fig" rid="pcbi.1011676.g001">Fig 1</xref>) involves unambiguous identification of the data used in an analysis, and enabling this is an important target for tools aiming to facilitate reproducible bioinformatics. This can be challenging to achieve, as it requires stable, unique identifiers, and if data can be mutated, the identifiers should be versioned. QIIME 2 uniquely identifies its data artifacts with UUIDs, and those artifacts are immutable (once created, they cannot be changed without the creation of a new data artifact with a different UUID). This ensures that analyses are <italic>reproducible</italic> and <italic>robust</italic> if a researcher has access to the data. Providing general purpose access to QIIME 2 Results is outside the scope of the system, so it remains the responsibility of the user to ensure their data are available to others. Sharing QIIME 2 Results and Provenance Replay reproducibility supplements in a single archive through a stable service such as FigShare is an excellent way for users to ensure that their analysis will be <italic>reproducible</italic>, <italic>replicable</italic>, and <italic>robust</italic>. QIIME 2 may facilitate data access in the future by enabling programmatic retrieval of data artifacts from Qiita [<xref ref-type="bibr" rid="pcbi.1011676.ref029">29</xref>], or integrating q2-fondue [<xref ref-type="bibr" rid="pcbi.1011676.ref030">30</xref>] commands with Provenance Replay results, to load data from the NCBI Sequence Read Archive into QIIME 2 as a step in replaying an analysis.</p>
<p>In addition to enabling unambiguous identification of data (including any reference data) that is used, there are several other important targets for tools aiming to facilitate reproducible bioinformatics. First, commands used to generate data must be recorded with all parameter settings, including default values, and unambiguously linked to their input and output data. While checksums of entire files are tempting unique identifiers, they take too long to compute on large files to be practical as identifiers. QIIME 2 uses version 4 UUIDs. Next, all relevant software versions, including versions of underlying dependencies, must be recorded. Details about the environment where command execution occurred are important to record, including the Operating System and its version and the version of the programming languages used. Differences in any of these variables could cause a failure to reproduce results. Minting and recording an execution identifier (e.g., as a UUID) can help to remove ambiguity regarding when or how a command or workflow was applied. And, while not essential for reproducibility, recording a timestamp of execution can be helpful when trying to make sense of collections of results. Finally, it is generally a good idea to optionally provide software tools in a containerized environment, to ensure that a working environment will be accessible in the future (e.g., if unavailability of compatible binaries prevents environment recreation). Khan et al. (2019) [<xref ref-type="bibr" rid="pcbi.1011676.ref015">15</xref>] provide a more detailed list of specific recommendations and corresponding justifications on best practices for ensuring reproducible computational workflows.</p>
<p>A future goal is to enable Provenance Replay to output replay scripts for users of QIIME 2 through graphical interfaces. QIIME 2 can be accessed through a Python 3 API, a command line interface (CLI), and through various workflow systems, including Galaxy and CWL. At present, Provenance Replay can output Python 3 scripts and bash scripts, providing documentation options for users of the API and CLI. A future development target is to provide reproducibility instructions for Galaxy [<xref ref-type="bibr" rid="pcbi.1011676.ref031">31</xref>] users as well. Providing complete documentation through higher-level (e.g., graphical) interfaces is more verbose, but ultimately expands the audience who can learn from that documentation.</p>
<p>Comprehensive study documentation is a necessary prerequisite to scientific reproducibility, but many researchers are unable to provide adequate documentation due to limited training, resources, and competing demands for their time. Tools such as Provenance Replay provide a means for ensuring study reproducibility while reducing the documentation burden on bioinformatics users, who may forget to record steps in their computational lab notebooks, or who may not be aware of all of the information that needs to be documented to ensure reproducibility. Provenance Replay largely automates <italic>in silico</italic> reproducibility in QIIME 2, and this approach can provide a model for other scientific computing platforms. Moving forward, computational tools that record data provenance for the user will be a major advancement for methods reproducibility, allowing researchers to more easily corroborate results, learn from the work of others, and build on the conclusions of scientific studies.</p>
</sec>
<sec id="sec005" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pcbi.1011676.s001" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1011676.s001" xlink:type="simple">
<label>S1 File</label>
<caption>
<title>qiime2-provenance-replay-code-and-tutorial.</title>
<p>The provenance replay code, as of QIIME 2 2023.9, and a brief usage tutorial with corresponding data. The two .qzv files included in this supplement are “QIIME Zipped Visualization” files. These can be used with QIIME 2 Provenance Replay to generate replay scripts, can be viewed using QIIME 2 View (<ext-link ext-link-type="uri" xlink:href="https://view.qiime2.org/" xlink:type="simple">https://view.qiime2.org</ext-link>), or can be unzipped with any typical unzip utility as they are .zip files with a specific internal structure that enables QIIME 2 to interpret them.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="pcbi.1011676.ref001"><label>1</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Cacioppo</surname> <given-names>JT</given-names></name>, <name name-style="western"><surname>Kaplan</surname> <given-names>RM</given-names></name>, <name name-style="western"><surname>Krosnick</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Olds</surname> <given-names>JL</given-names></name>, <name name-style="western"><surname>Dean</surname> <given-names>H</given-names></name>. <article-title>Social, behavioral, and economic sciences perspectives on robust and reliable science</article-title>. <source>Report of the Subcommittee on Replicability in Science Advisory Committee to the National Science Foundation Directorate for Social, Behavioral, and Economic Sciences.</source> <year>2015</year>;<fpage>1</fpage>. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.nsf.gov/sbe/AC_Materials/SBE_Robust_and_Reliable_Research_Report.pdf" xlink:type="simple">https://www.nsf.gov/sbe/AC_Materials/SBE_Robust_and_Reliable_Research_Report.pdf</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref002"><label>2</label><mixed-citation publication-type="journal" xlink:type="simple"><collab>University of California Museum of Paleontology</collab>. <source>How Science Works. Understanding Science</source>. <year>2022</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://undsci.berkeley.edu/lessons/pdfs/how_science_works.pdf" xlink:type="simple">https://undsci.berkeley.edu/lessons/pdfs/how_science_works.pdf</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref003"><label>3</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gazzaniga</surname> <given-names>MS</given-names></name>. <source>Psychological science 2018. 6th ed</source>. <name name-style="western"><surname>Norton</surname> <given-names>W. W.</given-names></name>; <year>2018</year>. Available from: <ext-link ext-link-type="uri" xlink:href="http://archive.org/details/dokumen.pub_psychological-science-1-6nbsped-9780393640403" xlink:type="simple">http://archive.org/details/dokumen.pub_psychological-science-1-6nbsped-9780393640403</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nicholas</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Watkinson</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Jamali</surname> <given-names>HR</given-names></name>, <name name-style="western"><surname>Herman</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Tenopir</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Volentine</surname> <given-names>R</given-names></name>, <etal>et al</etal>. <article-title>Peer review: still king in the digital age.</article-title> <source>Learn Publ</source>. <year>2015</year>;<volume>28</volume>: <fpage>15</fpage>–<lpage>21</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1087/20150104" xlink:type="simple">10.1087/20150104</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><collab>Open Science Collaboration</collab>. <article-title>Estimating the reproducibility of psychological science.</article-title> <source>Science</source>. <year>2015</year>;<volume>349</volume>: <fpage>aac4716</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1126/science.aac4716" xlink:type="simple">10.1126/science.aac4716</ext-link></comment> <object-id pub-id-type="pmid">26315443</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Baker</surname> <given-names>M.</given-names></name> <article-title>1,500 scientists lift the lid on reproducibility</article-title>. <source>Nature</source>. <year>2016</year>;<volume>533</volume>: <fpage>452</fpage>–<lpage>454</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/533452a" xlink:type="simple">10.1038/533452a</ext-link></comment> <object-id pub-id-type="pmid">27225100</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref007"><label>7</label><mixed-citation publication-type="other" xlink:type="simple">The Turing Way Community. The Turing Way: A handbook for reproducible, ethical and collaborative research. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.7625728" xlink:type="simple">10.5281/zenodo.7625728</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref008"><label>8</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gundersen</surname> <given-names>OE</given-names></name>, <name name-style="western"><surname>Kjensmo</surname> <given-names>S</given-names></name>. <source>State of the Art: Reproducibility in Artificial Intelligence</source>. <year>2018</year>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref009"><label>9</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Shiffrin</surname> <given-names>RM</given-names></name>, <name name-style="western"><surname>Börner</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Stigler</surname> <given-names>SM</given-names></name>. <article-title>Scientific progress despite irreproducibility: A seeming paradox</article-title>. <source>Proceedings of the National Academy of Sciences</source>. <year>2018</year>;<volume>115</volume>: <fpage>2632</fpage>–<lpage>2639</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1711786114" xlink:type="simple">10.1073/pnas.1711786114</ext-link></comment> <object-id pub-id-type="pmid">29531095</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref010"><label>10</label><mixed-citation publication-type="book" xlink:type="simple"><name name-style="western"><surname>Zhao</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Wilde</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Foster</surname> <given-names>I</given-names></name>. <chapter-title>Applying the Virtual Data Provenance Model.</chapter-title> In: <name name-style="western"><surname>Hutchison</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Kanade</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Kittler</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Kleinberg</surname> <given-names>JM</given-names></name>, <name name-style="western"><surname>Mattern</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Mitchell</surname> <given-names>JC</given-names></name>, <etal>et al</etal>., editors. <source>Provenance and Annotation of Data.</source> <publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer Berlin Heidelberg</publisher-name>; <year>2006</year>. pp. <fpage>148</fpage>–<lpage>161</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/11890850%5F16" xlink:type="simple">10.1007/11890850_16</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Munafò</surname> <given-names>MR</given-names></name>, <name name-style="western"><surname>Nosek</surname> <given-names>BA</given-names></name>, <name name-style="western"><surname>Bishop</surname> <given-names>DVM</given-names></name>, <name name-style="western"><surname>Button</surname> <given-names>KS</given-names></name>, <name name-style="western"><surname>Chambers</surname> <given-names>CD</given-names></name>, <name name-style="western"><surname>Percie du Sert</surname> <given-names>N</given-names></name>, <etal>et al</etal>. <article-title>A manifesto for reproducible science</article-title>. <source>Nature Human Behaviour</source>. <year>2017</year>;<volume>1</volume>: <fpage>1</fpage>–<lpage>9</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41562-016-0021" xlink:type="simple">10.1038/s41562-016-0021</ext-link></comment> <object-id pub-id-type="pmid">33954258</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mesirov</surname> <given-names>JP</given-names></name>. <article-title>Computer science. Accessible reproducible research</article-title>. <source>Science</source>. <year>2010</year>;<volume>327</volume>: <fpage>415</fpage>–<lpage>416</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1126/science.1179653" xlink:type="simple">10.1126/science.1179653</ext-link></comment> <object-id pub-id-type="pmid">20093459</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref013"><label>13</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Köster</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Rahmann</surname> <given-names>S</given-names></name>. <article-title>Snakemake—a scalable bioinformatics workflow engine</article-title>. <source>Bioinformatics</source>. <year>2012</year>;<volume>28</volume>: <fpage>2520</fpage>–<lpage>2522</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/bts480" xlink:type="simple">10.1093/bioinformatics/bts480</ext-link></comment> <object-id pub-id-type="pmid">22908215</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref014"><label>14</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Love</surname> <given-names>MI</given-names></name>, <name name-style="western"><surname>Soneson</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Hickey</surname> <given-names>PF</given-names></name>, <name name-style="western"><surname>Johnson</surname> <given-names>LK</given-names></name>, <name name-style="western"><surname>Pierce</surname> <given-names>NT</given-names></name>, <name name-style="western"><surname>Shepherd</surname> <given-names>L</given-names></name>, <etal>et al</etal>. <article-title>Tximeta: Reference sequence checksums for provenance identification in RNA-seq.</article-title> <source>PLoS Comput Biol</source>. <year>2020</year>;<volume>16</volume>: <fpage>e1007664</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1007664" xlink:type="simple">10.1371/journal.pcbi.1007664</ext-link></comment> <object-id pub-id-type="pmid">32097405</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref015"><label>15</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Khan</surname> <given-names>FZ</given-names></name>, <name name-style="western"><surname>Soiland-Reyes</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Sinnott</surname> <given-names>RO</given-names></name>, <name name-style="western"><surname>Lonie</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Goble</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Crusoe</surname> <given-names>MR</given-names></name>. <article-title>Sharing interoperable workflow provenance: A review of best practices and their practical application in CWLProv.</article-title> <source>Gigascience</source>. <year>2019</year>;<volume>8</volume>: <fpage>giz095</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/gigascience/giz095" xlink:type="simple">10.1093/gigascience/giz095</ext-link></comment> <object-id pub-id-type="pmid">31675414</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref016"><label>16</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bechhofer</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Buchan</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>De Roure</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Missier</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Ainsworth</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Bhagat</surname> <given-names>J</given-names></name>, <etal>et al</etal>. <article-title>Why linked data is not enough for scientists.</article-title> <source>Future Gener Comput Syst</source>. <year>2013</year>;<volume>29</volume>: <fpage>599</fpage>–<lpage>611</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.future.2011.08.004" xlink:type="simple">10.1016/j.future.2011.08.004</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref017"><label>17</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bolyen</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Rideout</surname> <given-names>JR</given-names></name>, <name name-style="western"><surname>Dillon</surname> <given-names>MR</given-names></name>, <name name-style="western"><surname>Bokulich</surname> <given-names>NA</given-names></name>, <name name-style="western"><surname>Abnet</surname> <given-names>CC</given-names></name>, <name name-style="western"><surname>Al-Ghalith</surname> <given-names>GA</given-names></name>, <etal>et al</etal>. <article-title>Reproducible, interactive, scalable and extensible microbiome data science using QIIME 2</article-title>. <source>Nat Biotechnol</source>. <year>2019</year>;<volume>37</volume>: <fpage>852</fpage>–<lpage>857</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41587-019-0209-9" xlink:type="simple">10.1038/s41587-019-0209-9</ext-link></comment> <object-id pub-id-type="pmid">31341288</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref018"><label>18</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Brown</surname> <given-names>AM</given-names></name>, <name name-style="western"><surname>Bolyen</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Raspet</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Altin</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Ladner</surname> <given-names>JT</given-names></name>. <article-title>PepSIRF + QIIME 2: software tools for automated, reproducible analysis of highly-multiplexed serology data.</article-title> <source>arXiv [q-bio.QM].</source> <year>2022</year>. Available from: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2207.11509" xlink:type="simple">http://arxiv.org/abs/2207.11509</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref019"><label>19</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bolyen</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Dillon</surname> <given-names>MR</given-names></name>, <name name-style="western"><surname>Bokulich</surname> <given-names>NA</given-names></name>, <name name-style="western"><surname>Ladner</surname> <given-names>JT</given-names></name>, <name name-style="western"><surname>Larsen</surname> <given-names>BB</given-names></name>, <name name-style="western"><surname>Hepp</surname> <given-names>CM</given-names></name>, <etal>et al</etal>. <article-title>Reproducibly sampling SARS-CoV-2 genomes across time, geography, and viral diversity.</article-title> <source>F1000Res</source>. <year>2020</year>;<volume>9</volume>: <fpage>657</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.12688/f1000research.24751.2" xlink:type="simple">10.12688/f1000research.24751.2</ext-link></comment> <object-id pub-id-type="pmid">33500774</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref020"><label>20</label><mixed-citation publication-type="journal" xlink:type="simple"><collab>Python Software Foundation</collab>. <source>Python Language Reference. Python Software Foundation</source>; <year>2001</year>. Available from: <ext-link ext-link-type="uri" xlink:href="http://www.python.org" xlink:type="simple">http://www.python.org</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref021"><label>21</label><mixed-citation publication-type="book" xlink:type="simple"><name name-style="western"><surname>Hagberg</surname> <given-names>AA</given-names></name>, <name name-style="western"><surname>Schult</surname> <given-names>DA</given-names></name>, <name name-style="western"><surname>Swart</surname> <given-names>PJ</given-names></name>. <chapter-title>Exploring Network Structure, Dynamics, and Function using NetworkX.</chapter-title> In: <name name-style="western"><surname>Varoquaux</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Vaught</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Millman</surname> <given-names>J</given-names></name>, editors. <source>Proceedings of the 7th Python in Science Conference.</source> <publisher-name>Pasadena</publisher-name>, <publisher-loc>CA USA</publisher-loc>; <year>2008</year>. pp. <fpage>11</fpage>–<lpage>15</lpage>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref022"><label>22</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Simonov K</surname> <given-names>YAML</given-names></name> <article-title>community. PyYAML</article-title>. <source>The YAML Project</source>; <year>2006</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://pyyaml.org/" xlink:type="simple">https://pyyaml.org/</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref023"><label>23</label><mixed-citation publication-type="other" xlink:type="simple">Boulogne F, Mangin O, Verney L, Al E. BibTexParser. sciunto-org; Available from: <ext-link ext-link-type="uri" xlink:href="https://bibtexparser.readthedocs.io/en/master/" xlink:type="simple">https://bibtexparser.readthedocs.io/en/master/</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref024"><label>24</label><mixed-citation publication-type="journal" xlink:type="simple"><collab>Pallets</collab>. <source>Click. Pallets</source>; <year>2014</year>. Available from: <ext-link ext-link-type="uri" xlink:href="https://click.palletsprojects.com/en/7.0.x/" xlink:type="simple">https://click.palletsprojects.com/en/7.0.x/</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1011676.ref025"><label>25</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Davis</surname> <given-names>FD</given-names></name>. <article-title>Perceived Usefulness, Perceived Ease of Use, and User Acceptance of Information Technology.</article-title> <source>Miss Q.</source> <year>1989</year>;<volume>13</volume>: <fpage>319</fpage>–<lpage>340</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2307/249008" xlink:type="simple">10.2307/249008</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref026"><label>26</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Keefe</surname> <given-names>CR</given-names></name>. <article-title>Improving In Silico Scientific Reproducibility With Provenance Replay Software.</article-title> <name name-style="western"><surname>Caporaso</surname> <given-names>JG</given-names></name>, editor. <source>Master of Science, Northern Arizona University.</source> <year>2022</year>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6084/m9.figshare.24217224.v1" xlink:type="simple">10.6084/m9.figshare.24217224.v1</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref027"><label>27</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Borsom</surname> <given-names>EM</given-names></name>, <name name-style="western"><surname>Conn</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Keefe</surname> <given-names>CR</given-names></name>, <name name-style="western"><surname>Herman</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Orsini</surname> <given-names>GM</given-names></name>, <name name-style="western"><surname>Hirsch</surname> <given-names>AH</given-names></name>, <etal>et al</etal>. <source>Predicting neurodegenerative disease using pre-pathology gut microbiota composition: a longitudinal study in mice modeling Alzheimer’s disease pathologies</source>. <year>2022</year> <month>Apr</month>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.21203/rs.3.rs-1538737/v1" xlink:type="simple">10.21203/rs.3.rs-1538737/v1</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1011676.ref028"><label>28</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Weninger</surname> <given-names>SN</given-names></name>, <name name-style="western"><surname>Herman</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Meyer</surname> <given-names>RK</given-names></name>, <name name-style="western"><surname>Beauchemin</surname> <given-names>ET</given-names></name>, <name name-style="western"><surname>Kangath</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Lane</surname> <given-names>AI</given-names></name>, <etal>et al</etal>. <article-title>Oligofructose improves small intestinal lipid-sensing mechanisms via alterations to the small intestinal microbiota.</article-title> <source>Microbiome</source>. <year>2023</year>;<volume>11</volume>: <fpage>169</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/s40168-023-01590-2" xlink:type="simple">10.1186/s40168-023-01590-2</ext-link></comment> <object-id pub-id-type="pmid">37533066</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref029"><label>29</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gonzalez</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Navas-Molina</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Kosciolek</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>McDonald</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Vázquez-Baeza</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Ackermann</surname> <given-names>G</given-names></name>, <etal>et al</etal>. <article-title>Qiita: rapid, web-enabled microbiome meta-analysis.</article-title> <source>Nat Methods</source>. <year>2018</year>;<volume>15</volume>: <fpage>796</fpage>–<lpage>798</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41592-018-0141-9" xlink:type="simple">10.1038/s41592-018-0141-9</ext-link></comment> <object-id pub-id-type="pmid">30275573</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref030"><label>30</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ziemski</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Adamov</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kim</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Flörl</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Bokulich</surname> <given-names>NA</given-names></name>. <article-title>Reproducible acquisition, management and meta-analysis of nucleotide sequence (meta)data using q2-fondue.</article-title> <source>Bioinformatics</source>. <year>2022</year>;<volume>38</volume>: <fpage>5081</fpage>–<lpage>5091</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btac639" xlink:type="simple">10.1093/bioinformatics/btac639</ext-link></comment> <object-id pub-id-type="pmid">36130056</object-id></mixed-citation></ref>
<ref id="pcbi.1011676.ref031"><label>31</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Afgan</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Baker</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Batut</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>van den Beek</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Bouvier</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Čech</surname> <given-names>M</given-names></name>, <etal>et al</etal>. <article-title>The Galaxy platform for accessible, reproducible and collaborative biomedical analyses: 2018 update</article-title>. <source>Nucleic Acids Res</source>. <year>2018</year>;<volume>46</volume>: <fpage>W537</fpage>–<lpage>W544</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gky379" xlink:type="simple">10.1093/nar/gky379</ext-link></comment> <object-id pub-id-type="pmid">29790989</object-id></mixed-citation></ref>
</ref-list>
</back>
<sub-article article-type="aggregated-review-documents" id="pcbi.1011676.r001" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1011676.r001</article-id>
<title-group>
<article-title>Decision Letter 0</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Alber</surname>
<given-names>Mark</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kreft</surname>
<given-names>Jan-Ulrich  </given-names>
</name>
<role>Guest Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2023</copyright-year>
<copyright-holder>Alber, Kreft</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1011676" document-id-type="doi" document-type="article" id="rel-obj001" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>0</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">19 Sep 2023</named-content>
</p>
<p>Dear Dr Caporaso,</p>
<p>Thank you very much for submitting your manuscript "Facilitating Bioinformatics Reproducibility with QIIME 2 Provenance Replay" for consideration at PLOS Computational Biology.</p>
<p>As with all papers reviewed by the journal, your manuscript was reviewed by members of the editorial board and by several independent reviewers. In light of the reviews (below this email), we would like to invite the resubmission of a significantly-revised version that takes into account the reviewers' comments.</p>
<p>Dear Authors,</p>
<p>Thank you for submitting your manuscript to PLoS Computational Biology. It has now been reviewed by two experts who have done a thorough job and made constructive suggestions for improvement. I have also read the manuscript and agree with their assessment and recommendations.</p>
<p>I agree it will be important to reduce the reliance on the master's thesis by going back to the original citations and this would also be fair to the authors of these studies.</p>
<p>Regarding the tension between developing the concepts about reproducibility and explaining the actual work sufficiently, I agree that the latter is essential but if the authors can manage to better explain the conceptual work in a concise way, I would be happy to have that included. However it must become clearer than it is at the moment. I felt the explanation was not clear enough for someone not familiar with the ideas already and for those who already are, it is less of a problem but not doing a great job as to what are new ideas and why they are important. If you decide to develop this into a separate publication that is fine. If you want to keep it, make it clearer while keeping it concise.</p>
<p>Best wishes,</p>
<p>Jan-Ulrich Kreft</p>
<p>Guest Editor</p>
<p>We cannot make any decision about publication until we have seen the revised manuscript and your response to the reviewers' comments. Your revised manuscript is also likely to be sent to reviewers for further evaluation.</p>
<p>When you are ready to resubmit, please upload the following:</p>
<p>[1] A letter containing a detailed list of your responses to the review comments and a description of the changes you have made in the manuscript. Please note while forming your response, if your article is accepted, you may have the opportunity to make the peer review history publicly available. The record will include editor decision letters (with reviews) and your responses to reviewer comments. If eligible, we will contact you to opt in or out.</p>
<p>[2] Two versions of the revised manuscript: one with either highlights or tracked changes denoting where the text has been changed; the other a clean version (uploaded as the manuscript file).</p>
<p>Important additional instructions are given below your reviewer comments.</p>
<p>Please prepare and submit your revised manuscript within 60 days. If you anticipate any delay, please let us know the expected resubmission date by replying to this email. Please note that revised manuscripts received after the 60-day due date may require evaluation and peer review similar to newly submitted manuscripts.</p>
<p>Thank you again for your submission. We hope that our editorial process has been constructive so far, and we welcome your feedback at any time. Please don't hesitate to contact us if you have any questions or comments.</p>
<p>Sincerely,</p>
<p>Jan-Ulrich Kreft</p>
<p>Guest Editor</p>
<p>PLOS Computational Biology</p>
<p>Mark Alber</p>
<p>Section Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************</p>
<p>Dear Authors,</p>
<p>Thank you for submitting your manuscript to PLoS Computational Biology. It has now been reviewed by two experts who have done a thorough job and made constructive suggestions for improvement. I have also read the manuscript and agree with their assessment and recommendations.</p>
<p>I agree it will be important to reduce the reliance on the master's thesis by going back to the original citations and this would also be fair to the authors of these studies.</p>
<p>Regarding the tension between developing the concepts about reproducibility and explaining the actual work sufficiently, I agree that the latter is essential but if the authors can manage to better explain the conceptual work in a concise way, I would be happy to have that included. However it must become clearer than it is at the moment. I felt the explanation was not clear enough for someone not familiar with the ideas already and for those who already are, it is less of a problem but not doing a great job as to what are new ideas and why they are important. If you decide to develop this into a separate publication that is fine. If you want to keep it, make it clearer while keeping it concise.</p>
<p>Best wishes,</p>
<p>Jan-Ulrich Kreft</p>
<p>Guest Editor</p>
<p>Reviewer's Responses to Questions</p>
<p><bold>Comments to the Authors:</bold></p>
<p><bold>Please note here if the review is uploaded as an attachment.</bold></p>
<p>Reviewer #1: The presented software article motivates and presents a python application capable of extracting metadata from a QIIME2 analysis. This metadata serves as documentation of the analyses performed within QIIME2, and can be used to re-run the analyses. The advantage of programmatically extracting this documentation, or data provenance information, is that it requires very little time and effort from the researcher and guarantees retrieval of accurate and complete information, which can be tedious to achieve manually.</p>
<p>The application is a useful and much needed addition to the QIIME2 software suite, making it easier and faster for researchers to provide provenance information in scientific publications and related works. The idea is not particular novel, and references to exiting work need to be added, but means to increase the chances that scientists provide good provenance data are very important.</p>
<p>The presentation of the application, its output, and the placement in the field should be improved before publication as outlined in the major comments below.</p>
<p>Major comments:</p>
<p>- The last sentence of the abstract says ‘… and discuss considerations for bioinformatics developers who wish to implement similar functionality in their software.’ I cannot find this back in the Results or Future Directions sections. I think it would be useful to list the key elements that should be reported for microbiome research, and any suggestions you might have for developers in the field.</p>
<p>- The manuscript heavily relies on the reference Keefe 2022, a master thesis which is not easily accessible, only upon request. I feel this is a major limitation and all information necessary to fully understand the paper, its motivation and the application should be given in the manuscript or supplement.</p>
<p>- No proper introduction of the field of workflow documentation is given, and references to similar approaches are missing. This needs to be added. For example, Snakemake generates an analysis graph (DAG) and provides documentation during runtime, the config file allows ‘replay’ at any time. Other post-analysis documentation tools exist but are not discussed and cited. See how Love et al, 2020, provided an overview over existing applications for provenance tracking of RNA-seq data for how it should look like.</p>
<p>- I like the reference to the reproducibility classes of the Turing Way categories/Gundersen. The figure legend (figure 1) is a bit hard to digest though. The concept is clear and easy to understand, so make the figure legend easy to comprehend as well, cite the references in the legend as well, and keep the details for the main text.</p>
<p>- The authors do not place their application in the context of Figure 1. Please do so and discuss the challenges associated with the more general reproducibility classes.</p>
<p>- Figure 2 is extremely difficult to interpret with the limited information given. Even if the reader is very familiar with the steps of the analysis, it is impossible to match input and output data and actions/methods to the DAG. The DAG needs to be sensibly annotated to be useful. Same holds for the Action details which are almost not human-readable in the presented form. Make sure the action details are easily matched to the DAG and the input/output files of the user, to make the DAG a useful feature of Provenance Replay.</p>
<p>Minor comments:</p>
<p>- Avoid spoken expressions like “aren’t” in the text (abstract).</p>
<p>References:</p>
<p>[Love et al., 2020] Love MI, Soneson C, Hickey PF, Johnson LK, Pierce NT, et al. (2020) Tximeta: Reference sequence checksums for provenance identification in RNA-seq. PLOS Computational Biology 16(2): e1007664. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1007664" xlink:type="simple">https://doi.org/10.1371/journal.pcbi.1007664</ext-link></p>
<p>Reviewer #2: Keefe et al have developed a useful software component that will, as advertised, serve as an aid to reproducibility in microbiome data analysis. The problem that the authors identify is real: even sophisticated bioinformatics scientists have trouble recording and documenting complex analytical workflows. The solution presented by the authors is to let the computer record the steps and software dependencies from the analysis, a task which seems to be so much more appropriate for computers than humans. While this collusion only covers one field of research on one bioinformatics platform, it does point the way forward for other areas of research and software.</p>
<p>Major comments</p>
<p>1. There is a tension within the article between developing broad ideas about reproducibility vs introducing the software. As the primary purpose of this article is to introduce new software, the authors should not develop broad ideas that are not absolutely necessary to motivate and describe the software presented. As just one example, is it necessary to merge reproducibility categories from two organizations into a unified hierarchy? Can’t the software’s utility be justified separately in the context of each framework, without merging them? Personally, I think the broad ideas are interesting and would make for an impactful, but separate, manuscript. The authors do not have the space to properly develop broad ideas about reproducibility here, and any attempts to do so will distract from the main focus of the paper.</p>
<p>2. Much of the background for this manuscript seems to come from a Master’s thesis written by the lead author. The thesis is primarily used as a source for the goals that the software seeks to meet. I tried but was unable to obtain a copy of the thesis from the library at Northern Arizona University. I understand referencing a thesis, but am uncomfortable with the degree to which the basis of the article relies on a thesis from the lead author. As I’ve not read the thesis, I’m not sure about the degree to which the goals are crafted by the author himself vs. aggregated from other sources. If the goals outlined in O’Keefe 2022 are collected from other sources, please cite them directly. Throughout the article, the reliance on O’Keefe 2022 should be minimized in favor of briefly describing the rationale and citing original sources.</p>
<p>3. In the Design and Implementation section, the authors say, “initial software design was based on literature review, existing API targets, and discussion with QIIME 2 developers, as well as an initial requirements engineering process and formal focus groups with prospective users.” Even if you don’t have space for all the details, we at least need to hear about the requirements engineering process, and we need to see some data from the focus groups. Because you have gone above and beyond the design practice in many labs, the readers need to see what a robust design process looks like. And if you did it right, these data will provide strong support for your design.</p>
<p>4. In Availability and Future Directions, the authors say, “this approach can provide a model for other scientific computing platforms,” but then don’t address HOW other computing platforms might use ideas from QIIME 2 Provenance Replay. This is the one glaring question that needs to be addressed by the authors. It seems to me that the design was so successful in QIIME 2 because the framework was built from the ground up to keep track of software used in each step. Is the plan to decouple QIIME 2 from microbiome data analysis and extend into other areas? To have separate QIIME 2-like frameworks for each field of research? To develop a general tool that tracks software dependencies on the command line? How does Galaxy fit into this picture? Do Jupyter Notebooks and RMarkdown have anything to learn from this? If you have to trim other parts of the article to make space for exploring this question, please do. This is where you can emphasize the payoff from your software and examine some broad ideas while staying relevant to the topic of the article.</p>
<p>Minor comments</p>
<p>1. The reproducibility hierarchy introduced by the authors, merging work from Gunderson and Kjensmo with ideas from the Turing Way, is presented in an overly technical way. Instead of “we contextualize key factors in generalizability within the broader goals of reproducible research,” you could just as easily say “we developed an expanded hierarchy that includes ideas from both groups.” The figure legend is impossible to understand without a full understanding of the text, which is itself difficult. This prevents the figure from serving as a visual introduction to your ideas about reproducibility. This may be a moot point if you follow the suggestion from major comment 1.</p>
<p>2. You probably mean to cite the Turing Way as “The Turing Way Community. (2021, November 10). The Turing Way: A handbook for reproducible, ethical and collaborative research. Zenodo. <ext-link ext-link-type="uri" xlink:href="http://doi.org/10.5281/zenodo.3233853”" xlink:type="simple">http://doi.org/10.5281/zenodo.3233853”</ext-link></p>
<p>**********</p>
<p><bold>Have the authors made all data and (if applicable) computational code underlying the findings in their manuscript fully available?</bold></p>
<p>The <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/materials-and-software-sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data and code underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data and code should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data or code —e.g. participant privacy or use of data from a third party—those must be specified.</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.</p>
<p>Reviewer #1: <bold>Yes: </bold>Julia C Engelmann</p>
<p>Reviewer #2: No</p>
<p><underline>Figure Files:</underline></p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <underline><ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com</ext-link></underline>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email us at <underline><email xlink:type="simple">figures@plos.org</email></underline>.</p>
<p><underline>Data Requirements:</underline></p>
<p>Please note that, as a condition of publication, PLOS' data policy requires that you make available all data used to draw the conclusions outlined in your manuscript. Data must be deposited in an appropriate repository, included within the body of the manuscript, or uploaded as supporting information. This includes all numerical values that were used to generate graphs, histograms etc.. For an example in PLOS Biology see here: <ext-link ext-link-type="uri" xlink:href="http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5" xlink:type="simple">http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5</ext-link>.</p>
<p><underline>Reproducibility:</underline></p>
<p>To enhance the reproducibility of your results, we recommend that you deposit your laboratory protocols in protocols.io, where a protocol can be assigned its own identifier (DOI) such that it can be cited independently in the future. Additionally, PLOS ONE offers an option to publish peer-reviewed clinical study protocols. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link></p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pcbi.1011676.r002">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1011676.r002</article-id>
<title-group>
<article-title>Author response to Decision Letter 0</article-title>
</title-group>
<related-object document-id="10.1371/journal.pcbi.1011676" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj002" link-type="rebutted-decision-letter" object-id="10.1371/journal.pcbi.1011676.r001" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">3 Nov 2023</named-content>
</p>
<supplementary-material id="pcbi.1011676.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1011676.s002" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">plos-comp-bio-response-to-reviewers.pdf</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1011676.r003" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1011676.r003</article-id>
<title-group>
<article-title>Decision Letter 1</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Alber</surname>
<given-names>Mark</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kreft</surname>
<given-names>Jan-Ulrich  </given-names>
</name>
<role>Guest Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2023</copyright-year>
<copyright-holder>Alber, Kreft</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1011676" document-id-type="doi" document-type="article" id="rel-obj003" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">10 Nov 2023</named-content>
</p>
<p>Dear Dr Caporaso,</p>
<p>Thank you for the thorough revision addressing all comments raised by the reviewers satisfactorily. There is therefore no need to send it out to the reviewers again and we are pleased to inform you that your manuscript 'Facilitating Bioinformatics Reproducibility with QIIME 2 Provenance Replay' has been provisionally accepted for publication in PLOS Computational Biology.</p>
<p>There are a few minor changes that I would like you to consider at the proofing stage. In the Design and Implementation section, you report percentages of focus group participants with 4 significant figures although there were only 19. The values should be rounded to two sig. fig. Reporting the TAM sores of e.g. 5.82 would be easier to interpret if you stated the maximum score. In Results you use the term computational literacy, it may be nicer to use experience instead.</p>
<p>Before your manuscript can be formally accepted you will need to complete some formatting changes, which you will receive in a follow up email. A member of our team will be in touch with a set of requests.</p>
<p>Please note that your manuscript will not be scheduled for publication until you have made the required changes, so a swift response is appreciated.</p>
<p>IMPORTANT: The editorial review process is now complete. PLOS will only permit corrections to spelling, formatting or significant scientific errors from this point onwards. Requests for major changes, or any which affect the scientific understanding of your work, will cause delays to the publication date of your manuscript.</p>
<p>Should you, your institution's press office or the journal office choose to press release your paper, you will automatically be opted out of early publication. We ask that you notify us now if you or your institution is planning to press release the article. All press must be co-ordinated with PLOS.</p>
<p>Thank you again for supporting Open Access publishing; we are looking forward to publishing your work in PLOS Computational Biology. </p>
<p>Best regards,</p>
<p>Jan-Ulrich Kreft</p>
<p>Guest Editor</p>
<p>PLOS Computational Biology</p>
<p>Mark Alber</p>
<p>Section Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************************************************</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1011676.r004" specific-use="acceptance-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1011676.r004</article-id>
<title-group>
<article-title>Acceptance letter</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Alber</surname>
<given-names>Mark</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kreft</surname>
<given-names>Jan-Ulrich  </given-names>
</name>
<role>Guest Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2023</copyright-year>
<copyright-holder>Alber, Kreft</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1011676" document-id-type="doi" document-type="article" id="rel-obj004" link-type="peer-reviewed-article"/>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">20 Nov 2023</named-content>
</p>
<p>PCOMPBIOL-D-23-00967R1 </p>
<p>Facilitating Bioinformatics Reproducibility with QIIME 2 Provenance Replay</p>
<p>Dear Dr Caporaso,</p>
<p>I am pleased to inform you that your manuscript has been formally accepted for publication in PLOS Computational Biology. Your manuscript is now with our production department and you will be notified of the publication date in due course.</p>
<p>The corresponding author will soon be receiving a typeset proof for review, to ensure errors have not been introduced during production. Please review the PDF proof of your manuscript carefully, as this is the last chance to correct any errors. Please note that major changes, or those which affect the scientific understanding of the work, will likely cause delays to the publication date of your manuscript. </p>
<p>Soon after your final files are uploaded, unless you have opted out, the early version of your manuscript will be published online. The date of the early version will be your article's publication date. The final article will be published to the same URL, and all versions of the paper will be accessible to readers.</p>
<p>Thank you again for supporting PLOS Computational Biology and open-access publishing. We are looking forward to publishing your work! </p>
<p>With kind regards,</p>
<p>Anita Estes</p>
<p>PLOS Computational Biology | Carlyle House, Carlyle Road, Cambridge CB4 3DN | United Kingdom <email xlink:type="simple">ploscompbiol@plos.org</email> | Phone +44 (0) 1223-442824 | <ext-link ext-link-type="uri" xlink:href="http://ploscompbiol.org" xlink:type="simple">ploscompbiol.org</ext-link> | @PLOSCompBiol</p>
</body>
</sub-article>
</article>