<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id>
<journal-title-group>
<journal-title>PLOS Computational Biology</journal-title>
</journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-23-01757</article-id>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Artificial intelligence</subject><subj-group><subject>Machine learning</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Bioengineering</subject><subj-group><subject>Macromolecular engineering</subject><subj-group><subject>Protein engineering</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Engineering and technology</subject><subj-group><subject>Bioengineering</subject><subj-group><subject>Macromolecular engineering</subject><subj-group><subject>Protein engineering</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Optimization</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Physiology</subject><subj-group><subject>Immune physiology</subject><subj-group><subject>Antibodies</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune system proteins</subject><subj-group><subject>Antibodies</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune system proteins</subject><subj-group><subject>Antibodies</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Biochemistry</subject><subj-group><subject>Proteins</subject><subj-group><subject>Immune system proteins</subject><subj-group><subject>Antibodies</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Numerical analysis</subject><subj-group><subject>Extrapolation</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Biochemistry</subject><subj-group><subject>Proteins</subject><subj-group><subject>Protein domains</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Neural networks</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Neural networks</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Benchmarking uncertainty quantification for protein engineering</article-title>
<alt-title alt-title-type="running-head">Benchmarking uncertainty quantification for protein engineering</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-6466-1401</contrib-id>
<name name-style="western">
<surname>Greenman</surname> <given-names>Kevin P.</given-names></name>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/validation/">Validation</role>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
<xref ref-type="fn" rid="econtrib001"><sup>‡</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-8601-6040</contrib-id>
<name name-style="western">
<surname>Amini</surname> <given-names>Ava P.</given-names></name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff004"><sup>4</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9045-6826</contrib-id>
<name name-style="western">
<surname>Yang</surname> <given-names>Kevin K.</given-names></name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff004"><sup>4</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>Department of Chemical Engineering, Catholic Institute of Technology, Cambridge, Massachusetts, United States of America</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Department of Chemistry, Catholic Institute of Technology, Cambridge, Massachusetts, United States of America</addr-line>
</aff>
<aff id="aff003">
<label>3</label>
<addr-line>Department of Chemical Engineering, Massachusetts Institute of Technology, Cambridge, Massachusetts, United States of America</addr-line>
</aff>
<aff id="aff004">
<label>4</label>
<addr-line>Microsoft Research, Cambridge, Massachusetts, United States of America</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Kolodny</surname> <given-names>Rachel</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>University of Haifa, ISRAEL</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<fn fn-type="other" id="econtrib001">
<p>‡Work done in part during an internship at Microsoft Research</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">ava.amini@microsoft.com</email> (APA); <email xlink:type="simple">yang.kevin@microsoft.com</email> (KKY)</corresp>
</author-notes>
<pub-date pub-type="collection">
<month>1</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="epub">
<day>7</day>
<month>1</month>
<year>2025</year>
</pub-date>
<volume>21</volume>
<issue>1</issue>
<elocation-id>e1012639</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>10</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-year>2025</copyright-year>
<copyright-holder>Greenman et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pcbi.1012639"/>
<abstract>
<p>Machine learning sequence-function models for proteins could enable significant advances in protein engineering, especially when paired with state-of-the-art methods to select new sequences for property optimization and/or model improvement. Such methods (Bayesian optimization and active learning) require calibrated estimations of model uncertainty. While studies have benchmarked a variety of deep learning uncertainty quantification (UQ) methods on standard and molecular machine-learning datasets, it is not clear if these results extend to protein datasets. In this work, we implemented a panel of deep learning UQ methods on regression tasks from the Fitness Landscape Inference for Proteins (FLIP) benchmark. We compared results across different degrees of distributional shift using metrics that assess each UQ method’s accuracy, calibration, coverage, width, and rank correlation. Additionally, we compared these metrics using one-hot encoding and pretrained language model representations, and we tested the UQ methods in retrospective active learning and Bayesian optimization settings. Our results indicate that there is no single best UQ method across all datasets, splits, and metrics, and that uncertainty-based sampling is often unable to outperform greedy sampling in Bayesian optimization. These benchmarks enable us to provide recommendations for more effective design of biological sequences using machine learning.</p>
</abstract>
<abstract abstract-type="summary">
<title>Author summary</title>
<p>Protein engineering has previously benefited from the use of machine learning models to guide the choice of new experiments. In many cases, the goal of conducting new experiments is optimizing for a property or improving the machine learning model. Many standard methods for these two tasks require good estimates of the uncertainty in the model’s predictions. Several methods for quantifying this uncertainty exist and have been benchmarked on datasets from other domains (e.g. small molecules), but it is not clear whether these results also apply for proteins. To address this, we evaluated a range of uncertainty quantification approaches on tasks derived from a protein-focused benchmark dataset. We tested performance on different degrees of distributional shift between the training and testing sets and on different representations of the sequences, and we assessed performance in terms of several standard metrics. Finally, we used the uncertainties for property optimization and model improvement. Our findings indicate that no single uncertainty estimation method excels across all scenarios. Moreover, uncertainty-based strategies for property optimization often did not outperform simpler methods that did not consider uncertainty. This research offers insights for the more efficacious application of machine learning in the realm of biological sequence design.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/100000001</institution-id>
<institution>National Science Foundation</institution>
</institution-wrap>
</funding-source>
<award-id>1745302</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-6466-1401</contrib-id>
<name name-style="western">
<surname>Greenman</surname> <given-names>Kevin P.</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>K.P.G. was supported by a Microsoft Research (<ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/en-us/research/" xlink:type="simple">https://www.microsoft.com/en-us/research/</ext-link>) micro-internship and by the National Science Foundation Graduate Research Fellowship Program under Grant No. 1745302. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="6"/>
<table-count count="0"/>
<page-count count="19"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>PLOS Publication Stage</meta-name>
<meta-value>vor-update-to-uncorrected-proof</meta-value>
</custom-meta>
<custom-meta>
<meta-name>Publication Update</meta-name>
<meta-value>2025-01-17</meta-value>
</custom-meta>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>The code for the models, uncertainty methods, and evaluation metrics in this work is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/microsoft/protein-uq" xlink:type="simple">https://github.com/microsoft/protein-uq</ext-link> and archived at <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/doi/10.5281/zenodo.7839141" xlink:type="simple">https://zenodo.org/doi/10.5281/zenodo.7839141</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Machine learning (ML) has already begun to accelerate the field of protein engineering by providing low-cost predictions of phenomena that require time- and resource-intensive labeling by experiments or physics-based simulations [<xref ref-type="bibr" rid="pcbi.1012639.ref001">1</xref>]. It is often necessary to have an estimate of model uncertainty in addition to the property prediction, as the performance of an ML model can be highly dependent on the domain shift between its training and testing data [<xref ref-type="bibr" rid="pcbi.1012639.ref002">2</xref>]. Because protein engineering data is often collected in a manner that violates the independent and identically distributed (i.i.d.) assumptions of many ML approaches [<xref ref-type="bibr" rid="pcbi.1012639.ref003">3</xref>], tailored ML methods are required to guide the selection of new experiments from a protein landscape. Uncertainty quantification (UQ) can inform the selection of experiments in order to improve a ML model or optimize protein function through active learning (AL) or Bayesian optimization (BO).</p>
<p>In chemistry and materials science, several studies have benchmarked common UQ methods against one another on standard datasets and have used or developed appropriate metrics to quantify the quality of these uncertainty estimates [<xref ref-type="bibr" rid="pcbi.1012639.ref004">4</xref>–<xref ref-type="bibr" rid="pcbi.1012639.ref009">9</xref>]. These works have illustrated that the best choice of UQ method can depend on the dataset and other considerations such as representation and scaling. While some protein engineering work has leveraged uncertainty estimates, these studies have been mostly limited to single UQ methods such as convolutional neural network (CNN) ensembles [<xref ref-type="bibr" rid="pcbi.1012639.ref010">10</xref>] or Gaussian processes (GPs) [<xref ref-type="bibr" rid="pcbi.1012639.ref011">11</xref>, <xref ref-type="bibr" rid="pcbi.1012639.ref012">12</xref>].</p>
<p>Gruver et al. compared CNN ensembles to GPs (using traditional representations and pre-trained BERT [<xref ref-type="bibr" rid="pcbi.1012639.ref013">13</xref>] language model embeddings) in Bayesian optimization tasks [<xref ref-type="bibr" rid="pcbi.1012639.ref014">14</xref>]. They found that CNN ensembles are often more robust to distribution shift than other types of models. Additionally, they report that most model types have more poorly calibrated uncertainties on out-of-domain samples. However, a more comprehensive study of CNN UQ methods, evaluated using a variety of uncertainty quality metrics, has not been done. A comparison of uncertainty methods on different protein representations (e.g., one-hot encodings or embeddings from protein language models) in an active learning setting is also lacking.</p>
<p>In this work, we evaluate a panel of UQ methods for protein sequence-function prediction on a set of standardized, public protein datasets (<xref ref-type="fig" rid="pcbi.1012639.g001">Fig 1</xref>). Our chosen datasets included splits with varied degrees of domain extrapolation, which enabled method evaluation in a setting similar to what might be experienced while collecting new experimental data for protein engineering. We assessed each model using a variety of metrics that captured different aspects of desired performance, including accuracy, calibration, coverage, width, and rank correlation. Additionally, we compared the performance of the UQ methods on one-hot encoded sequence representations and on embeddings computed from the ESM-1b protein masked language model [<xref ref-type="bibr" rid="pcbi.1012639.ref015">15</xref>]. We find that the quality of UQ estimates are dependent on the landscape, task, and embedding, and that no single method consistently outperforms all others. We also evaluated the UQ methods in an active learning setting with several acquisition functions, and demonstrated that uncertainty-based sampling often outperforms random sampling (especially in later stages of active learning), although better calibrated uncertainty does not necessarily equate to better active learning. Finally, we tested the UQ methods in Bayesian optimization and found that while BO typically outperformed random sampling, none were better than a greedy baseline. We envision that the understanding gained from this work will enable more effective development and application of UQ techniques to machine learning in protein engineering.</p>
<fig id="pcbi.1012639.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Approach, datasets, and tasks.</title>
<p>(A) Schematic of the approach for benchmarking uncertainty quantification (UQ) in machine learning for protein engineering. A panel of UQ methods were evaluated on protein fitness datasets to assess the quality of the uncertainty estimates and their utility in active learning and Bayesian optimization. (B) Our study utilized three protein datasets/landscapes and different train-validation-test split tasks within each dataset. These datasets and tasks covered a range of sample diversities and domain shifts (task difficulties).</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g001" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec002" sec-type="conclusions">
<title>Results and discussion</title>
<sec id="sec003">
<title>Uncertainty quantification</title>
<p>Our first goal was to evaluate the calibration and quality of a variety of UQ methods. We implemented seven uncertainty methods for this benchmark: linear Bayesian ridge regression (BRR) [<xref ref-type="bibr" rid="pcbi.1012639.ref016">16</xref>, <xref ref-type="bibr" rid="pcbi.1012639.ref017">17</xref>], Gaussian processes (GPs) [<xref ref-type="bibr" rid="pcbi.1012639.ref018">18</xref>], and five methods using variations on a convolutional neural network (CNN) architecture. The CNN implementation from FLIP [<xref ref-type="bibr" rid="pcbi.1012639.ref003">3</xref>] provided the core architecture used by our dropout [<xref ref-type="bibr" rid="pcbi.1012639.ref019">19</xref>], ensemble [<xref ref-type="bibr" rid="pcbi.1012639.ref020">20</xref>], evidential [<xref ref-type="bibr" rid="pcbi.1012639.ref021">21</xref>], mean-variance estimation (MVE) [<xref ref-type="bibr" rid="pcbi.1012639.ref022">22</xref>], and last-layer stochastic variational inference (SVI) [<xref ref-type="bibr" rid="pcbi.1012639.ref023">23</xref>] methods. Additional model details are provided in the Methods section.</p>
<p>The landscapes used in this work were taken from the Fitness Landscape Inference for Proteins (FLIP) benchmark [<xref ref-type="bibr" rid="pcbi.1012639.ref003">3</xref>]. These include the binding domain of an immunoglobulin binding protein (GB1), adeno-associated virus stability (AAV), and thermostability (Meltome) data landscapes, which cover a large sequence space and a broad range of protein families. The FLIP benchmark includes several train-test splits, or tasks, for each landscape. Most of these tasks are designed to mimic common, real-world data collection scenarios and are thus a more realistic assessment of generalization than random train-test splits. However, random splits are also included as a point of reference. We chose 8 of the 15 FLIP tasks to benchmark the panel of uncertainty methods. We selected these tasks to be representative of several regimes of domain shift—random sampling with no domain shift (AAV/Random, Meltome/Random, and GB1/Random); the highest (and most relevant) domain-shift regimes (AAV/Random vs. Designed and GB1/1 vs. Rest); and less aggressive domain shifts (AAV/7 vs. Rest, GB1/2 vs. Rest, and GB1/3 vs. Rest). The Datasets section of the Methods provides notes on the nomenclature used for these tasks.</p>
<p>We trained the seven models on each of the eight tasks described above and evaluated their performance on the test set using the metrics described in the Evaluation Metrics section. We compare model calibration and accuracy in <xref ref-type="fig" rid="pcbi.1012639.g002">Fig 2</xref> and the percent coverage versus average width relative to range in <xref ref-type="fig" rid="pcbi.1012639.g003">Fig 3</xref>. These figures illustrate the results for models trained on the embeddings from a pretrained ESM language model [<xref ref-type="bibr" rid="pcbi.1012639.ref015">15</xref>]; the corresponding results using one-hot encodings are shown in Figs A and B in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref>.</p>
<fig id="pcbi.1012639.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Miscalibration area vs. root mean square error (RMSE).</title>
<p>For the (A) AAV, (B) Meltome, and (C) GB1 landscapes. Miscalibration area (also called the area under the calibration error curve or AUCE) quantifies the absolute difference between the calibration plot and perfect calibration. It is desirable to have a model that is both accurate and well-calibrated, so the best performing points are those closest to the lower left corner of the plots. Each point represents an average of 5 models trained using different random seeds for initialization of the CNN parameters and batching / stochastic gradient descent. Fig A in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref> shows the corresponding results for the OHE representation. See the Uncertainty Methods section for an explanation of points for which experiments were not feasible (e.g. there is no GP Continuous model result for the AAV landscape due to memory constraints for training these models).</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g002" xlink:type="simple"/>
</fig>
<fig id="pcbi.1012639.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Coverage vs. average width / range.</title>
<p>For the (A) AAV, (B) Meltome, and (C) GB1 landscapes. Coverage is the percentage of true values that fall within the 95% confidence interval (±2<italic>σ</italic>) of each prediction, and the width is the size of the 95% confidence region relative to the range of the training set (4<italic>σ</italic>/<italic>R</italic> where <italic>R</italic> is the range of the training set). A good model exhibits high coverage and low width, which corresponds to the upper left of each plot. The horizontal dashed line indicates 95% coverage. Each point represents an average of 5 models trained using different random seeds for initialization of the CNN parameters and batching / stochastic gradient descent. Fig B in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref> shows the corresponding results for the OHE representation. See the Uncertainty Methods section for an explanation of several points for which experiments were not feasible (e.g. there is no GP Continuous model result for the AAV landscape due to memory constraints for training these models).</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g003" xlink:type="simple"/>
</fig>
<p>As expected, the splits with the least required domain extrapolation tend to have more accurate models (lower RMSE; <xref ref-type="fig" rid="pcbi.1012639.g002">Fig 2</xref>). However, the relationship between miscalibration area and extrapolation is less clear; some models are highly calibrated on the most difficult (highest domain shift) splits, while others are poorly calibrated even on random splits. There is no single method that performs consistently well across splits and landscapes, but some trends can be observed. For example, ensembling is often one of the highest accuracy CNN models, but also one of the most poorly calibrated. Additionally, GP and BRR models are often better calibrated than CNN models. For the AAV and GB1 landscapes (<xref ref-type="fig" rid="pcbi.1012639.g002">Fig 2a and 2c</xref>), model miscalibration area usually increases slightly while RMSE increases more substantially with increasing domain shift.</p>
<p>In addition to accuracy and calibration, we assessed each method in terms of the coverage and width of its uncertainty estimates. A good uncertainty method results in high coverage (ideally, the true value falls within the 95% confidence region 95% of the time) while still maintaining a small average width. The latter is necessary because predicting a very large and uniform value of uncertainty for every point would result in good coverage, so coverage alone is not sufficient. <xref ref-type="fig" rid="pcbi.1012639.g003">Fig 3</xref> illustrates that many methods perform relatively well in either coverage or width (corresponding to the the top and left limits of the plot, respectively), but few methods perform well in both. Similarly to <xref ref-type="fig" rid="pcbi.1012639.g002">Fig 2</xref>, there is some observable trend that more challenging splits are further from the optimal part (upper left) of the plot; this trend is more clear for the GB1 splits (<xref ref-type="fig" rid="pcbi.1012639.g003">Fig 3b</xref>) than for the AAV splits. Most models trained on the AAV landscape (<xref ref-type="fig" rid="pcbi.1012639.g003">Fig 3a</xref>) have a similar average width/range ratio for all splits, but for the GB1 landscape (<xref ref-type="fig" rid="pcbi.1012639.g003">Fig 3c</xref>), this ratio typically increases as the domain shift increases. The locations of the sets of points for each model type shared some similarities across landscapes. CNN SVI often has low coverage and low width, CNN MVE often has moderate coverage and moderate width, and CNN Evidential and BRR often have high coverage and high width. These trends across landscapes could point to a general problem of under- or over-confidence with some model types, and indicates that post-hoc calibration may be necessary. The results for all prediction and uncertainty metrics (along with their standard deviations across 5 different initialization seeds) are shown in Tables A to AR in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref>.</p>
<p>We next assessed how target predictions and uncertainty estimates depended on the degree of domain shift. Across datasets and splits, we compared the ranking performance of each method in terms of predictions relative to true values and uncertainty estimates relative to true errors (ESM in <xref ref-type="fig" rid="pcbi.1012639.g004">Fig 4</xref> and one-hot encodings (OHE) in Fig C in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref>). The splits are ordered according to domain shift within their respective landscapes (lowest to highest shift from left to right). We observe that the rank correlation of the predictions to the true labels generally decreases moving from less to more domain shift within a landscape, consistent with expectation, with the exception of AAV/Random vs. Designed models performing better than AAV/7 vs. Rest models (<xref ref-type="fig" rid="pcbi.1012639.g004">Fig 4a</xref>). Most methods exhibit similar performance in Spearman rank correlations of predictions to targets (<italic>ρ</italic>) within the same task. For many tasks, GP and BRR models perform as well or better than CNN models. Performance on Spearman rank correlations of uncertainties to prediction residuals (<italic>ρ</italic><sub><italic>unc</italic></sub>) is generally much worse than that on <italic>ρ</italic>, with some results showing negative correlation (<xref ref-type="fig" rid="pcbi.1012639.g004">Fig 4b</xref>). MVE and evidential uncertainty methods are most performant in <italic>ρ</italic><sub><italic>unc</italic></sub> for most cases of low to moderate domain shift. Most methods have <italic>ρ</italic><sub><italic>unc</italic></sub> near zero for the most challenging splits. Despite the relatively good performance of MVE on tasks with low to moderate domain shift, it performs poorly in cases of high domain shift, which is consistent with its intended use as an estimator of aleatoric (data-dependent) uncertainty.</p>
<fig id="pcbi.1012639.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g004</object-id>
<label>Fig 4</label>
<caption>
<title>Spearman rank correlations.</title>
<p>Of (A) predictions (<italic>ρ</italic>) and (B) uncertainties (<italic>ρ</italic><sub><italic>unc</italic></sub>) vs. extrapolation. Within each landscape (AAV, Meltome, and GB1), splits are ordered by the amount of domain shift between train and test sets, with the lowest domain shift on the left and the highest domain shift on the right. Error bars on the CNN results represent the 95% confidence interval calculated from 5 different random seeds for initialization of the CNN parameters and batching / stochastic gradient descent. Fig C in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref> shows the corresponding results for the OHE representation. See the Uncertainty Methods section for an explanation of several points for which experiments were not feasible.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g004" xlink:type="simple"/>
</fig>
<p>We find that the models trained on ESM embeddings outperform those trained on one-hot encodings in 21 out of 51 cases for rank correlation of test set predictions, and 29 out of 51 cases for rank correlation of test set uncertainties. The relative performance of the two representations on prediction and uncertainty rank correlation is shown in Fig D in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref>. In terms of predictions, ESM embeddings often yield substantially better performance for tasks with high domain shift (e.g. GB1/1 vs. Rest and Meltome/Random), while OHE performs slightly better on tasks with lower domain shift (e.g. AAV/Random and GB1/3 vs. Rest). The relative uncertainty rank correlation performance, on the other hand, does not have a clear relationship to domain shift.</p>
<p>Since there is no single best UQ method across datasets, splits, and metrics, it is prudent for practitioners to quantify the performance of uncertainty estimates on each new task and to prioritize metrics according to the situation (e.g. prioritize high coverage over low width in high-risk or safety-critical situations).</p>
</sec>
<sec id="sec004">
<title>Active learning</title>
<p>In protein engineering, the purpose of uncertainty estimation is typically to intelligently prioritize sample acquisition to facilitate downstream experimentation. One such use case is in active learning, where uncertainty estimates are used to inform sampling with the goal of improving model predictions overall (i.e., to achieve an accurate model with less training data; <xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5a</xref>). Having assessed the calibration and accuracy of the panel of UQ methods above, we next evaluated whether uncertainty-based active learning could make the learning process more sample-efficient. Across all datasets and splits using the pretrained ESM embeddings, data acquisition was simulated as iterative selection from the data library according to a given sampling strategy (acquisition function; see <xref ref-type="sec" rid="sec007">Methods</xref> for details). The results are summarized in <xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5</xref> for Spearman rank correlation (<italic>ρ</italic>) on three methods and one split per landscape, and additional results are shown in the Figs H to BH in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref> for other metrics, uncertainty methods, and splits. Across most models, the performance difference between the start of active learning (10% of training data) and end of active learning (100% of training data) is relatively small, and many models begin to plateau in performance before reaching 100% of training data. In addition to the active learning experiments run with 10% of the training data in the initial sample, we also show results starting with 1% and 5% of the training data in Figs F and G in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref>, respectively. These results showed worse initial performance given the smaller initial training data, but otherwise similar trends to the trials starting at 10%.</p>
<fig id="pcbi.1012639.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g005</object-id>
<label>Fig 5</label>
<caption>
<title>Active learning.</title>
<p>(A) Schematic of active learning approach. A model is trained on an initial dataset, and is then retrained in each iteration by adding more points to the training set based on some selection criteria. (B-D) Uncertainty-guided active learning in protein sequence-function prediction. Spearman rank correlation of predictions (<italic>ρ</italic>) for the CNN ensemble, CNN evidential, and GP methods evaluated on the AAV/Random (B), Meltome/Random (C), and GB1/Random (D) splits. The “random” strategy acquired sequences with all unseen points having equal probabilities, the “explorative sample” strategy acquired sequences with random sampling weighted by uncertainty, and the “explorative greedy” strategy acquired the previously unseen sequences with the highest uncertainty. See the Uncertainty Methods section for an explanation of why GP experiments for the AAV landscape were not feasible.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g005" xlink:type="simple"/>
</fig>
<p>The “explorative greedy” and “explorative sample” acquisition functions (which sample based on uncertainty alone or sample randomly weighted by uncertainty, respectively) sometimes outperform random sampling, but this is not true across all methods and landscapes (<xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5b–5d</xref>). In some cases, the performance of the uncertainty-based sampling strategies also varies depending on the fraction of the total training data available to the model. For example, for the Meltome/Random split and CNN evidential model (<xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5c</xref>), explorative greedy sampling results in a decrease in model performance after the first round of active learning while the explorative sample strategy increases performance. By the fourth round of active learning for this task, the two explorative strategies outperform random sampling. This indicates that in the early stages of active learning when a model’s uncertainty estimates are poorly calibrated, it may be advantageous to sample with at least some randomness included in an uncertainty-based acquisition function. We also analyzed how the mean test set uncertainty changed as more data was acquired during active learning. Fig E in <xref ref-type="supplementary-material" rid="pcbi.1012639.s001">S1 Appendix</xref> illustrates that in some cases, the mean test set uncertainty decreased with increasing training data, while in other cases, it increased. While one might expect adding more data from uncertain sequences would always cause a decrease in the mean uncertainty, this assumes that (1) the added training data comes from the same distribution as the test data, and (2) the uncertainty estimates are well-calibrated at the beginning and remain so after retraining with more data. Overall, the results indicate that uncertainty-informed active learning can outperform random sampling and thus lead to more accurate machine learning models with fewer training points needing to be measured (<xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5b–5d</xref>).</p>
</sec>
<sec id="sec005">
<title>Bayesian optimization</title>
<p>Uncertainty estimates can also be leveraged to identify top-performing sequences with as few samples as possible. The true objective can be approximated with a surrogate model, and we can use the predictions of this model as well as the uncertainties in these model predictions to guide a search toward higher or lower values of the true objective. This approach, referred to as Bayesian optimization, can also be represented by <xref ref-type="fig" rid="pcbi.1012639.g005">Fig 5a</xref>. Bayesian optimization methods use an acquisition function computed from the predicted mean and uncertainty values to trade off exploration and exploitation when choosing new sequences to sample, in contrast to the acquisition functions used in active learning that maximize exploration.</p>
<p>We compared two popular acquisition functions (upper-confidence bound (UCB) and Thompson sampling (TS)) against random and greedy baselines (see <xref ref-type="sec" rid="sec007">Methods</xref> for details). UCB and TS are intended to accelerate identification of top-performing instances over random and greedy baselines by taking into account both the predicted objective and the uncertainty in that prediction. <xref ref-type="fig" rid="pcbi.1012639.g006">Fig 6</xref> shows the % of top-100 scores found versus fraction of training data seen for three UQ methods and one split per landscape. Across these cases, the uncertainty-based methods almost always perform better than the random baseline but never outperform greedily sampling the sequences with the highest predicted values. In most cases, UCB sampling performs about the same as greedy sampling, with the notable exception of evidential uncertainty on the AAV sampled dataset (<xref ref-type="fig" rid="pcbi.1012639.g006">Fig 6a</xref>), for which it performed worse than random sampling. The performance of TS (a probabalistic method) was typically intermediate between greedy/UCB (deterministic methods) and random sampling. Overall, the uncertainty-based methods do not outperform greedy baselines, suggesting that better UQ methods are needed for protein engineering or that the landscapes studied here are simple enough that they can be optimized by pure exploitation.</p>
<fig id="pcbi.1012639.g006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012639.g006</object-id>
<label>Fig 6</label>
<caption>
<title>Bayesian optimization.</title>
<p>(A-C) Bayesian optimization in protein sequence-function prediction. % of top-100 scores in training set found for the CNN ensemble, CNN evidential, and GP methods evaluated on the AAV/Random (A), Meltome/Random (B), and GB1/Random (C) splits. The “greedy” strategy acquired sequences with the best predicted property values. The “UCB” and “TS” strategies acquired sequences based on the upper confidence bound (UCB) and Thompson sampling (TS) approaches, respectively. The “random” strategy acquired sequences with all unseen points having equal probabilities. See the Uncertainty Methods section for an explanation of why GP experiments for the AAV landscape were not feasible. Note that in several plots, including the Gaussian process plots for Meltome and GB1 and the evidential plot for Meltome, the “greedy” strategy performance is nearly identical to and is covered up by the “UCB” strategy.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.g006" xlink:type="simple"/>
</fig>
</sec>
</sec>
<sec id="sec006" sec-type="conclusions">
<title>Conclusions</title>
<p>Calibrated uncertainty estimations for ML predictions of biomolecular properties are necessary for effective model improvement using active learning or property optimization using Bayesian methods. In this work, we benchmarked a panel of uncertainty quantification (UQ) methods on protein datasets, including on train-test splits that are representative of real-world data collection practices. After evaluating each method based on accuracy, calibration, coverage, width, rank correlation, and performance in active learning and Bayesian optimization, we find that there is no method that performs consistently well across all metrics or all landscapes and splits.</p>
<p>We also examined how models trained using one-hot-encoding representations of sequences compare to those trained on more informative and generalizable representations such as embeddings from a pretrained ESM language model. This comparison illustrated that while the pretrained embeddings do improve model accuracy and uncertainty correlation/calibration in some cases, particularly on splits with higher domain shifts, this is not universally true and in some cases makes performance worse.</p>
<p>While the UQ evaluation metrics used in this work provide valuable information, they are ultimately only a proxy for expected performance in Bayesian optimization and active learning. We found that UQ evaluation metrics are not well-correlated with gains in accuracy from one active learning iteration to another on these datasets. This suggests that future work in UQ should include retrospective Bayesian optimization and/or active learning studies rather than relying on UQ evaluation metrics alone. Our retrospective active learning studies using holdouts of the training sets demonstrate that many of the uncertainty methods outperform random sampling baselines. In some of our experiments, we observe that the uncertainty-based sampling strategies perform worse than random sampling during the earliest stages of active learning, then perform better as a model’s accuracy and quality of uncertainty estimates improve in later stages. Our Bayesian optimization experiments demonstrate that while uncertainty-based methods typically perform better than a random approach, including uncertainty in the acquisition function does not necessarily confer a benefit over a greedy approach that considers only property predictions. While previous work has successfully used BO to optimize proteins [<xref ref-type="bibr" rid="pcbi.1012639.ref024">24</xref>–<xref ref-type="bibr" rid="pcbi.1012639.ref027">27</xref>], it is not clear that uncertainty helped these campaigns because they do not compare directly to greedy sampling. Taken together, these results indicate that there is a need for further development of UQ methods and/or sampling strategies to improve protein engineering performance in AL and BO settings.</p>
<p>Future work in this area could expand on methods (e.g. Bayesian neural networks [<xref ref-type="bibr" rid="pcbi.1012639.ref028">28</xref>] and conformal prediction [<xref ref-type="bibr" rid="pcbi.1012639.ref029">29</xref>, <xref ref-type="bibr" rid="pcbi.1012639.ref030">30</xref>]), metrics (e.g. sharpness [<xref ref-type="bibr" rid="pcbi.1012639.ref005">5</xref>], dispersion [<xref ref-type="bibr" rid="pcbi.1012639.ref031">31</xref>], and tightness [<xref ref-type="bibr" rid="pcbi.1012639.ref032">32</xref>]), and representations (e.g. ESM-2 [<xref ref-type="bibr" rid="pcbi.1012639.ref033">33</xref>] or using an attention layer rather than mean aggregation on our ESM-1b embeddings). In addition to further study of existing methods, future work should focus on designing novel UQ methods that give a clear performance benefit in AL and BO for protein engineering. Future work could also examine other sampling regimes for AL and BO, such as training an initial model on random data and sampling from designed sequences. While this work considered uncertainty predictions as directly output by the models, further study is needed to understand the effects of post-hoc calibration methods (e.g. scalar recalibration [<xref ref-type="bibr" rid="pcbi.1012639.ref031">31</xref>] or CRUDE [<xref ref-type="bibr" rid="pcbi.1012639.ref034">34</xref>]). Future work should consider additional active learning and Bayesian optimization strategies, such as those that consider batch diversity in the acquisition function [<xref ref-type="bibr" rid="pcbi.1012639.ref035">35</xref>], and methods that consider the desired domain shift. Ultimately, this work contributes to a more thorough understanding of the performance and utility of UQ for sequence-function models and provides a foundation for future work to enable more effective protein engineering.</p>
</sec>
<sec id="sec007" sec-type="materials|methods">
<title>Methods</title>
<sec id="sec008">
<title>Regression tasks</title>
<p>All tasks studied in this work are regression problems, in which we attempt to fit a model to a dataset with <inline-formula id="pcbi.1012639.e001"><alternatives><graphic id="pcbi.1012639.e001g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e001" xlink:type="simple"/><mml:math display="inline" id="M1"><mml:mi mathvariant="script">D</mml:mi></mml:math></alternatives></inline-formula> data points (<italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>). <italic>x</italic><sub><italic>i</italic></sub> is a protein sequence representation (either a one-hot encoding or an embedding vector from an ESM language model), and <inline-formula id="pcbi.1012639.e002"><alternatives><graphic id="pcbi.1012639.e002g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e002" xlink:type="simple"/><mml:math display="inline" id="M2"><mml:mrow><mml:msub><mml:mi>y</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> is a scalar-valued target property from the protein landscapes described in the Datasets section.</p>
</sec>
<sec id="sec009">
<title>Datasets</title>
<p>The landscapes and splits in this work are taken from the FLIP benchmark [<xref ref-type="bibr" rid="pcbi.1012639.ref003">3</xref>]. GB1 is a landscape commonly used for investigating epistasis (interactions between mutations) using the binding domain of protein G, an immunoglobulin binding protein in Streptococcal bacteria. These splits are designed primarily to test generalization from few- to many-mutation sequences. The AAV landscape is based on data collected for the Adeno-associated virus capsid protein, which help the virus integrate a DNA payload into a target cell. The mutations in this landscape are restricted to a subset of positions within a much longer sequence. The Meltome landscape includes data from proteins across 13 different species for a non-protein-specific property (thermostability), so it includes both local and global variations. The total number of data points in the GB1, AAV, and Meltome sets are 8,733, 284,009, and 27,951, respectively. In the AAV set, 82,583 are sampled (mutations) and 201,426 are designed. For AAV, only the 82,583 sampled sequences are used for the Random and 7 vs. Rest tasks, while all 284,009 are used for the Sampled vs. Designed task.</p>
<p>The names of several of the tasks were changed slightly from the original FLIP nomenclature for clarity: GB1/Random was originally called GB1/Sampled, AAV/Random was originally called AAV/Sampled, AAV/7 vs. Rest was originally called AAV/7 vs. Many, AAV/Sampled vs. Designed was originally called AAV/Mut-Des, and Meltome/Random was originally called Meltome/Mixed.</p>
</sec>
<sec id="sec010">
<title>ESM embeddings</title>
<p>We used the pretrained, 650M-parameter ESM-1b model (<monospace specific-use="no-wrap">esm1b_t33_650M_UR50S</monospace>) from [<xref ref-type="bibr" rid="pcbi.1012639.ref015">15</xref>] to generate embeddings of the protein sequences in this study and to compare these embeddings to one-hot encoding representations. Sequence embeddings from the final representation layer (layer 33) were mean pooled per amino acid over the length of each protein sequence, which resulted in a fixed embedding size of 1280 for each sequence. In other words, the output of the ESM-1b model is a tensor of size <italic>L</italic> × 1280, and we averaged over each sequence to obtain a representation vector of size 1280 for each sample.</p>
</sec>
<sec id="sec011">
<title>Base CNN model architectures</title>
<p>The base architecture of all CNN models in this work was taken from the CNNs in the FLIP benchmark [<xref ref-type="bibr" rid="pcbi.1012639.ref003">3</xref>], which took the architecture from previous work [<xref ref-type="bibr" rid="pcbi.1012639.ref036">36</xref>]. For the one-hot encoding inputs (with a vocabulary of 22 tokens), this was comprised of a convolution with 1024 output channels and kernel width 5, a ReLU non-linear activation function, a linear mapping to 2048 dimensions, a max pool over the sequence, and a linear mapping to 1 dimension. For ESM embedding inputs (of size 1280), the architecture was the same except with 1280 input channels rather than 1024, and a linear mapping to 2560 dimensions rather than 2048.</p>
</sec>
<sec id="sec012">
<title>CNN Model training procedures</title>
<p>To train our CNN models, we used a batch size of 256 (GB1, AAV) or 30 (Meltome). Adam [<xref ref-type="bibr" rid="pcbi.1012639.ref037">37</xref>] was used for optimization with the following learning rates: 0.001 for the convolution weights, 0.00005 for the first linear mapping, and 0.000005 for the second linear mapping. Weight decay was set to 0.05 for both the first and second linear mappings. CNNs were trained with early stopping using a patience of 20 epochs. Each model was trained on an NVIDIA Volta V100 GPU. Reported metrics in Figs <xref ref-type="fig" rid="pcbi.1012639.g002">2</xref>–<xref ref-type="fig" rid="pcbi.1012639.g004">4</xref> are the average of training 5 models per split with different seeds for initialization of the CNN parameters and batching / stochastic gradient descent. Code, data, and instructions needed to reproduce results can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/microsoft/protein-uq" xlink:type="simple">https://github.com/microsoft/protein-uq</ext-link>.</p>
</sec>
<sec id="sec013">
<title>Uncertainty methods</title>
<p>For all models and landscapes, the sequences were featurized using either one-hot encodings or embeddings from a pretrained language model (see the ESM Embeddings section).</p>
<p>We used the <monospace specific-use="no-wrap">scikit-learn</monospace> [<xref ref-type="bibr" rid="pcbi.1012639.ref038">38</xref>] implementation of Bayesian ridge regression (BRR) with default hyperparameters. BRR for one-hot encodings of the Meltome/Random split was not feasible because the required work array was too large to perform the computation with standard 32-bit LAPACK in <monospace specific-use="no-wrap">scipy</monospace>.</p>
<p>For Gaussian processes (GPs), we used the GPyTorch [<xref ref-type="bibr" rid="pcbi.1012639.ref039">39</xref>] implementation with the constant mean module, scaled rational quadratic (RQ) kernel covariance module, and Gaussian likelihood. Some GP models (for AAV one-hot encodings and ESM embeddings, and Meltome one-hot encodings) were not feasible to train due to GPU-memory requirements for exact GP models, so these are omitted from the results.</p>
<p>For our uncertainty methods that rely on sampling (dropout, ensemble, and SVI), the final model prediction is defined as the mean of the set of inference samples, and the uncertainty is the standard deviation of these samples. In other words, for a set of predictions <inline-formula id="pcbi.1012639.e003"><alternatives><graphic id="pcbi.1012639.e003g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e003" xlink:type="simple"/><mml:math display="inline" id="M3"><mml:mrow><mml:mi mathvariant="script">E</mml:mi> <mml:mo>=</mml:mo> <mml:mo>{</mml:mo> <mml:msub><mml:mi>G</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>,</mml:mo> <mml:msub><mml:mi>G</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>,</mml:mo> <mml:mo>…</mml:mo> <mml:mo>,</mml:mo> <mml:msub><mml:mi>G</mml:mi> <mml:mi>n</mml:mi></mml:msub> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>}</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> (each coming from an individual model <italic>G</italic><sub><italic>i</italic></sub>), the final prediction is defined as
<disp-formula id="pcbi.1012639.e004"><alternatives><graphic id="pcbi.1012639.e004g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e004" xlink:type="simple"/><mml:math display="block" id="M4"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mover accent="true"><mml:mi>G</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>G</mml:mi> <mml:mo>∈</mml:mo> <mml:mi mathvariant="script">E</mml:mi></mml:mrow></mml:munder> <mml:mfrac><mml:mrow><mml:mi>G</mml:mi> <mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mi>n</mml:mi></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(1)</label></disp-formula>
and the uncertainty <italic>U</italic>(<italic>x</italic>) is defined as
<disp-formula id="pcbi.1012639.e005"><alternatives><graphic id="pcbi.1012639.e005g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e005" xlink:type="simple"/><mml:math display="block" id="M5"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi>U</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msqrt><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>G</mml:mi> <mml:mo>∈</mml:mo> <mml:mi mathvariant="script">E</mml:mi></mml:mrow></mml:munder> <mml:mfrac><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mover accent="true"><mml:mi>G</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>-</mml:mo> <mml:mi>G</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mi>n</mml:mi></mml:mfrac></mml:mrow></mml:msqrt></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(2)</label></disp-formula></p>
<p>The uncertainty is sometimes defined as the variance <italic>U</italic><sup>2</sup>, but using the standard deviation puts the uncertainty in the same units as the predictions.</p>
<p>For dropout uncertainty [<xref ref-type="bibr" rid="pcbi.1012639.ref019">19</xref>], a single model <italic>G</italic> was trained normally. At inference time, we applied <italic>n</italic> = 10 random dropout masks with dropout probability <italic>p</italic> to obtain the set of predictions <inline-formula id="pcbi.1012639.e006"><alternatives><graphic id="pcbi.1012639.e006g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e006" xlink:type="simple"/><mml:math display="inline" id="M6"><mml:mi mathvariant="script">E</mml:mi></mml:math></alternatives></inline-formula> for each input <italic>x</italic><sub><italic>i</italic></sub>. We tested dropout rates of <italic>p</italic> ∈ {0.1, 0.2, 0.3, 0.4, 0.5} and reported the model with the lowest miscalibration area.</p>
<p>Similarly for last-layer stochastic variational inference (SVI) [<xref ref-type="bibr" rid="pcbi.1012639.ref023">23</xref>], we obtained <inline-formula id="pcbi.1012639.e007"><alternatives><graphic id="pcbi.1012639.e007g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e007" xlink:type="simple"/><mml:math display="inline" id="M7"><mml:mi mathvariant="script">E</mml:mi></mml:math></alternatives></inline-formula> using <italic>n</italic> = 10 samples from a set of models where each <italic>G</italic><sub><italic>i</italic></sub> has the weight and bias terms of its last layer themselves sampled from a distribution <italic>q</italic>(<italic>θ</italic>) that has been trained to approximate the true posterior <inline-formula id="pcbi.1012639.e008"><alternatives><graphic id="pcbi.1012639.e008g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e008" xlink:type="simple"/><mml:math display="inline" id="M8"><mml:mrow><mml:mi>p</mml:mi> <mml:mo>(</mml:mo> <mml:mi>θ</mml:mi> <mml:mo>|</mml:mo> <mml:mi mathvariant="script">D</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:math></alternatives></inline-formula>.</p>
<p>Traditional model ensembling calculated <inline-formula id="pcbi.1012639.e009"><alternatives><graphic id="pcbi.1012639.e009g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e009" xlink:type="simple"/><mml:math display="inline" id="M9"><mml:mi mathvariant="script">E</mml:mi></mml:math></alternatives></inline-formula> using <italic>n</italic> = 5 models trained using different random seeds for initialization of the CNN parameters and batching / stochastic gradient descent. The computational cost of this approach is 5 times that of a standard CNN model since the cost scales linearly with the size of the ensemble.</p>
<p>In mean-variance estimation (MVE) models, we adapt the base CNN architecture to produce 2 outputs (<italic>θ</italic> = {<italic>μ</italic>, <italic>σ</italic><sup>2</sup>}) for each data point (<italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>) in the last layer rather than 1, and we train using the negative log-likelihood loss:
<disp-formula id="pcbi.1012639.e010"><alternatives><graphic id="pcbi.1012639.e010g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e010" xlink:type="simple"/><mml:math display="block" id="M10"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="script">L</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>θ</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:mfrac><mml:mn>1</mml:mn> <mml:mi>N</mml:mi></mml:mfrac> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>i</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>N</mml:mi></mml:munderover> <mml:mfrac><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>y</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>-</mml:mo> <mml:mi>μ</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mrow><mml:mn>2</mml:mn> <mml:msup><mml:mi>σ</mml:mi> <mml:mn>2</mml:mn></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac> <mml:mo>+</mml:mo> <mml:mfrac><mml:mn>1</mml:mn> <mml:mn>2</mml:mn></mml:mfrac> <mml:mtext>log</mml:mtext> <mml:mrow><mml:mo>(</mml:mo> <mml:mn>2</mml:mn> <mml:mi>π</mml:mi> <mml:msup><mml:mi>σ</mml:mi> <mml:mn>2</mml:mn></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(3)</label></disp-formula></p>
<p>In practice, the variance (<italic>σ</italic><sup>2</sup>) is clamped to a minimum value of 10<sup>−6</sup> to prevent division by 0.</p>
<p>Evidential deep learning modifies the loss function of the traditional CNN to jointly maximize the model’s fit to data while also minimizing its evidence on errors (increasing uncertainty on unreliable predictions) [<xref ref-type="bibr" rid="pcbi.1012639.ref021">21</xref>]:
<disp-formula id="pcbi.1012639.e011"><alternatives><graphic id="pcbi.1012639.e011g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e011" xlink:type="simple"/><mml:math display="block" id="M11"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="script">L</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:msup><mml:mi mathvariant="script">L</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mi>L</mml:mi> <mml:mi>L</mml:mi></mml:mrow></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>+</mml:mo> <mml:mo>λ</mml:mo> <mml:msup><mml:mi mathvariant="script">L</mml:mi> <mml:mi>R</mml:mi></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(4)</label></disp-formula>
where <inline-formula id="pcbi.1012639.e012"><alternatives><graphic id="pcbi.1012639.e012g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e012" xlink:type="simple"/><mml:math display="inline" id="M12"><mml:mrow><mml:msup><mml:mi mathvariant="script">L</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mi>L</mml:mi> <mml:mi>L</mml:mi></mml:mrow></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula> is the negative log-likelihood loss defined above, <inline-formula id="pcbi.1012639.e013"><alternatives><graphic id="pcbi.1012639.e013g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e013" xlink:type="simple"/><mml:math display="inline" id="M13"><mml:mrow><mml:msup><mml:mi mathvariant="script">L</mml:mi> <mml:mi>R</mml:mi></mml:msup> <mml:mrow><mml:mo>(</mml:mo> <mml:mi>x</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula> is the evidence regularizer as defined in Amini et al. [<xref ref-type="bibr" rid="pcbi.1012639.ref021">21</xref>], and λ controls the trade-off between these two terms. In this study, we use λ = 1 for all evidential models. In these models, the last layer of the model produces 4 outputs <bold>m</bold> = {<italic>γ</italic>, <italic>ν</italic>, <italic>α</italic>, <italic>β</italic>} that parameterize the Normal-Inverse-Gamma distribution. This distribution assumes that targets <italic>y</italic><sub><italic>i</italic></sub> are drawn i.i.d. from a Gaussian distribution with unknown mean and variance <italic>θ</italic> = {<italic>μ</italic>, <italic>σ</italic><sup>2</sup>}, where the mean is drawn from a Gaussian and the variance is drawn from an Inverse-Gamma distribution. The output of the evidential model can be divided into the prediction and the epistemic (model) and aleatoric (data) uncertainty components following the analysis of Amini et al. [<xref ref-type="bibr" rid="pcbi.1012639.ref021">21</xref>]:
<disp-formula id="pcbi.1012639.e014"><alternatives><graphic id="pcbi.1012639.e014g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1012639.e014" xlink:type="simple"/><mml:math display="block" id="M14"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:munder><mml:munder accentunder="true"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi> <mml:mo>[</mml:mo> <mml:mi>μ</mml:mi> <mml:mo>]</mml:mo> <mml:mo>=</mml:mo> <mml:mi>γ</mml:mi></mml:mrow> <mml:mo>︸</mml:mo></mml:munder> <mml:mtext>prediction</mml:mtext></mml:munder> <mml:mo>,</mml:mo> <mml:mspace width="1em"/><mml:munder><mml:munder accentunder="true"><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi> <mml:mrow><mml:mo>[</mml:mo> <mml:msup><mml:mi>σ</mml:mi> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:mfrac><mml:mi>β</mml:mi> <mml:mrow><mml:mi>α</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mrow> <mml:mo>︸</mml:mo></mml:munder> <mml:mtext>aleatoric</mml:mtext></mml:munder> <mml:mo>,</mml:mo> <mml:mspace width="1em"/><mml:munder><mml:munder accentunder="true"><mml:mrow><mml:mtext>Var</mml:mtext> <mml:mrow><mml:mo>[</mml:mo> <mml:mi>μ</mml:mi> <mml:mo>]</mml:mo></mml:mrow> <mml:mo>=</mml:mo> <mml:mfrac><mml:mi>β</mml:mi> <mml:mrow><mml:mi>ν</mml:mi> <mml:mo>(</mml:mo> <mml:mi>α</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn> <mml:mo>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow> <mml:mo>︸</mml:mo></mml:munder> <mml:mtext>epistemic</mml:mtext></mml:munder></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(5)</label></disp-formula></p>
<p>We report the sum of the aleatoric and epistemic uncertainties as the total uncertainty.</p>
</sec>
<sec id="sec014">
<title>Evaluation metrics</title>
<p>To give a comprehensive report of model accuracy, we computed the following metrics on the test sets: root mean square error (RMSE), mean absolute error (MAE), coefficient of determination (<italic>R</italic><sup>2</sup>), and Spearman rank correlation (<italic>ρ</italic>). RMSE is more sensitive to outliers than MAE, so while both are informative independently, the combination of the two gives additional information about the distribution of errors. <italic>R</italic><sup>2</sup> and <italic>ρ</italic> are both unitless and are thus more easily interpreted and compared across datasets.</p>
<p>We evaluated the quality of the uncertainty estimates using four metrics. First, <italic>ρ</italic><sub><italic>unc</italic></sub> is the Spearman rank correlation between uncertainty and absolute prediction error. This metric may be particularly relevant in an active learning context, where one wants to acquire labels for the most uncertain points hoping that these are also the highest-error points. This application does not require the uncertainties to be well-calibrated.</p>
<p>Second, the miscalibration area (also called the area under the calibration error curve or AUCE) quantifies the absolute difference between the calibration plot and perfect calibration in a single number [<xref ref-type="bibr" rid="pcbi.1012639.ref040">40</xref>]. Good calibration may be more important in safety-critical applications.</p>
<p>Following Kompa et al. [<xref ref-type="bibr" rid="pcbi.1012639.ref041">41</xref>], we measured the coverage as the percentage of true values that fall within the 95% confidence interval (±2<italic>σ</italic>) of each prediction. This is a indication of the reliability of the uncertainty estimates. A model with high coverage is appropriately cautious in its predictions, which may be most important in applications where safety is a major consideration.</p>
<p>Kompa et al. [<xref ref-type="bibr" rid="pcbi.1012639.ref041">41</xref>] also define another metric, the width, as the size of the 95% confidence region (4<italic>σ</italic>). We normalized this width relative to the range (<italic>R</italic>) of the training set as 4<italic>σ</italic>/<italic>R</italic> to make these values unitless and thus more interpretable across datasets. The width is a measure of precision in the uncertainty. In practical applications, narrower intervals (lower width) can help in making more precise and cost-effective decisions. Ideal uncertainties have high coverage and low width, but in some cases, there may be trade-offs between the two. For example, wider widths can help to detect distribution shift, but these wider intervals may not be reliable if coverage is low. The coverage and width metrics may also be more easily interpretable than other calibration metrics [<xref ref-type="bibr" rid="pcbi.1012639.ref041">41</xref>].</p>
</sec>
<sec id="sec015">
<title>Active learning</title>
<p>Each active learning run began with a random sample of 1%, 5%, or 10% of the full training data, which was taken from the random splits of the three landscapes. We evaluated several alternatives for adding to this initial dataset using different sampling strategies (acquisition functions): explorative greedy, explorative sample, and random. “Explorative greedy” sampled the sequences with the highest uncertainty; “explorative sample” sampled the data according to the probability of sampling a sequence equal to the ratio of its uncertainty to the sum of all uncertainties in the dataset (i.e. random sampling weighted by uncertainty); and “random” sampled the data uniformly from all unobserved sequences. We employed these sampling strategies 5 times in each active learning run, with the 5 training set sizes equally spaced on a log scale. We repeated this process using 3 folds (different random seeds for sampling initial dataset and “explorative sample” probabilities) and calculated the mean and standard deviation across these folds.</p>
</sec>
<sec id="sec016">
<title>Bayesian optimization</title>
<p>Similarly to active learning, each Bayesian optimization run began with a random sample of 10% of the full training data, which was taken from the random splits of the three landscapes. We used the following acquisition functions: greedy, upper-confidence bound (UCB) [<xref ref-type="bibr" rid="pcbi.1012639.ref042">42</xref>], Thompson sampling (TS) [<xref ref-type="bibr" rid="pcbi.1012639.ref043">43</xref>], and random. In greedy sampling, the sequence with the best predicted value was selected. The UCB strategy added the predicted uncertainties to the predicted values and selected the sequence with the largest sum. For TS, we added each predicted value to a number sampled randomly from a Gaussian distribution with a mean of 0 and a standard deviation of the corresponding predicted uncertainty, and again selected the largest sum. The “random” strategy used in Bayesian optimization was the same as that used in active learning (new points were sampled with uniform probability). As with active learning, we used these strategies 5 times in each run, with the 5 training set sizes equally spaced on a log scale. We report the mean and standard deviation across 3 folds (different random seeds for sampling the initial dataset).</p>
</sec>
</sec>
<sec id="sec017" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pcbi.1012639.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.s001" xlink:type="simple">
<label>S1 Appendix</label>
<caption>
<title>Supporting information.</title>
<p>Code availability, OHE results, OHE vs. ESM comparison, additional prediction and uncertainty evaluation metrics, and additional active learning results.</p>
<p>(PDF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ack>
<p>The authors thank the MIT Lincoln Laboratory Supercloud cluster [<xref ref-type="bibr" rid="pcbi.1012639.ref044">44</xref>] at the Massachusetts Green High Performance Computing Center (MGHPCC) for providing high-performance computing resources to train our machine learning models.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pcbi.1012639.ref001">
<label>1</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Yang</surname> <given-names>KK</given-names></name>, <name name-style="western"><surname>Wu</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Arnold</surname> <given-names>FH</given-names></name>. <article-title>Machine-learning-guided directed evolution for protein engineering</article-title>. <source>Nature Methods</source>. <year>2019</year>;<volume>16</volume>(<issue>8</issue>):<fpage>687</fpage>–<lpage>694</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41592-019-0496-6" xlink:type="simple">10.1038/s41592-019-0496-6</ext-link></comment> <object-id pub-id-type="pmid">31308553</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref002">
<label>2</label>
<mixed-citation publication-type="other" xlink:type="simple">Kendall A, Gal Y. What Uncertainties Do We Need in Bayesian Deep Learning for Computer Vision? In: Guyon I, Luxburg UV, Bengio S, Wallach H, Fergus R, Vishwanathan S, et al., editors. Advances in Neural Information Processing Systems. vol. 30. Curran Associates, Inc.; 2017. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/file/2650d6089a6d640c5e85b2b88265dc2b-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper_files/paper/2017/file/2650d6089a6d640c5e85b2b88265dc2b-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref003">
<label>3</label>
<mixed-citation publication-type="other" xlink:type="simple">Dallago C, Mou J, Johnston KE, Wittmann BJ, Bhattacharya N, Goldman S, et al.. FLIP: Benchmark tasks in fitness landscape inference for proteins; BioRxiv [Preprint]. 2021. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2021.11.09.467890v2" xlink:type="simple">https://www.biorxiv.org/content/10.1101/2021.11.09.467890v2</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref004">
<label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Scalia</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Grambow</surname> <given-names>CA</given-names></name>, <name name-style="western"><surname>Pernici</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Li</surname> <given-names>YP</given-names></name>, <name name-style="western"><surname>Green</surname> <given-names>WH</given-names></name>. <article-title>Evaluating scalable uncertainty estimation methods for deep learning-based molecular property prediction</article-title>. <source>Journal of Chemical Information and Modeling</source>. <year>2020</year>;<volume>60</volume>(<issue>6</issue>):<fpage>2697</fpage>–<lpage>2717</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/acs.jcim.9b00975" xlink:type="simple">10.1021/acs.jcim.9b00975</ext-link></comment> <object-id pub-id-type="pmid">32243154</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref005">
<label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Tran</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Neiswanger</surname> <given-names>W</given-names></name>, <name name-style="western"><surname>Yoon</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Xing</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Ulissi</surname> <given-names>ZW</given-names></name>. <article-title>Methods for comparing uncertainty quantifications for material property predictions</article-title>. <source>Machine Learning: Science and Technology</source>. <year>2020</year>;<volume>1</volume>(<issue>2</issue>):<fpage>025006</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1088/2632-2153/ab7e1a" xlink:type="simple">10.1088/2632-2153/ab7e1a</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref006">
<label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Hirschfeld</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Swanson</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Yang</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Barzilay</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Coley</surname> <given-names>CW</given-names></name>. <article-title>Uncertainty quantification using neural networks for molecular property prediction</article-title>. <source>Journal of Chemical Information and Modeling</source>. <year>2020</year>;<volume>60</volume>(<issue>8</issue>):<fpage>3770</fpage>–<lpage>3780</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/acs.jcim.0c00502" xlink:type="simple">10.1021/acs.jcim.0c00502</ext-link></comment> <object-id pub-id-type="pmid">32702986</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref007">
<label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Nigam</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Pollice</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Hurley</surname> <given-names>MF</given-names></name>, <name name-style="western"><surname>Hickman</surname> <given-names>RJ</given-names></name>, <name name-style="western"><surname>Aldeghi</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Yoshikawa</surname> <given-names>N</given-names></name>, <etal>et al</etal>. <article-title>Assigning confidence to molecular property prediction</article-title>. <source>Expert Opinion on Drug Discovery</source>. <year>2021</year>;<volume>16</volume>(<issue>9</issue>):<fpage>1009</fpage>–<lpage>1023</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1080/17460441.2021.1925247" xlink:type="simple">10.1080/17460441.2021.1925247</ext-link></comment> <object-id pub-id-type="pmid">34126827</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref008">
<label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Soleimany</surname> <given-names>AP</given-names></name>, <name name-style="western"><surname>Amini</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Goldman</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Rus</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Bhatia</surname> <given-names>SN</given-names></name>, <name name-style="western"><surname>Coley</surname> <given-names>CW</given-names></name>. <article-title>Evidential deep learning for guided molecular property prediction and discovery</article-title>. <source>ACS Central Science</source>. <year>2021</year>;<volume>7</volume>(<issue>8</issue>):<fpage>1356</fpage>–<lpage>1367</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/acscentsci.1c00546" xlink:type="simple">10.1021/acscentsci.1c00546</ext-link></comment> <object-id pub-id-type="pmid">34471680</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref009">
<label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Gruich</surname> <given-names>CJ</given-names></name>, <name name-style="western"><surname>Madhavan</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Goldsmith</surname> <given-names>BR</given-names></name>. <article-title>Clarifying trust of materials property predictions using neural networks with distribution-specific uncertainty quantification</article-title>. <source>Machine Learning: Science and Technology</source>. <year>2023</year>;<volume>4</volume>(<issue>2</issue>):<fpage>025019</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref010">
<label>10</label>
<mixed-citation publication-type="other" xlink:type="simple">Mariet Z, Jerfel G, Wang Z, Angermüller C, Belanger D, Vora S, et al. Deep Uncertainty and the Search for Proteins. In: NeurIPS Workshop: Machine Learning for Molecules; 2020. Available from: <ext-link ext-link-type="uri" xlink:href="https://ml4molecules.github.io/papers2020/ML4Molecules_2020_paper_23.pdf" xlink:type="simple">https://ml4molecules.github.io/papers2020/ML4Molecules_2020_paper_23.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref011">
<label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Hie</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Bryson</surname> <given-names>BD</given-names></name>, <name name-style="western"><surname>Berger</surname> <given-names>B</given-names></name>. <article-title>Leveraging Uncertainty in Machine Learning Accelerates Biological Discovery and Design</article-title>. <source>Cell Systems</source>. <year>2020</year>;<volume>11</volume>(<issue>5</issue>):<fpage>461</fpage>–<lpage>477.e9</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.cels.2020.09.007" xlink:type="simple">10.1016/j.cels.2020.09.007</ext-link></comment> <object-id pub-id-type="pmid">33065027</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref012">
<label>12</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Parkinson</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>W</given-names></name>. <article-title>Linear-Scaling kernels for protein sequences and small molecules outperform deep learning while providing uncertainty quantitation and improved interpretability</article-title>. <source>Journal of Chemical Information and Modeling</source>. <year>2023</year>;<volume>63</volume>(<issue>15</issue>):<fpage>4589</fpage>–<lpage>4601</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/acs.jcim.3c00601" xlink:type="simple">10.1021/acs.jcim.3c00601</ext-link></comment> <object-id pub-id-type="pmid">37498239</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref013">
<label>13</label>
<mixed-citation publication-type="other" xlink:type="simple">Devlin J, Chang MW, Lee K, Toutanova K. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding; arXiv:1810.04805 [Preprint]. 2019. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1810.04805" xlink:type="simple">https://arxiv.org/abs/1810.04805</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref014">
<label>14</label>
<mixed-citation publication-type="other" xlink:type="simple">Gruver N, Stanton S, Kirichenko P, Finzi M, Maffettone P, Myers V, et al. Effective surrogate models for protein design with bayesian optimization. In: ICML Workshop on Computational Biology; 2021. Available from: <ext-link ext-link-type="uri" xlink:href="https://icml-compbio.github.io/2021/papers/WCBICML2021_paper_61.pdf" xlink:type="simple">https://icml-compbio.github.io/2021/papers/WCBICML2021_paper_61.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref015">
<label>15</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Rives</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Meier</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Sercu</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Goyal</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Lin</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Liu</surname> <given-names>J</given-names></name>, <etal>et al</etal>. <article-title>Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences</article-title>. <source>Proceedings of the National Academy of Sciences</source>. <year>2021</year>;<volume>118</volume>(<issue>15</issue>):<fpage>e2016239118</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.2016239118" xlink:type="simple">10.1073/pnas.2016239118</ext-link></comment> <object-id pub-id-type="pmid">33876751</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref016">
<label>16</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>MacKay</surname> <given-names>DJ</given-names></name>. <article-title>Bayesian interpolation</article-title>. <source>Neural Computation</source>. <year>1992</year>;<volume>4</volume>(<issue>3</issue>):<fpage>415</fpage>–<lpage>447</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1162/neco.1992.4.3.415" xlink:type="simple">10.1162/neco.1992.4.3.415</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref017">
<label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Tipping</surname> <given-names>ME</given-names></name>. <article-title>Sparse Bayesian learning and the relevance vector machine</article-title>. <source>Journal of Machine Learning Research</source>. <year>2001</year>;<volume>1</volume>(<issue>Jun</issue>):<fpage>211</fpage>–<lpage>244</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref018">
<label>18</label>
<mixed-citation publication-type="other" xlink:type="simple">Williams CK, Rasmussen CE. Gaussian Processes for Machine Learning. vol. 2. MIT Press Cambridge, MA; 2006.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref019">
<label>19</label>
<mixed-citation publication-type="other" xlink:type="simple">Gal Y, Ghahramani Z. Dropout as a Bayesian Approximation: Representing Model Uncertainty in Deep Learning. In: Balcan MF, Weinberger KQ, editors. Proceedings of The 33rd International Conference on Machine Learning. vol. 48 of Proceedings of Machine Learning Research. New York, New York, USA: PMLR; 2016. p. 1050–1059. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v48/gal16.html" xlink:type="simple">https://proceedings.mlr.press/v48/gal16.html</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref020">
<label>20</label>
<mixed-citation publication-type="other" xlink:type="simple">Lakshminarayanan B, Pritzel A, Blundell C. Simple and Scalable Predictive Uncertainty Estimation using Deep Ensembles. In: Guyon I, Luxburg UV, Bengio S, Wallach H, Fergus R, Vishwanathan S, et al., editors. Advances in Neural Information Processing Systems. vol. 30. Curran Associates, Inc.; 2017. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2017/file/9ef2ed4b7fd2c810847ffa5fa85bce38-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper/2017/file/9ef2ed4b7fd2c810847ffa5fa85bce38-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref021">
<label>21</label>
<mixed-citation publication-type="other" xlink:type="simple">Amini A, Schwarting W, Soleimany A, Rus D. Deep Evidential Regression. In: Larochelle H, Ranzato M, Hadsell R, Balcan MF, Lin H, editors. Advances in Neural Information Processing Systems. vol. 33. Curran Associates, Inc.; 2020. p. 14927–14937. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2020/file/aab085461de182608ee9f607f3f7d18f-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper/2020/file/aab085461de182608ee9f607f3f7d18f-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref022">
<label>22</label>
<mixed-citation publication-type="other" xlink:type="simple">Nix DA, Weigend AS. Estimating the mean and variance of the target probability distribution. In: Proceedings of 1994 IEEE International Conference on Neural Networks (ICNN’94). vol. 1. IEEE; 1994. p. 55–60.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref023">
<label>23</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Hoffman</surname> <given-names>MD</given-names></name>, <name name-style="western"><surname>Blei</surname> <given-names>DM</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Paisley</surname> <given-names>J</given-names></name>. <article-title>Stochastic variational inference</article-title>. <source>Journal of Machine Learning Research</source>. <year>2013</year>;.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref024">
<label>24</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Romero</surname> <given-names>PA</given-names></name>, <name name-style="western"><surname>Krause</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Arnold</surname> <given-names>FH</given-names></name>. <article-title>Navigating the protein fitness landscape with Gaussian processes</article-title>. <source>Proceedings of the National Academy of Sciences</source>. <year>2013</year>;<volume>110</volume>(<issue>3</issue>):<fpage>E193</fpage>–<lpage>E201</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1215251110" xlink:type="simple">10.1073/pnas.1215251110</ext-link></comment> <object-id pub-id-type="pmid">23277561</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref025">
<label>25</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Bedbrook</surname> <given-names>CN</given-names></name>, <name name-style="western"><surname>Yang</surname> <given-names>KK</given-names></name>, <name name-style="western"><surname>Rice</surname> <given-names>AJ</given-names></name>, <name name-style="western"><surname>Gradinaru</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Arnold</surname> <given-names>FH</given-names></name>. <article-title>Machine learning to design integral membrane channelrhodopsins for efficient eukaryotic expression and plasma membrane localization</article-title>. <source>PLOS Computational Biology</source>. <year>2017</year>;<volume>13</volume>(<issue>10</issue>):<fpage>e1005786</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1005786" xlink:type="simple">10.1371/journal.pcbi.1005786</ext-link></comment> <object-id pub-id-type="pmid">29059183</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref026">
<label>26</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Bedbrook</surname> <given-names>CN</given-names></name>, <name name-style="western"><surname>Yang</surname> <given-names>KK</given-names></name>, <name name-style="western"><surname>Robinson</surname> <given-names>JE</given-names></name>, <name name-style="western"><surname>Mackey</surname> <given-names>ED</given-names></name>, <name name-style="western"><surname>Gradinaru</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Arnold</surname> <given-names>FH</given-names></name>. <article-title>Machine learning-guided channelrhodopsin engineering enables minimally invasive optogenetics</article-title>. <source>Nature Methods</source>. <year>2019</year>;<volume>16</volume>(<issue>11</issue>):<fpage>1176</fpage>–<lpage>1184</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41592-019-0583-8" xlink:type="simple">10.1038/s41592-019-0583-8</ext-link></comment> <object-id pub-id-type="pmid">31611694</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref027">
<label>27</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Greenhalgh</surname> <given-names>JC</given-names></name>, <name name-style="western"><surname>Fahlberg</surname> <given-names>SA</given-names></name>, <name name-style="western"><surname>Pfleger</surname> <given-names>BF</given-names></name>, <name name-style="western"><surname>Romero</surname> <given-names>PA</given-names></name>. <article-title>Machine learning-guided acyl-ACP reductase engineering for improved in vivo fatty alcohol production</article-title>. <source>Nature Communications</source>. <year>2021</year>;<volume>12</volume>(<issue>1</issue>):<fpage>5825</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41467-021-25831-w" xlink:type="simple">10.1038/s41467-021-25831-w</ext-link></comment> <object-id pub-id-type="pmid">34611172</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref028">
<label>28</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Neal</surname> <given-names>RM</given-names></name>. <source>Bayesian learning for neural networks</source>. <volume>vol. 118</volume>. <publisher-name>Springer Science &amp; Business Media</publisher-name>; <year>2012</year>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref029">
<label>29</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Norinder</surname> <given-names>U</given-names></name>, <name name-style="western"><surname>Carlsson</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Boyer</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Eklund</surname> <given-names>M</given-names></name>. <article-title>Introducing conformal prediction in predictive modeling. A transparent and flexible alternative to applicability domain determination</article-title>. <source>Journal of Chemical Information and Modeling</source>. <year>2014</year>;<volume>54</volume>(<issue>6</issue>):<fpage>1596</fpage>–<lpage>1603</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/ci5001168" xlink:type="simple">10.1021/ci5001168</ext-link></comment> <object-id pub-id-type="pmid">24797111</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref030">
<label>30</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Fannjiang</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Bates</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Angelopoulos</surname> <given-names>AN</given-names></name>, <name name-style="western"><surname>Listgarten</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Jordan</surname> <given-names>MI</given-names></name>. <article-title>Conformal prediction under feedback covariate shift for biomolecular design</article-title>. <source>Proceedings of the National Academy of Sciences</source>. <year>2022</year>;<volume>119</volume>(<issue>43</issue>):<fpage>e2204569119</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.2204569119" xlink:type="simple">10.1073/pnas.2204569119</ext-link></comment> <object-id pub-id-type="pmid">36256807</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref031">
<label>31</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Levi</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Gispan</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Giladi</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Fetaya</surname> <given-names>E</given-names></name>. <article-title>Evaluating and calibrating uncertainty prediction in regression tasks</article-title>. <source>Sensors</source>. <year>2022</year>;<volume>22</volume>(<issue>15</issue>):<fpage>5540</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/s22155540" xlink:type="simple">10.3390/s22155540</ext-link></comment> <object-id pub-id-type="pmid">35898047</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref032">
<label>32</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Gneiting</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Raftery</surname> <given-names>AE</given-names></name>. <article-title>Strictly proper scoring rules, prediction, and estimation</article-title>. <source>Journal of the American Statistical Association</source>. <year>2007</year>;<volume>102</volume>(<issue>477</issue>):<fpage>359</fpage>–<lpage>378</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1198/016214506000001437" xlink:type="simple">10.1198/016214506000001437</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref033">
<label>33</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Lin</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Akin</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Rao</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Hie</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Zhu</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Lu</surname> <given-names>W</given-names></name>, <etal>et al</etal>. <article-title>Evolutionary-scale prediction of atomic-level protein structure with a language model</article-title>. <source>Science</source>. <year>2023</year>;<volume>379</volume>(<issue>6637</issue>):<fpage>1123</fpage>–<lpage>1130</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1126/science.ade2574" xlink:type="simple">10.1126/science.ade2574</ext-link></comment> <object-id pub-id-type="pmid">36927031</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref034">
<label>34</label>
<mixed-citation publication-type="other" xlink:type="simple">Zelikman E, Healy C, Zhou S, Avati A. CRUDE: Calibrating Regression Uncertainty Distributions Empirically; arXiv:2005.12496 [Preprint]. 2021. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2005.12496" xlink:type="simple">https://arxiv.org/abs/2005.12496</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref035">
<label>35</label>
<mixed-citation publication-type="other" xlink:type="simple">Kirsch A, van Amersfoort J, Gal Y. BatchBALD: Efficient and Diverse Batch Acquisition for Deep Bayesian Active Learning. In: Wallach H, Larochelle H, Beygelzimer A, d'Alché-Buc F, Fox E, Garnett R, editors. Advances in Neural Information Processing Systems. vol. 32. Curran Associates, Inc.; 2019. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2019/file/95323660ed2124450caaac2c46b5ed90-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper_files/paper/2019/file/95323660ed2124450caaac2c46b5ed90-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref036">
<label>36</label>
<mixed-citation publication-type="other" xlink:type="simple">Shanehsazzadeh A, Belanger D, Dohan D. Is Transfer Learning Necessary for Protein Landscape Prediction?; arXiv:2011.03443 [Preprint]. 2020. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2011.03443" xlink:type="simple">https://arxiv.org/abs/2011.03443</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref037">
<label>37</label>
<mixed-citation publication-type="other" xlink:type="simple">Kingma DP, Ba J. Adam: A Method for Stochastic Optimization; arXiv:1412.6980 [Preprint]. 2017. Available from: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1412.6980" xlink:type="simple">https://arxiv.org/abs/1412.6980</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref038">
<label>38</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Pedregosa</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Varoquaux</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Gramfort</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Michel</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Thirion</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Grisel</surname> <given-names>O</given-names></name>, <etal>et al</etal>. <article-title>Scikit-learn: Machine Learning in Python</article-title>. <source>Journal of Machine Learning Research</source>. <year>2011</year>;<volume>12</volume>:<fpage>2825</fpage>–<lpage>2830</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref039">
<label>39</label>
<mixed-citation publication-type="other" xlink:type="simple">Gardner J, Pleiss G, Weinberger KQ, Bindel D, Wilson AG. GPyTorch: Blackbox Matrix-Matrix Gaussian Process Inference with GPU Acceleration. In: Bengio S, Wallach H, Larochelle H, Grauman K, Cesa-Bianchi N, Garnett R, editors. Advances in Neural Information Processing Systems. vol. 31. Curran Associates, Inc.; 2018. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2018/file/27e8e17134dd7083b050476733207ea1-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper_files/paper/2018/file/27e8e17134dd7083b050476733207ea1-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref040">
<label>40</label>
<mixed-citation publication-type="other" xlink:type="simple">Gustafsson FK, Danelljan M, Schon TB. Evaluating scalable bayesian deep learning methods for robust computer vision. In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition workshops; 2020. p. 318–319.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref041">
<label>41</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Kompa</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Snoek</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Beam</surname> <given-names>AL</given-names></name>. <article-title>Empirical Frequentist Coverage of Deep Learning Uncertainty Quantification Procedures</article-title>. <source>Entropy</source>. <year>2021</year>;<volume>23</volume>(<issue>12</issue>). <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/e23121608" xlink:type="simple">10.3390/e23121608</ext-link></comment> <object-id pub-id-type="pmid">34945914</object-id></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref042">
<label>42</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Srinivas</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Krause</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kakade</surname> <given-names>SM</given-names></name>, <name name-style="western"><surname>Seeger</surname> <given-names>MW</given-names></name>. <article-title>Information-Theoretic Regret Bounds for Gaussian Process Optimization in the Bandit Setting</article-title>. <source>IEEE Transactions on Information Theory</source>. <year>2012</year>;<volume>58</volume>(<issue>5</issue>):<fpage>3250</fpage>–<lpage>3265</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/TIT.2011.2182033" xlink:type="simple">10.1109/TIT.2011.2182033</ext-link></comment></mixed-citation>
</ref>
<ref id="pcbi.1012639.ref043">
<label>43</label>
<mixed-citation publication-type="other" xlink:type="simple">Chapelle O, Li L. An Empirical Evaluation of Thompson Sampling. In: Shawe-Taylor J, Zemel R, Bartlett P, Pereira F, Weinberger KQ, editors. Advances in Neural Information Processing Systems. vol. 24. Curran Associates, Inc.; 2011. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2011/file/e53a0a2978c28872a4505bdb51db06dc-Paper.pdf" xlink:type="simple">https://proceedings.neurips.cc/paper_files/paper/2011/file/e53a0a2978c28872a4505bdb51db06dc-Paper.pdf</ext-link>.</mixed-citation>
</ref>
<ref id="pcbi.1012639.ref044">
<label>44</label>
<mixed-citation publication-type="other" xlink:type="simple">Reuther A, Kepner J, Byun C, Samsi S, Arcand W, Bestor D, et al. Interactive supercomputing on 40,000 cores for machine learning and data analysis. In: 2018 IEEE High Performance extreme Computing Conference (HPEC). IEEE; 2018. p. 1–6.</mixed-citation>
</ref></ref-list>
</back>
<sub-article article-type="aggregated-review-documents" id="pcbi.1012639.r001" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r001</article-id>
<title-group>
<article-title>Decision Letter 0</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ben-Tal</surname>
<given-names>Nir</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kolodny</surname>
<given-names>Rachel</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2025</copyright-year>
<copyright-holder>Ben-Tal, Kolodny</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="article" id="rel-obj001" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>0</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">11 Feb 2024</named-content>
</p>
<p>Dear  Yang,</p>
<p>Thank you very much for submitting your manuscript "Benchmarking uncertainty quantification for protein engineering" for consideration at PLOS Computational Biology.</p>
<p>As with all papers reviewed by the journal, your manuscript was reviewed by members of the editorial board and by several independent reviewers. In light of the reviews (below this email), we would like to invite the resubmission of a significantly-revised version that takes into account the reviewers' comments.</p>
<p>We cannot make any decision about publication until we have seen the revised manuscript and your response to the reviewers' comments. Your revised manuscript is also likely to be sent to reviewers for further evaluation.</p>
<p>When you are ready to resubmit, please upload the following:</p>
<p>[1] A letter containing a detailed list of your responses to the review comments and a description of the changes you have made in the manuscript. Please note while forming your response, if your article is accepted, you may have the opportunity to make the peer review history publicly available. The record will include editor decision letters (with reviews) and your responses to reviewer comments. If eligible, we will contact you to opt in or out.</p>
<p>[2] Two versions of the revised manuscript: one with either highlights or tracked changes denoting where the text has been changed; the other a clean version (uploaded as the manuscript file).</p>
<p>Important additional instructions are given below your reviewer comments.</p>
<p>Please prepare and submit your revised manuscript within 60 days. If you anticipate any delay, please let us know the expected resubmission date by replying to this email. Please note that revised manuscripts received after the 60-day due date may require evaluation and peer review similar to newly submitted manuscripts.</p>
<p>Thank you again for your submission. We hope that our editorial process has been constructive so far, we apologize for the long time it took, and we welcome your feedback at any time. Please don't hesitate to contact us if you have any questions or comments.</p>
<p>Sincerely,</p>
<p>Rachel Kolodny</p>
<p>Academic Editor</p>
<p>PLOS Computational Biology</p>
<p>Nir Ben-Tal</p>
<p>Section Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************</p>
<p>Reviewer's Responses to Questions</p>
<p><bold>Comments to the Authors:</bold></p>
<p><bold>Please note here if the review is uploaded as an attachment.</bold></p>
<p>Reviewer #1: This paper is about quantifying the ability of different protein sequence-based models to accurately predict uncertainty in different protein design data regimes and modalities. The authors tried difference sequence representations, model architectures and uncertainty quantification methods across FLIP benchmark tasks. Although the amount of experiments involved in this work is impressive, the paper is unfortunately written in a way that mostly lists those dense results, hence making it difficult to get any insight or grasp the relationship between them. Moreover, some of the paper's claims could benefit from better statistical analysis support.</p>
<p>**Major comments**</p>
<p>The paper is results-dense and would benefit from some curation of metrics and/or models that would make it easier to ingest. For example, I suspect that RMSE and correlation are highly correlated as well as calibration plot and uncertainty correlation, but they both have their own distinct figures. Some of these results could be moved to supplementary.</p>
<p>The figures are hard to read or interpret since they tend to have too much information and don’t use the space efficiently by repeating the same axis and legend that end up using all the space. For example, the bar plots in Figure 4 are hard to read, especially the B panel for models that have close to zero correlation that seems to be missing.</p>
<p>Can the authors comment on the apparent strong relationship found between the coverage and width metrics for the same task across different models? It almost points toward models not better quantifying uncertainty per point, but just being more confident in general. It almost feels that a simple re-scaling of the uncertainties on a calibration set from training sets would make those trends and claims vanish.</p>
<p>Claims about models being better than others on task are hard to support without any statistical testing. The fact that using different starting seeds for training the CNN model gives similar performance across Figure 4 almost indicates that some of the trends found in previous figures could also vanish. Also, the active learning experiments with the random acquisition function seem to indicate that a simple re-sampling of the training data could be used to better estimate confidence intervals on metrics and support claims about model performance versus each other. Can the authors re-sample their training data to generate different training data and get a better estimation of their metrics? Can they also choose at least 3-4 different starting points in BO and active learning to not let the starting 10% dictate the conclusion?</p>
<p>Can the authors comment on the expected relationship between the different metrics? It is not clear why the miscalibration curve should be plotted against the RMSE.</p>
<p>**Minor comments**</p>
<p>The authors do cite their uncertainty metrics, but given how central they are to the paper, can they describe them more?</p>
<p>I might have missed these details, but what are the data used for active learning and BO. Is it the designed AAV or just the random? Would it make more sense to start from sampled and then sample from designed? (or maybe it is the case)</p>
<p>In the active learning experiments, are the models getting more confident as they get more details?</p>
<p>A summary table of the different rankings of selected models across tasks would be helpful since the paper has so many results that it is hard to keep track of performance across different regimes and modalities. Ideally, these rankings would be derived from robust statistical testing.</p>
<p>Some curves are not visible in Figure 6 since they are hidden by others.</p>
<p>Reviewer #2: This paper considers the problem of uncertainty prediction for the problem of protein engineering. The paper reports extensive experimentation, with the goal to evaluate different types of predictors, and specifically evaluate their uncertainty predictions. The experiment is conducted in three different types of tasks, with varying degree of domain shift between the training data and test data. The uncertainty prediction is evaluated using different metrics, and through two downstream applications, namely, active learning and Bayesian optimization. The results of the experiments show that:</p>
<p>1. There is no single method that performs better across all tasks.</p>
<p>2. Current uncertainty predictors are not always useful for downstream tasks like active learning (where they are useful in some cases), and Bayesian optimization (where a greedy, uncertainty-agnostic approach performs better).</p>
<p>3. Representing proteins using the ESM language model embeddings rather than one-hot encoding, improves results in some cases but not all.</p>
<p>The main conclusion is that uncertainty predictors cannot be assumed to work out of the box for protein engineering tasks and need to be further developed, and/or carefully evaluated per task.</p>
<p>Strengths:</p>
<p>1. The paper performs an extensive and thorough evaluation of the methods under different varying conditions, and gives both a detailed description of each setup and the big picture of the status of current methods.</p>
<p>2. The conclusion is a useful and practical contribution to the community that often considers relying on such predictions of uncertainty.</p>
<p>Weaknesses:</p>
<p>1. The paper does not present a novel model or evaluation methodology, and is a relatively straightforward implementation of various experiments. In my view this should not prevent the paper from being accepted as the experimentation is thorough and therefore valuable to the community.</p>
<p>2. The results do not show a clear winning method that can directly inform practitioners on ways to improve their research on downstream tasks. However, like I mentioned above, there is value in empirically demonstrating the limitations of current models, which should serve as a warning for practitioners of downstream tasks, and an invitation to researchers to perform more research on uncertainty prediction.</p>
<p>3. There are a few issues that were not clear to me. See questions below. I believe that fixing those issues, would make the paper publishable.</p>
<p>Questions:</p>
<p>1. The representation and CNN architecture is not clear to me. What dimension exactly is being averaged in the ESM embeddings? At the end, what is the dimension of the protein representation both for ESM and one-hot encoding? Is it constant or varying with sequence length? Why do you need a CNN to process this representation, as opposed to a fully connected MLP? Is there some local invariance or smoothness property that should be captured by convolutions? If so over what dimension?</p>
<p>2. It is not clear from the description in section 4.6 that all methods use the same number of samples in evaluation. If this is the case it should be stated, and if not, it should be discussed and justified.</p>
<p>3. Some methods were not fully described, for example it was mentioned that the evidential CNN uses a loss termed L^R, without elaborating.</p>
<p>3. Section 4.7 is confusing in listing the different metrics. First it describes MAE and R^2 metrics which I failed to see in any of the epxeriment results. Second, it states there are four uncertainty metrics but I could only count three (ro_unc, coverage, AUCE). Third, the description of the coverage states that the range is 4\\sigma/R rather than 4\\sigma, however in figure 3 these are shown on different axes (coverage vs. width/R) - I’m not sure what’s going on there.</p>
<p>Some general remarks:</p>
<p>1. The fact that the linear model and GP are performing better than the CNN suggest that maybe there is not enough training data for a deep learning method to work. Are there larger datasets that can be tested?</p>
<p>2. I’m not sure there is a way to do this better than the thorough experiments setup already presented, but in the evaluation through downstream tasks (active learning and Bayesian optimization), there is still some conflation between the quality of the prediction and the quality of the uncertainty predictions. It might be beneficial to try to compare a predictor where the uncertainty is computed exactly from ground truth data. This can serve as some kind of upper bound on the performance.</p>
<p>**********</p>
<p><bold>Have the authors made all data and (if applicable) computational code underlying the findings in their manuscript fully available?</bold></p>
<p>The <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/materials-and-software-sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data and code underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data and code should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data or code —e.g. participant privacy or use of data from a third party—those must be specified.</p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p>PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.</p>
<p>Reviewer #1: No</p>
<p>Reviewer #2: No</p>
<p><underline>Figure Files:</underline></p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <underline><ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com</ext-link></underline>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email us at <underline><email xlink:type="simple">figures@plos.org</email></underline>.</p>
<p><underline>Data Requirements:</underline></p>
<p>Please note that, as a condition of publication, PLOS' data policy requires that you make available all data used to draw the conclusions outlined in your manuscript. Data must be deposited in an appropriate repository, included within the body of the manuscript, or uploaded as supporting information. This includes all numerical values that were used to generate graphs, histograms etc.. For an example in PLOS Biology see here: <ext-link ext-link-type="uri" xlink:href="http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5" xlink:type="simple">http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5</ext-link>.</p>
<p><underline>Reproducibility:</underline></p>
<p>To enhance the reproducibility of your results, we recommend that you deposit your laboratory protocols in protocols.io, where a protocol can be assigned its own identifier (DOI) such that it can be cited independently in the future. Additionally, PLOS ONE offers an option to publish peer-reviewed clinical study protocols. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link></p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pcbi.1012639.r002">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r002</article-id>
<title-group>
<article-title>Author response to Decision Letter 0</article-title>
</title-group>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj002" link-type="rebutted-decision-letter" object-id="10.1371/journal.pcbi.1012639.r001" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">28 Jun 2024</named-content>
</p>
<supplementary-material id="pcbi.1012639.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.s002" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">protein_uq_plos_compbio-reviewer_responses.pdf</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="aggregated-review-documents" id="pcbi.1012639.r003" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r003</article-id>
<title-group>
<article-title>Decision Letter 1</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ben-Tal</surname>
<given-names>Nir</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kolodny</surname>
<given-names>Rachel</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2025</copyright-year>
<copyright-holder>Ben-Tal, Kolodny</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="article" id="rel-obj003" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">29 Aug 2024</named-content>
</p>
<p>Dear  Yang,</p>
<p>Thank you very much for submitting your manuscript "Benchmarking uncertainty quantification for protein engineering" for consideration at PLOS Computational Biology. As with all papers reviewed by the journal, your manuscript was reviewed by members of the editorial board and by several independent reviewers. The reviewers appreciated the attention to an important topic. Based on the reviews, we would like to accept this manuscript for publication, but please modify the small changes the reviewer requested. </p>
<p>Please prepare and submit your revised manuscript within 30 days. If you anticipate any delay, please let us know the expected resubmission date by replying to this email.</p>
<p>When you are ready to resubmit, please upload the following:</p>
<p>[1] A letter containing a detailed list of your responses to all review comments, and a description of the changes you have made in the manuscript. Please note while forming your response, if your article is accepted, you may have the opportunity to make the peer review history publicly available. The record will include editor decision letters (with reviews) and your responses to reviewer comments. If eligible, we will contact you to opt in or out</p>
<p>[2] Two versions of the revised manuscript: one with either highlights or tracked changes denoting where the text has been changed; the other a clean version (uploaded as the manuscript file).</p>
<p>Important additional instructions are given below your reviewer comments.</p>
<p>Thank you again for your submission to our journal. We hope that our editorial process has been constructive so far, and we welcome your feedback at any time. Please don't hesitate to contact us if you have any questions or comments.</p>
<p>Sincerely,</p>
<p>Rachel Kolodny</p>
<p>Academic Editor</p>
<p>PLOS Computational Biology</p>
<p>Nir Ben-Tal</p>
<p>Section Editor</p>
<p>PLOS Computational Biology</p>
<p>***********************</p>
<p>A link appears below if there are any accompanying review attachments. If you believe any reviews to be missing, please contact <email xlink:type="simple">ploscompbiol@plos.org</email> immediately:</p>
<p>Reviewer's Responses to Questions</p>
<p><bold>Comments to the Authors:</bold></p>
<p><bold>Please note here if the review is uploaded as an attachment.</bold></p>
<p>Reviewer #1: The author has addressed most of my comments.</p>
<p>Regarding the statistical evidence discussion, I do understand that creating new splits is its own endeavor, and this paper is not about FLIP but about uncertainty quantification. However, since the author already did the hard work of training 5 models per split with different seeds, can they also report the standard deviation for all metrics in the supplementary table. This could help the reader have some context on how different models perform relative to each other in a given context</p>
<p>**********</p>
<p><bold>Have the authors made all data and (if applicable) computational code underlying the findings in their manuscript fully available?</bold></p>
<p>The <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/materials-and-software-sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data and code underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data and code should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data or code —e.g. participant privacy or use of data from a third party—those must be specified.</p>
<p>Reviewer #1: Yes</p>
<p>**********</p>
<p>PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/ploscompbiol/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.</p>
<p>Reviewer #1: No</p>
<p>Figure Files:</p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com" xlink:type="simple">https://pacev2.apexcovantage.com</ext-link>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email us at <email xlink:type="simple">figures@plos.org</email>.</p>
<p>Data Requirements:</p>
<p>Please note that, as a condition of publication, PLOS' data policy requires that you make available all data used to draw the conclusions outlined in your manuscript. Data must be deposited in an appropriate repository, included within the body of the manuscript, or uploaded as supporting information. This includes all numerical values that were used to generate graphs, histograms etc.. For an example in PLOS Biology see here: <ext-link ext-link-type="uri" xlink:href="http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5" xlink:type="simple">http://www.plosbiology.org/article/info%3Adoi%2F10.1371%2Fjournal.pbio.1001908#s5</ext-link>.</p>
<p>Reproducibility:</p>
<p>To enhance the reproducibility of your results, we recommend that you deposit your laboratory protocols in protocols.io, where a protocol can be assigned its own identifier (DOI) such that it can be cited independently in the future. Additionally, PLOS ONE offers an option to publish peer-reviewed clinical study protocols. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link></p>
<p>References:</p>
<p>Review your reference list to ensure that it is complete and correct. If you have cited papers that have been retracted, please include the rationale for doing so in the manuscript text, or remove these references and replace them with relevant current references. Any changes to the reference list should be mentioned in the rebuttal letter that accompanies your revised manuscript.</p>
<p><italic>If you need to cite a retracted article, indicate the article’s retracted status in the References list and also include a citation and full reference for the retraction notice.</italic></p>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pcbi.1012639.r004">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r004</article-id>
<title-group>
<article-title>Author response to Decision Letter 1</article-title>
</title-group>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj004" link-type="rebutted-decision-letter" object-id="10.1371/journal.pcbi.1012639.r003" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">12 Nov 2024</named-content>
</p>
<supplementary-material id="pcbi.1012639.s003" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012639.s003" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">protein_uq_plos_compbio-reviewer_responses2.pdf</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1012639.r005" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r005</article-id>
<title-group>
<article-title>Decision Letter 2</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ben-Tal</surname>
<given-names>Nir</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kolodny</surname>
<given-names>Rachel</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2025</copyright-year>
<copyright-holder>Ben-Tal, Kolodny</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="article" id="rel-obj005" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>2</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">14 Nov 2024</named-content>
</p>
<p>Dear  Yang,</p>
<p>We are pleased to inform you that your manuscript 'Benchmarking uncertainty quantification for protein engineering' has been provisionally accepted for publication in PLOS Computational Biology.</p>
<p>Before your manuscript can be formally accepted you will need to complete some formatting changes, which you will receive in a follow up email. A member of our team will be in touch with a set of requests.</p>
<p>Please note that your manuscript will not be scheduled for publication until you have made the required changes, so a swift response is appreciated.</p>
<p>IMPORTANT: The editorial review process is now complete. PLOS will only permit corrections to spelling, formatting or significant scientific errors from this point onwards. Requests for major changes, or any which affect the scientific understanding of your work, will cause delays to the publication date of your manuscript.</p>
<p>Should you, your institution's press office or the journal office choose to press release your paper, you will automatically be opted out of early publication. We ask that you notify us now if you or your institution is planning to press release the article. All press must be co-ordinated with PLOS.</p>
<p>Thank you again for supporting Open Access publishing; we are looking forward to publishing your work in PLOS Computational Biology. </p>
<p>Best regards,</p>
<p>Rachel Kolodny</p>
<p>Academic Editor</p>
<p>PLOS Computational Biology</p>
<p>Nir Ben-Tal</p>
<p>Section Editor</p>
<p>PLOS Computational Biology</p>
<p>Feilim Mac Gabhann</p>
<p>Editor-in-Chief</p>
<p>PLOS Computational Biology</p>
<p>Jason Papin</p>
<p>Editor-in-Chief</p>
<p>PLOS Computational Biology</p>
<p>***********************************************************</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pcbi.1012639.r006" specific-use="acceptance-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012639.r006</article-id>
<title-group>
<article-title>Acceptance letter</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Ben-Tal</surname>
<given-names>Nir</given-names>
</name>
<role>Section Editor</role>
</contrib>
<contrib contrib-type="author">
<name name-style="western">
<surname>Kolodny</surname>
<given-names>Rachel</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2025</copyright-year>
<copyright-holder>Ben-Tal, Kolodny</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pcbi.1012639" document-id-type="doi" document-type="article" id="rel-obj006" link-type="peer-reviewed-article"/>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">11 Dec 2024</named-content>
</p>
<p>PCOMPBIOL-D-23-01757R2 </p>
<p>Benchmarking uncertainty quantification for protein engineering</p>
<p>Dear Dr Yang,</p>
<p>I am pleased to inform you that your manuscript has been formally accepted for publication in PLOS Computational Biology. Your manuscript is now with our production department and you will be notified of the publication date in due course.</p>
<p>The corresponding author will soon be receiving a typeset proof for review, to ensure errors have not been introduced during production. Please review the PDF proof of your manuscript carefully, as this is the last chance to correct any errors. Please note that major changes, or those which affect the scientific understanding of the work, will likely cause delays to the publication date of your manuscript. </p>
<p>Soon after your final files are uploaded, unless you have opted out, the early version of your manuscript will be published online. The date of the early version will be your article's publication date. The final article will be published to the same URL, and all versions of the paper will be accessible to readers.</p>
<p>Thank you again for supporting PLOS Computational Biology and open-access publishing. We are looking forward to publishing your work! </p>
<p>With kind regards,</p>
<p>Dorothy Lannert</p>
<p>PLOS Computational Biology | Carlyle House, Carlyle Road, Cambridge CB4 3DN | United Kingdom <email xlink:type="simple">ploscompbiol@plos.org</email> | Phone +44 (0) 1223-442824 | <ext-link ext-link-type="uri" xlink:href="http://ploscompbiol.org" xlink:type="simple">ploscompbiol.org</ext-link> | @PLOSCompBiol</p>
</body>
</sub-article>
</article>