<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id><journal-title-group>
<journal-title>PLoS Computational Biology</journal-title></journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, USA</publisher-loc></publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-14-00334</article-id>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1003757</article-id>
<article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Biology and life sciences</subject><subj-group><subject>Computational biology</subject></subj-group><subj-group><subject>Evolutionary biology</subject><subj-group><subject>Population genetics</subject></subj-group></subj-group><subj-group><subject>Genetics</subject><subj-group><subject>Mutation</subject></subj-group></subj-group><subj-group><subject>Molecular biology</subject><subj-group><subject>Molecular biology techniques</subject><subj-group><subject>Sequencing techniques</subject><subj-group><subject>Sequence analysis</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Medicine and health sciences</subject><subj-group><subject>Clinical genetics</subject><subj-group><subject>Personalized medicine</subject></subj-group></subj-group><subj-group><subject>Infectious diseases</subject><subj-group><subject>Viral diseases</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Analysis of Stop-Gain and Frameshift Variants in Human Innate Immunity Genes</article-title>
<alt-title alt-title-type="running-head">Stop-Gains and Frameshifts in Human Innate Immunity Genes</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Rausell</surname><given-names>Antonio</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="aff" rid="aff4"><sup>4</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Mohammadi</surname><given-names>Pejman</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff5"><sup>5</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>McLaren</surname><given-names>Paul J.</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="aff" rid="aff6"><sup>6</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Bartha</surname><given-names>Istvan</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="aff" rid="aff6"><sup>6</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Xenarios</surname><given-names>Ioannis</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff4"><sup>4</sup></xref><xref ref-type="aff" rid="aff7"><sup>7</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Fellay</surname><given-names>Jacques</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff6"><sup>6</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Telenti</surname><given-names>Amalio</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="corresp" rid="cor1"><sup>*</sup></xref></contrib>
</contrib-group>
<aff id="aff1"><label>1</label><addr-line>SIB Swiss Institute of Bioinformatics, Lausanne and Basel, Lausanne, Switzerland</addr-line></aff>
<aff id="aff2"><label>2</label><addr-line>Department of Laboratories, University Hospital of Lausanne, Lausanne, Switzerland</addr-line></aff>
<aff id="aff3"><label>3</label><addr-line>University of Lausanne, Lausanne, Switzerland</addr-line></aff>
<aff id="aff4"><label>4</label><addr-line>Vital-IT group, SIB Swiss Institute of Bioinformatics Lausanne, Lausanne, Switzerland</addr-line></aff>
<aff id="aff5"><label>5</label><addr-line>Computational Biology Group, ETH Zurich, Zurich, Switzerland</addr-line></aff>
<aff id="aff6"><label>6</label><addr-line>School of Life Sciences, École Polytechnique Fédérale de Lausanne, Lausanne, Switzerland</addr-line></aff>
<aff id="aff7"><label>7</label><addr-line>Swiss-Prot group, SIB Swiss Institute of Bioinformatics, Lausanne, Switzerland</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple"><name name-style="western"><surname>Quintana-Murci</surname><given-names>Lluis</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/></contrib>
</contrib-group>
<aff id="edit1"><addr-line>Institut Pasteur, France</addr-line></aff>
<author-notes>
<corresp id="cor1">* E-mail: <email xlink:type="simple">Amalio.telenti@chuv.ch</email></corresp>
<fn fn-type="conflict"><p>The authors have declared that no competing interests exist.</p></fn>
<fn fn-type="con"><p>Conceived and designed the experiments: AR AT. Performed the experiments: AR PM PJM. Analyzed the data: AR PM PJM. Contributed reagents/materials/analysis tools: IB IX JF. Wrote the paper: AR AT.</p></fn>
</author-notes>
<pub-date pub-type="collection"><month>7</month><year>2014</year></pub-date>
<pub-date pub-type="epub"><day>24</day><month>7</month><year>2014</year></pub-date>
<volume>10</volume>
<issue>7</issue>
<elocation-id>e1003757</elocation-id>
<history>
<date date-type="received"><day>22</day><month>2</month><year>2014</year></date>
<date date-type="accepted"><day>16</day><month>6</month><year>2014</year></date>
</history>
<permissions>
<copyright-year>2014</copyright-year>
<copyright-holder>Rausell et al</copyright-holder><license xlink:type="simple"><license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions>
<abstract>
<p>Loss-of-function variants in innate immunity genes are associated with Mendelian disorders in the form of primary immunodeficiencies. Recent resequencing projects report that stop-gains and frameshifts are collectively prevalent in humans and could be responsible for some of the inter-individual variability in innate immune response. Current computational approaches evaluating loss-of-function in genes carrying these variants rely on gene-level characteristics such as evolutionary conservation and functional redundancy across the genome. However, innate immunity genes represent a particular case because they are more likely to be under positive selection and duplicated. To create a ranking of severity that would be applicable to innate immunity genes we evaluated 17,764 stop-gain and 13,915 frameshift variants from the NHLBI Exome Sequencing Project and 1,000 Genomes Project. Sequence-based features such as loss of functional domains, isoform-specific truncation and nonsense-mediated decay were found to correlate with variant allele frequency and validated with gene expression data. We integrated these features in a Bayesian classification scheme and benchmarked its use in predicting pathogenic variants against Online Mendelian Inheritance in Man (OMIM) disease stop-gains and frameshifts. The classification scheme was applied in the assessment of 335 stop-gains and 236 frameshifts affecting 227 interferon-stimulated genes. The sequence-based score ranks variants in innate immunity genes according to their potential to cause disease, and complements existing gene-based pathogenicity scores. Specifically, the sequence-based score improves measurement of functional gene impairment, discriminates across different variants in a given gene and appears particularly useful for analysis of less conserved genes.</p>
</abstract>
<abstract abstract-type="summary"><title>Author Summary</title>
<p>There are well-characterized severe immunodeficiencies associated with loss-of-function variants in innate immunity genes. Genome sequencing projects identify rare stop-gain and frameshift variants in innate immunity genes whose phenotype is uncharacterized. Current methods to estimate the severity of rare stop-gains and frameshifts are based on evolutionary conservation of the gene, the likelihood for redundancy in its function or mutational burden. These parameters are not always applicable to innate immunity genes. We evaluated sequence-level characteristics of more than 30'000 stop-gains and frameshifts and prioritized variants according to their predicted functional consequences. Our scoring approach complements existing tools in the prediction of innate immunity OMIM disease variants and associates with functional readouts such as gene expression. In this framework, we show that many individuals do carry highly pathogenic variants in genes participating in antiviral defense. The clinical assessment of these variants is of significant interest.</p>
</abstract>
<funding-group><funding-statement>This work is funded by the Swiss National Science Foundation (grant #CRSII3_147665). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts><page-count count="12"/></counts></article-meta>
</front>
<body><sec id="s1">
<title>Introduction</title>
<p>There is considerable variability in the human immune response to pathogens. The observation of genetic causes of a number of primary immunodeficiencies underscores the fundamental role of variants in immune genes - in many cases resulting in severe, pathogen-specific disorders <xref ref-type="bibr" rid="pcbi.1003757-QuintanaMurci1">[1]</xref>. A main challenge in the analysis of genome variation today is the assignment of a functional role to rare variants <xref ref-type="bibr" rid="pcbi.1003757-MacArthur1">[2]</xref>. Here, large numbers of study participants would not necessarily provide the statistical power to associate a genotype with a phenotype. In this context, efforts are put toward to the computational identification of features allowing prioritization of variants for follow-up in genetic and functional analysis. Strategies to attribute a severity score to a variant, recently reviewed in <xref ref-type="bibr" rid="pcbi.1003757-Peterson1">[3]</xref>, include approaches based on evolutionary, physico-chemical and structural properties (Polyphen2 <xref ref-type="bibr" rid="pcbi.1003757-Adzhubei1">[4]</xref>, SIFT <xref ref-type="bibr" rid="pcbi.1003757-Kumar1">[5]</xref>), methods based on analysis of mutation load (e.g. the Residual Variation Intolerance Score, RVIS <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref>), and integrative pipelines <xref ref-type="bibr" rid="pcbi.1003757-Khurana1">[7]</xref>–<xref ref-type="bibr" rid="pcbi.1003757-Kircher1">[10]</xref>.</p>
<p>Of special interest in the study of inter-individual variability in innate immunity is the evaluation of stop-gains and frameshifts. Such variants are prevalent, having an estimated number of 100 to 200 occurrences per human genome <xref ref-type="bibr" rid="pcbi.1003757-Genomes1">[11]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-MacArthur2">[12]</xref>. Stop-gains and frameshifts may lead to functional consequences due to protein truncation, degradation of the transcript by Nonsense-Mediated Decay (NMD) <xref ref-type="bibr" rid="pcbi.1003757-Nagy1">[13]</xref> and dominant negative influences of protein species. In particular, rare and young variants that have not undergone purifying selection may contribute to burden of disease in a population <xref ref-type="bibr" rid="pcbi.1003757-Nelson1">[14]</xref>–<xref ref-type="bibr" rid="pcbi.1003757-Fu1">[16]</xref>. Despite a stop-gain or frameshift variant, however, the function of a protein may be preserved because of limited truncation of functional and structural domains, or because the variant affects only one of the splice forms. A less understood possibility is the occurrence of stop-codon read-through <xref ref-type="bibr" rid="pcbi.1003757-Jungreis1">[17]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Wills1">[18]</xref>.</p>
<p>Analyses based on gene characteristics such as evolutionary conservation and non-redundancy in the genome <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>, or mutational burden analysis <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref> are used to predict the severity of stop-gain and frameshift variants. Herein, we refer to these analyses as “<italic>gene-based</italic>”. However, innate immunity genes tend to be less conserved and more duplicated than the genome average <xref ref-type="bibr" rid="pcbi.1003757-Rausell1">[20]</xref> and other features may be needed to assess functional relevance of a variant. The aim of this study is to explore sequence characteristics that may improve the understanding of the functional consequences of stop-gain and frameshift variants in innate immunity genes. Herein, we will refer to these analyses as “<italic>sequence-based</italic>”. For this, we first evaluated two sets of publicly available data from a total of 7595 individuals <xref ref-type="bibr" rid="pcbi.1003757-Fu1">[16]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Genomes2">[21]</xref> including gene expression data from 421 of them <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>. Specific sequence features of truncating variants were found to correlate with allele frequency and gene expression levels. These features were used to generate a pathogenicity score that was evaluated through benchmark against OMIM disease variants. The approach was applied to assess functional consequences of stop-gain and frameshift variants in innate immunity genes, with particular attention to antiviral interferon-stimulated genes (ISGs).</p>
</sec><sec id="s2">
<title>Results</title>
<sec id="s2a">
<title>Variant set</title>
<p>We analysed gene variant data from a total of 7595 individuals from the NHLBI GO Exome Sequencing Project (ESP) <xref ref-type="bibr" rid="pcbi.1003757-Fu1">[16]</xref> and the 1000 Genomes Project <xref ref-type="bibr" rid="pcbi.1003757-Genomes2">[21]</xref>. We considered 17764 stop-gain and 13915 frameshift variants collectively affecting 11369 autosomal protein coding genes reliably annotated by the Consensus CDS (CCDS) project <xref ref-type="bibr" rid="pcbi.1003757-Pruitt1">[23]</xref>. The distributions of gene truncating variants according to allele frequency and study are presented in <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s008">Table S1</xref></bold>.</p>
</sec><sec id="s2b">
<title>Distribution of variants in sequence-based features</title>
<p>Consistent with previous reports <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Yngvadottir1">[24]</xref>, we observed that the distribution of stop-gain and frameshift variants along the protein coding sequence of genes is biased by allele frequency (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s001">Figure S1</xref></bold>). Variants with very low allele frequency (MAF≤0.001) are evenly distributed, with a modest 3′ terminal enrichment. However, the distribution of stop-gain and frameshift variants becomes less uniform with increasing allele frequencies, yet does not show a clear pattern. In contrast, we observed marked distribution trends in association with the following sequence features: (i) loss of functional domains; (ii) disruption of constitutive exons (i.e. exons present in all isoforms), or of principal isoforms; (iii) localization in potential NMD-targeted regions. In comparison to rare truncating variants, common stop-gain and frameshift variants were clearly depleted at positions leading to the loss of a functional domain (<xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1A</bold></xref>). Analysis of splicing-dependent effects was limited to genes with multiple annotated transcripts in CCDS (n = 5203). We observed an enrichment of common stop-gain and frameshift variants in alternative isoforms (<xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1B</bold></xref>) and a depletion of common variants in principal isoforms (<xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1C</bold></xref>). Defining principal isoform on the basis of highest expression level across tissues <xref ref-type="bibr" rid="pcbi.1003757-GonzalezPorta1">[25]</xref> showed comparable results. We observed that common gene truncating variants occurred less frequently in regions more than fifty nucleotides upstream the last exon-exon junction, possibly triggering NMD-mediated transcript degradation (<xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1D</bold></xref>). For all the features discussed above, gene-truncating variants associated with disease in the Online Mendelian Inheritance in Man (OMIM) database exhibited a distribution bias opposite to what was observed for common stop-gain and frameshift variants (<xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1</bold></xref>). The same trends were observed when ESP and 1000 Genomes variant datasets were analysed separately (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s002">Figure S2</xref></bold>).</p>
<fig id="pcbi-1003757-g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g001</object-id><label>Figure 1</label><caption>
<title>Distribution of variants according to sequence features and allele frequency.</title>
<p>The y-axis represents the percentage of variants for the allele frequencies and categories represented in the x-axis. <bold>Panel A</bold>, percentage of variants upstream of a functional domain. <bold>Panel B</bold>, in alternatively spliced sites. <bold>Panel C</bold>, in the principal isoform. <bold>Panel D</bold>, in regions targeted by NMD. The distribution is shown for synonymous (green), missense (blue), stop-gain (red) and frameshift (orange) variants according to minor allele frequency (MAF) intervals, where singletons (variants detected only in one individual) are represented separately. The pattern of OMIM disease variants and homozygous variants for each feature is shown. The corresponding coding genome background (measured as the percentage of nucleotides displaying the feature) is shown as a grey line (partly hidden by the distribution of synonymous variants in some panels). Numbers of variants in each category are reported in <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s009">Table S2</xref></bold>. Logistic regression was used to model the relationship between observing a given sequence feature in a given type of variant as a function of the logarithm of the minor allele frequency (MAF). The odds ratio estimates for stop-gain variants were significantly different from those of synonymous variants in all panels (p-values&lt;5e-05, heterogeneity test <xref ref-type="bibr" rid="pcbi.1003757-Randall1">[40]</xref>; for frameshifts, in panels B, C and D (p-values&lt;5e-03).</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g001" position="float" xlink:type="simple"/></fig></sec><sec id="s2c">
<title>Expression analysis</title>
<p>We used expression data from 421 individuals to assess the functional impact of stop-gain and frameshift variants <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>. In particular, we evaluated differences between protein truncating variants localized to NMD-targeted region compared to those that were not. Stop-gains predicted to trigger NMD (n = 756) had a significantly lower expression level (median Z-score = −0.59) than stop-gains predicted to escape NMD (n = 379, median = −0.10) and lower than a reference distribution of synonymous variants (median = −0.04, one-sided Wilcoxon rank-sum test p-value&lt;2.2e-16) (<xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2A</bold></xref>). Among stop-gains predicted to trigger NMD, singletons (n = 488, median Z-score = −0.75) showed a stronger decrease in expression level compared to non-singletons (n = 268, median = −0.26, p-value = 1.3e-10, <xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2B</bold></xref>), which is an indication that they represent actual variants and not sequencing or bioinformatics errors. We did not observe a similar reduction when considering 87 of the 172 frameshift variants with expression data mapping to potential NMD-target regions (median = −0.14, p-value = 5.7e-02).</p>
<fig id="pcbi-1003757-g002" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g002</object-id><label>Figure 2</label><caption>
<title>Association of NMD-target variants with gene expression.</title>
<p><bold>Panel A</bold> shows the distribution of average expression z-scores for genes from individuals carrying different types of variants (synonymous, missense, frameshift and stop-gain). Peer-factor normalized RPKM from <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref> were used. The black sector represents the distribution of variants outside the NMD-target region and the colored sector those within the NMD-target region. Statistically significant differences were observed for stop-gain variants predicted to trigger NMD (n = 756) compared to synonymous variants (one-sided Wilcoxon rank-sum test p-value&lt;2.2e-16). <bold>Panel B</bold> shows the distribution of average expression z-scores described in panel A for synonymous (grey) and stop-gain (dark and light purple) variants within the NMD-target region. The distribution of NMD-target stop-gains is represented separately for singletons (dark purple, n = 488) and non-singletons (light purple, n = 268). Distributions are statistically different (one-sided Wilcoxon rank-sum test = 1.3e-10). <bold>Panel C</bold> shows the distribution of average expression z-scores described in panel A for synonymous (grey) and stop-gain (dark and light pink) variants within the NMD-target region of genes with multiple isoforms described in CCDS. The distribution of NMD-target stop-gain is represented separately for those affecting all isoforms (dark pink, n = 216) and those affecting only a fraction of isoforms (light pink, n = 85). Distributions are statistically different (one-sided Wilcoxon rank-sum test = 2.5e-03). Results were reproduced using RPKM normalized expression values (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s003">Figure S3</xref></bold>).</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g002" position="float" xlink:type="simple"/></fig>
<p>To further evaluate a splicing-dependent impact on gene expression levels, we limited the analysis to 301 stop-gains predicted to trigger NMD and affecting genes with multiple isoforms described in CCDS. We observed a significant decrease in gene expression levels of NMD-triggering stop-gains affecting all isoforms (n = 216, median = −0.64) compared to those affecting only a fraction of isoforms (n = 85, median = −0.22, one-sided Wilcoxon rank-sum test p-value = 2.5e-03) (<xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2C</bold></xref>). Similar results were obtained using RPKM normalized expression values (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s003">Figure S3</xref></bold>). These observations confirmed the functional impact of stop-gains consistently with predictions of degradation by NMD and current annotation of isoforms.</p>
</sec><sec id="s2d">
<title>Pathogenicity scores</title>
<p>We then evaluated the predictive value for pathogenicity of the sequence-based features characterized in the previous sections: percentage of sequence affected, loss of functional domains, proportion of isoforms affected, principal isoform damage, and NMD-target region. We integrated them into a naïve Bayes classifier (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s010">Table S3</xref></bold>) and assessed its performance over a dataset of 1160 pathogenic stop-gain variants found in the OMIM database and 125 common stop-gain variants that are not known to be pathogenic. Predictive performance of the pathogenicity score was validated over unseen variants excluded from the learning data using successive random subsampling (see <xref ref-type="sec" rid="s4">Methods</xref>). The classifier was benchmarked against a state of the art gene-based probability score proposed by MacArthur et al <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>. This gene-based score relies on conservation and protein interaction network proximity to genes associated to a recessive disease as predictive features. In the case of stop-gain variants, the performance of the gene-based method was consistent with the reported results in the original work (Area Under the Curve (AUC) = 0.83, <xref ref-type="fig" rid="pcbi-1003757-g003"><bold>Figure 3A</bold></xref>). Similar ROC curves were obtained with the gene-based score RVIS <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref> that provides a measure of the departure from the average number of common functional mutations in genes with a similar amount of mutational burden (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s004">Figure S4</xref></bold>). The score based on sequence features alone showed a lower predictive value (AUC = 0.67). However, optimal ROCs were achieved by combining sequence and gene-based scores (<xref ref-type="fig" rid="pcbi-1003757-g003"><bold>Figure 3</bold></xref>). We observe that at a False Positive Rate (FPR) of less than 0.1 there is no improvement from the combined sequence-based and from the MacArthur gene-based score that used network proximity OMIM recessive disease genes in its design. Improvement at low FPR occurs in the combination of the sequence-based score with RVIS, which does not rely on OMIM annotations. While the AUC improvement is modest, it is consistent across two datasets (ESP and 1000 Genomes), over the two gene-based scores, and for the two types of variants (stop-gains and frameshifts), <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s004">Figure S4</xref></bold>. These results demonstrate that sequence features can be incorporated as an additional source of information to improve current pathogenicity prediction.</p>
<fig id="pcbi-1003757-g003" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g003</object-id><label>Figure 3</label><caption>
<title>Receiver operating characteristic of the performance of pathogenicity scores for stop-gain variants.</title>
<p><bold>Panel A</bold>: Classification power of three pathogenicity scores was evaluated on a set of 1160 pathogenic stop-gain variants in the OMIM database, and 125 common stop-gain variants not known to be pathogenic. Shown are the ROC curves for the sequence-based classifier (SB) developed in this work, for the gene-based score reported in <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref> (GB), and for the joint classifier (SB×GB). Dashed curves correspond to a randomization test in which rows in sequence features are shuffled column-wise (denoted by SB<sup>(r)</sup>). <bold>Panel B</bold>: AUC improvement achieved when combining the sequence-based scores with a gene-based score. The panels shows AUC values of ROC curves using two independent gene-based scores (MacArthur 2012 <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref> and RVIS <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref>), on two independent datasets of variants (ESP and 1000 Genomes) and two types of variants: stop-gains and frameshifts. Corresponding ROC curves and number of pathogenic and common variants used for benchmark is shown in <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s004">Figure S4</xref></bold>. Inclusion of sequence features led to an increased area under the ROC curve in all evaluated settings.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g003" position="float" xlink:type="simple"/></fig></sec><sec id="s2e">
<title>Correlation and complementarity of sequence-based and gene-based scores</title>
<p>The marginal improvement obtained when scores were combined motivated us to explore whether the different approaches were capturing independent information. We observed a very low correlation between gene-based and sequence-based scores (Spearman rank correlation &lt;0.13, p-value: &lt;2.2e-16 ([0.10,0.13] 95% CI from 10,000 bootstrap samples). The reason for this observation is that the various scores are based on different criteria: gene conservation and centrality (MacArthur 2012), burden of variation (RVIS) and sequence features (current work). Correlations were not increased in analyses limited to OMIM disease variants (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s005">Figure S5</xref></bold>). Based on these results we explored the potential for complementarity across scores.</p>
<p>First, we analysed whether the sequence-based score was better powered to detect functional impact as measured by effect on gene expression. We observed a stronger correlation with expression levels for the sequence-based score (Spearman rank correlation = 0.21±0.03, p-value: &lt;5e-12) than either gene-based scores (0.06±0.04, p-value&gt;0.05 for MacArthur 2012 score and 0.13±0.03, p-value&lt;5e-05, for RVIS score), <xref ref-type="fig" rid="pcbi-1003757-g004"><bold>Figure 4</bold></xref>. Second, we analysed OMIM genes that carry variants annotated as pathogenic in OMIM as well as unknown or non-pathogenic variants. Here, the variants are scored differently using a sequence-based approach, while all share the same gene-based score. <xref ref-type="fig" rid="pcbi-1003757-g005"><bold>Figure 5</bold></xref> depicts this situation for 95 OMIM disease genes carrying multiple stop-gains. The genes with the highest pathogenicity gene-based scores also carried variants with very low severity as determined by a sequence-based score. Third, we checked whether the performance of the sequence-based score varies depending on the degree of gene conservation, as measured by dN/dS ratio in the same set of OMIM disease genes. <xref ref-type="fig" rid="pcbi-1003757-g006"><bold>Figure 6</bold></xref> shows that, for genes below the protein-coding genome average dN/dS (0.261), the MacArthur and RVIS gene-based scores resulted in higher pathogenicity estimates than the sequence-based score; however without discriminating between pathogenic and non-pathogenic/non-annotated variants. In contrast, for genes with dN/dS≥0.261, the sequence-based score performed similarly for pathogenic variants while attributing less pathogenicity to non-pathogenic/non-annotated variants of the same gene (Wilcoxon signed rank test p-value&lt;0.012). We note that OMIM variants used here were not considered for learning in the Bayesian classification (see <xref ref-type="sec" rid="s4">Methods</xref>).</p>
<fig id="pcbi-1003757-g004" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g004</object-id><label>Figure 4</label><caption>
<title>Correlation between pathogenicity scores of truncating variants and impact in gene-expression levels.</title>
<p>Shown are the distributions (y-axis) of three pathogenicity scores (<bold>Panel A</bold>: the sequence-based score developed in this work, <bold>Panel B</bold>: the gene-based score from MacArthur 2012 <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>; <bold>Panel C</bold>: the gene-based score RVIS <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref>) within quintile bins (x-axis) of the average expression z-scores from individuals carrying stop-gain variants (Peer-factor normalized RPKM from <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref> were used; see <xref ref-type="sec" rid="s4">Methods</xref> and <xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2</bold></xref>). A total of 1060 stop-gain variants are represented, 212 in each quintile. Quintiles from 1 to 5 are ordered in decreasing impact on gene expression levels and correspond to the following intervals respectively: z-score&lt;−1.25, (−1.25, −0.66], (−0.66, −0.23], (−0.23, 0.23], (0.23, 5.15]. To allow comparison across scores, they are represented as rank percentiles, where the value of a given variant accounts for the percentage of all stop-variants that had a score more pathogenic than the variant. Therefore, a rank percentile of “0” indicates a variant with the highest predicted probability of being pathogenic while a rank percentile of “100” indicates a variant with the lowest predicted severity. A stronger correlation with expression levels was observed for the sequence-based score (Spearman rank correlation = 0.21±0.03, p-value: &lt;5e-12) than either gene-based scores (0.06±0.04, p-value&gt;0.05 for MacArthur 2012 score and 0.13±0.03, p-value&lt;5e-05, for RVIS score). None of the scores associated frameshift variants with gene expression levels.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g004" position="float" xlink:type="simple"/></fig><fig id="pcbi-1003757-g005" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g005</object-id><label>Figure 5</label><caption>
<title>Complementarity between sequence-based and gene-based pathogenicity scores illustrated for OMIM genes with both pathogenic and non-pathogenic/non-annotated stop-gain variants.</title>
<p>Shown are the sequence-based score (x-axis) for 273 stop-gain variants reported by the ESP and 1000 Genomes datasets in 75 OMIM genes carrying both OMIM pathogenic-variants (grey dots) and a non-pathogenic/non-annotated variants (orange dots). Genes are displayed by blocks from 1 to 9 (y-axis on the right) corresponding to deciles of the gene-based MacArthur 2012 rank percentile (e.g. 1: &lt; = 10; 2: (10,20], etc). Grey triangles beside the panels represent the direction of increasing pathogenicity for the corresponding scores.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g005" position="float" xlink:type="simple"/></fig><fig id="pcbi-1003757-g006" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g006</object-id><label>Figure 6</label><caption>
<title>Discrimination of pathogenic and non-pathogenic variants within OMIM genes according to the degree of gene conservation.</title>
<p>Shown are boxplots representing the distribution of the average sequence-based score of pathogenic (dark grey) and non-pathogenic/non-annotated (orange) stop-gain variants in OMIM genes depicted in <xref ref-type="fig" rid="pcbi-1003757-g005"><bold>Figure 5</bold></xref>. The distributions of the corresponding MacArthur 2012 and RVIS gene-based scores are shown in light grey. Genes are represented in two categories according to their conservation level in primates: dN/dS ratio below (<bold>Panel A</bold>; n = 54) and above (<bold>Panel B</bold>; n = 20) the protein-coding genome average.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g006" position="float" xlink:type="simple"/></fig>
<p>From these results, we conclude that the two types of scores are complementary. Specifically, the sequence-based score improves measurement of functional gene impairment, discriminates across different variants in a given gene and appears particularly useful for analysis of less conserved genes.</p>
</sec><sec id="s2f">
<title>Analysis of innate immunity genes</title>
<p>To test the ability to rank the functional consequences of gene truncating variants in innate immunity genes, we analysed the distribution of both the sequenced-based and gene-based pathogenicity scores in 1503 genes involved in innate immunity <xref ref-type="bibr" rid="pcbi.1003757-Rausell1">[20]</xref>, including 387 interferon stimulated genes (ISGs, <xref ref-type="bibr" rid="pcbi.1003757-Schoggins1">[26]</xref> <xref ref-type="bibr" rid="pcbi.1003757-Schoggins2">[27]</xref>). We identified 856 innate immunity genes, including 230 ISGs, carrying rare gene truncating variants (MAF&lt;1%). Globally, innate immunity and OMIM genes ranked higher than the background set of the genome for both scores (<xref ref-type="fig" rid="pcbi-1003757-g007"><bold>Figure 7</bold></xref><bold> and <xref ref-type="supplementary-material" rid="pcbi.1003757.s006">Figure S6</xref></bold>). However, the highest scores were obtained for stop-gain variants in OMIM genes, particularly for variants in innate immunity genes that are not observed in the ESP or the 1000 Genomes Project samples (<xref ref-type="fig" rid="pcbi-1003757-g007"><bold>Figure 7</bold></xref>). The latter result is consistent with their extreme rarity and severity. We note that OMIM variants used here for validation were not considered for learning in the Bayesian classification (see <xref ref-type="sec" rid="s4">Methods</xref>). Despite their apparent agreement in <xref ref-type="fig" rid="pcbi-1003757-g007"><bold>Figure 7</bold></xref>, correlation between the sequenced-based and gene-based pathogenicity scores was very low (Spearman correlation below 0.31 in all sets of genes analyzed) indicating that both scores provide complementary information.</p>
<fig id="pcbi-1003757-g007" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1003757.g007</object-id><label>Figure 7</label><caption>
<title>Pathogenicity score distributions for rare stop-gain variants in innate immunity genes.</title>
<p>Rank percentile distributions of pathogenicity scores for rare stop-gain variants (MAF&lt;1%) are shown in different sets of genes: protein coding genome background (grey, “Genome”), innate immunity genes (light turquoise, “Inn Imm”) and their subset of interferon stimulated genes (dark turquoise, “ISGs”). The same categories are shown for OMIM disease variants. All variants are reported in ESP and 1000 Genomes datasets except for sets indicated with the § symbol (dashed boxes) which present scores for OMIM disease variants only reported in the OMIM database. Only three variants reported in ESP and 1000 Genomes were found to affect ISGs and annotated as pathogenic in OMIM; this category is not represented in the figure. Variants with the highest probability of being pathogenic have rank percentiles closer to zero (top of the panels). <bold>Panel A</bold> represents precomputed gene-based pathogenicity scores from <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>. <bold>Panel B</bold> represents sequence-based pathogenicity scores, i.e. posterior probabilities using the features described in the present work (see main text). Distributions of rank percentiles are represented as boxes where each box spans between 1st and 3rd quantile, and the median is denoted by a bold line in the middle. Total number of variants within each distribution is indicated. Differences in number of variants in equivalent categories between panel A and B originate from unavailability of the gene-based scores for some genes. Statistical differences against the genome reference (one-sided Wilcoxon rank sum tests) are indicated with asterisks according to Bonferroni corrected p-values: &lt;5e-02 (*), &lt;5e-03 (**) and &lt;5e-04 (***). The genome-wide median is denoted by a red line. Spearman correlation between the sequenced-based and gene-based pathogenicity scores was below 0.31 in all sets of genes analyzed (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s005">Figure S5</xref></bold>).</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1003757.g007" position="float" xlink:type="simple"/></fig>
<p>Given the observation that truncating variants can be associated with important differences in functional impact, we estimated the number of individuals in the study population that carried variants consistently annotated as highly pathogenic by one or several scores. Among 7595 individuals and 1503 innate immunity genes, 33 individuals carried rare (MAF&lt;0.01) stop-gains and 85 carried rare frameshifts that scored with high severity (pathogenicity rank percentile &lt; = 20%) in all scores (sequence-based, MacArthur 2012 and RVIS). For the smaller set of 387 ISGs, we identified 8 individuals carrying rare stop-gains and 4 carrying rare frameshifts with high severity in all scores.</p>
<p>We then focused on truncating variants in genes associated with viral inhibition in cellular assays <xref ref-type="bibr" rid="pcbi.1003757-Schoggins1">[26]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Schoggins2">[27]</xref>. A total of 13 out of 42 genes carried such variants (observed in at least 2 people), which were very rare overall (MAF&lt;0.0053; <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s011">Table S4</xref></bold>). Specifically, two genes had variants with high predicted pathogenicity based on both scores: <italic>MX1</italic> which controls Influenza A virus <italic>in vitro</italic> and <italic>HPSE</italic> which is involved in metapneumovirus, respiratory syncytial virus and yellow fever virus control. While the gene-based scores were by definition identical for all variants affecting a same gene, the sequenced-based score sharply distinguished the variants according to different predictive pathogenicity (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s010">Table S3</xref></bold>). This observation was consistent with the observed differences in gene expression levels available for some of the variants.</p>
</sec></sec><sec id="s3">
<title>Discussion</title>
<p>Numerous Mendelian disorders leading to severe infection are caused by rare functional variation of innate immunity genes <xref ref-type="bibr" rid="pcbi.1003757-QuintanaMurci1">[1]</xref>. Here, we identified multiple stop-gain and frameshift variants in this family of genes in the general population, especially among interferon stimulated genes. These are generally heterozygous rare variants that may or may not result in clinical consequences. To understand the nature and possible consequences of these variants, we first analyzed their characteristics at the genome level. The genome-wide analysis of more than 30'000 variants provided the statistical power to identify sequence specific features for severity and to build a pathogenicity score. This sequence-based pathogenicity score was then applied to the analysis of variants in interferon stimulated genes with antiviral activity.</p>
<p>We observed that the distribution of stop-gain and frameshift variants in the sequence is biased by the allele frequency. Thus, we speculated that tolerance to these variants would reflect their impact on functional domains, on isoforms, and on degradation by NMD. Our results clearly underscore that rare stop-gain and frameshift variants are subject to purifying selection <xref ref-type="bibr" rid="pcbi.1003757-Tennessen1">[15]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Montgomery1">[28]</xref>. Indeed, those variants are kept at very low frequency when they result in the loss of functional domains, when they are located in NMD-targeted regions, or when they disrupt the principal isoform or constitutively spliced exons. The potential molecular impact of heterozygous rare truncating variants was examined using mRNA expression data <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>. Stop-gain variants predicted to trigger NMD degradation resulted in a measurable decrease in global expression levels. This is in line with recent findings showing a reduction in expression levels of the variant allele compared to the reference allele in heterozygous individuals when stop-gains occur in NMD target regions <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Montgomery2">[29]</xref>. In all analyses, singleton variants associated with highest functional impact, consistent with higher severity of lower frequency rare variants and indicative of general accuracy in variant calling. A possible limitation to our analysis is that we use lymphoblastoid cell line expression data <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>; the impact of specific variants may be allele and tissue-specific <xref ref-type="bibr" rid="pcbi.1003757-Kukurba1">[30]</xref>.</p>
<p>To further explore the functional consequences of gene truncating variants, we analysed the collective contribution of various severity features to the prediction of pathogenicity. For this we built a model on a learning set that was validated through benchmark against OMIM disease variants. These sequence-based features improved the ranking of OMIM variants when added to a predictive model that use gene-based features. Specifically, the sequence-based score appeared particularly suited for functional prediction (gene expression) and for the analysis of variants in less conserved genes. We provide a web-based tool (<ext-link ext-link-type="uri" xlink:href="http://nutvar.labtelenti.org/" xlink:type="simple">http://nutvar.labtelenti.org/</ext-link>) allowing the analysis of user-provided variants.</p>
<p>We hypothesized that such a sequence-based approach would be of particular interest for the study of innate immunity genes because, as a group, these genes tend to be less conserved than the genome average and hence need special consideration. The analysis showed that our sequence-based score is able to rank variants in innate immunity genes according to their pathogenicity and provides complementary information to previously proposed gene-based scores. Indeed we found that in the case of the antiviral genes <italic>MX1</italic> and <italic>HPSE</italic>, truncating variants ranked very highly in pathogenicity on the basis of gene-based scores while important differences were observed at sequence level suggesting significant differences in functional impact. For example the <italic>MX1</italic> stop-gain rs35132725 exhibits all the features of severity and a negative effect on expression levels. In contrast, the <italic>MX1</italic> frameshift rs199916659 is not expected to alter protein function.</p>
<p>Overall, among 387 ISGs examined in 7595 individuals, more than half of the genes carried a stop-gain or frameshift variant in 1 or more individuals, usually at low allele frequency. Of these, 12 individuals carried truncating variants consistently interpreted as highly pathogenic by the three evaluated scores. This rate of 1.5 per 1000 carriers could be a genomic substrate of occasional homozygosity with unknown phenotypic consequences.</p>
<p>We then evaluated those instances that concerned genes for which an antiviral effect has been established through a gain-of-function screen <italic>in vitro</italic>. This last analysis provided a short list of genes and reliable variants that could modulate responses to various viruses, including common human pathogens such as influenza. Of note, the <italic>in vitro</italic> virological inhibition data represents a technical readout, and there are a number of considerations that may diminish the <italic>in vivo</italic> consequences of these rare variants, including issues of redundancy and robustness in innate immunity networks, and the possibility of stop codon read-through. There are other limitations to the predictions based on sequence features, particularly the incomplete understanding of the functional role of alternative isoforms and their tissue specificity.</p>
<p>Rare gene truncating variants predicted to have high pathogenicity risk in innate immunity genes should be examined for phenotypic consequences in the population. Exceptional homozygous individuals may be at risk for severe infection while heterozygous individuals could have adequate compensation or subtler phenotypes. However, there is increasing awareness of the relevance of haploinsufficiency <xref ref-type="bibr" rid="pcbi.1003757-White1">[31]</xref>, and thus, it is not excluded that heterozygosity may be associated with apparent clinical phenotypes. Thus, the next step should include assessment <italic>in vivo</italic> of high risk variants, which requires the capacity to re-contact carrier individuals for collection of biological specimens and in-depth phenotypic assessment.</p>
</sec><sec id="s4" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="s4a">
<title>Human variation sets</title>
<p>Two genetic variant and annotation datasets were used: 1) 6503 individuals from the NHLBI GO Exome Sequencing Project (ESP) <xref ref-type="bibr" rid="pcbi.1003757-Fu1">[16]</xref> and 2) 1092 individuals from the 1000 Genomes Project <xref ref-type="bibr" rid="pcbi.1003757-Genomes2">[21]</xref>. Variants (SNPs and INDELs) and annotations for the ESP exomes (file ESP6500SI-V2-SSA137.dbSNP138-rsIDs.snps_indels.txt.tar.gz) were downloaded from the Exome Variant Server, NHLBI GO Exome Sequencing Project, Seattle, WA (<ext-link ext-link-type="uri" xlink:href="http://evs.gs.washington.edu/EVS/" xlink:type="simple">http://evs.gs.washington.edu/EVS/</ext-link>, accessed July 2013). Only variants assigned to the following categories were considered for further analysis: “stop-gained” (including “stop-gained-near-splice”), “frameshift”, “coding-synonymous” (including “coding-synonymous-near-splice”) and “missense” (including “missense-near-splice”). One base was added to the genomic coordinates reported for frameshifts in the ESP dataset to consider the actual location of the insertion/deletion event (<ext-link ext-link-type="uri" xlink:href="http://evs.gs.washington.edu/EVS/HelpDescriptions.jsp?tab=tabs-1" xlink:type="simple">http://evs.gs.washington.edu/EVS/HelpDescriptions.jsp?tab=tabs-1</ext-link>). Variants and genotypes from the 1000 Genome Project <xref ref-type="bibr" rid="pcbi.1003757-Genomes2">[21]</xref> correspond to phase 1 version 3 of the 20110521 release (<ext-link ext-link-type="uri" xlink:href="ftp://ftp.1000genomes.ebi.ac.uk/vol1/ftp/release/20110521/" xlink:type="simple">ftp://ftp.1000genomes.ebi.ac.uk/vol1/ftp/release/20110521/</ext-link>, accessed August 2013). SnpEff Variant Analysis software <xref ref-type="bibr" rid="pcbi.1003757-Cingolani1">[32]</xref> (version 3.3h build 2013-08-11) was used to annotated 1000Genome variants against SnpEff's pre-built human database (GRCh37.71). SnpEff categories labeled with errors or warnings in the EFF field were disregarded. Only variants assigned to the following categories were considered for further analysis: “stop_gained”, “frame_shift”, “synonymous_coding” (including “synonymous_start” and “synonymous_stop”) and “non_synonymous_coding” (missense). Hardy-Weinberg equilibrium (HWE) was tested with R package GWASExactHW (<ext-link ext-link-type="uri" xlink:href="http://cran.r-project.org/web/packages/GWASExactHW/" xlink:type="simple">http://cran.r-project.org/web/packages/GWASExactHW/</ext-link>, version 1.1). A fraction of variants significantly deviated from HWE (Fisher's exact p-values&lt;0.05), mainly due to an excess of homozygous rare allele calls, likely indicating technical artifacts. All variants not in HWE were filtered out. When both datasets where considered together, the following criteria were adopted: i) Genomic coordinates of frameshift variants reported by both datasets were treated as reported for the ESP dataset. ii) Allele frequencies and HWE of variants present in both datasets were derived from the sum of individuals from both studies; allele frequencies of variants present in only one dataset were taken as originally reported by the corresponding dataset. To exclude bias due to previous assumptions, results were reproduced for the two datasets considered separately as well as combined. For the combined analysis allele frequencies of variants present in only one dataset are estimated over all 7593 individuals.</p>
</sec><sec id="s4b">
<title>Annotation of variants in reference human transcript and protein sequences</title>
<p>The analysis pipeline implemented to annotate genetic variants is depicted in <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s007">Figure S7</xref></bold>. We restricted the analysis to protein coding genes and transcripts annotated by the Consensus CDS (CCDS) project <xref ref-type="bibr" rid="pcbi.1003757-Pruitt1">[23]</xref> (<ext-link ext-link-type="uri" xlink:href="http://ftp.ncbi.nlm.nih.gov/pub/CCDS/" xlink:type="simple">ftp.ncbi.nlm.nih.gov/pub/CCDS/</ext-link>, Release 12 04/30/2013). We considered only variants affecting a core set of human protein coding regions consistently annotated and of high quality. Only genes on the 22 autosomes were retained and only CCDS entries with a public status and an identical match were kept.</p>
<p>Domains of human protein sequences were retrieved from the InterPro database <xref ref-type="bibr" rid="pcbi.1003757-Hunter1">[33]</xref> (release 44.0, 23/09/2013). Data were downloaded through BioMart Central Portal <xref ref-type="bibr" rid="pcbi.1003757-Guberman1">[34]</xref> (<ext-link ext-link-type="uri" xlink:href="http://central.biomart.org/" xlink:type="simple">http://central.biomart.org/</ext-link>, accessed 04/10/2013), filtering fragments and considered domain boundaries corresponding to InterPro “supermatches”. Mapping from InterPro coordinates on UniProt protein sequences to CCDS sequences was done by exact matching of the complete amino acid sequences using UniProt database (release 2013_07; <xref ref-type="bibr" rid="pcbi.1003757-UniProt1">[35]</xref>).</p>
<p>A position within a protein coding gene was considered alternatively spliced if it was shared by only a fraction of all protein coding transcripts reported by the Consensus CCDS Project for that gene. Otherwise it was considered constitutively spliced for the purpose of the study. Annotation of principal isoforms used APPRIS (<xref ref-type="bibr" rid="pcbi.1003757-Rodriguez1">[36]</xref>; file APPRIS-g15.v3.15Jul2013/appris_data.principal.homo_sapiens.tsv accessed 03/09/2013 at URL: <ext-link ext-link-type="uri" xlink:href="http://appris.bioinfo.cnio.es" xlink:type="simple">http://appris.bioinfo.cnio.es</ext-link>), a computational pipeline and database for annotations of human splice isoforms. APPRIS selects a specific transcript as principal isoform, i.e. the one computationally predicted as responsible of the main cellular function, being expressed in most of the tissues or developmental stages and more evolutionary conserved. Selection of the principal isoform is based on protein structure, function and interspecies conservation of transcripts. As an alternative definition of principal isoform, we identified the transcript with a recurrent highest expression level across tissues as provided by <xref ref-type="bibr" rid="pcbi.1003757-GonzalezPorta1">[25]</xref>.</p>
<p>We accounted for nonsense-mediated decay (NMD) following HAVANA annotation guidelines v.20 (05/04/2012) (<ext-link ext-link-type="uri" xlink:href="http://www.sanger.ac.uk/research/projects/vertebrategenome/havana/assets/guidelines.pdf" xlink:type="simple">http://www.sanger.ac.uk/research/projects/vertebrategenome/havana/assets/guidelines.pdf</ext-link>), Specifically, the NMD-target region of a transcript was defined as those positions more than 50 nucleotides upstream the 3′-most exon-exon junction. Transcripts bearing stop-gain variants at these regions are predicted to be degraded by NMD <xref ref-type="bibr" rid="pcbi.1003757-Nagy1">[13]</xref>.</p>
</sec><sec id="s4c">
<title>Functional validation using mRNA expression data</title>
<p>Geuvadis RNA sequencing data from 421 lymphoblastoid cell lines from the 1000 Genomes Project (phase 1 version 3 of the 20110521 release, see above; <xref ref-type="bibr" rid="pcbi.1003757-Genomes2">[21]</xref>) were obtained from Lappalainen et al. 2013 <xref ref-type="bibr" rid="pcbi.1003757-Lappalainen1">[22]</xref>. Gene expression quantifications of protein-coding genes were downloaded from EBI ArrayExpress accession E-GEUV-1 (accessed 05/11/2013). Analyses were independently performed on both RPKM and Peer-factor normalized RPKM values. As a measure of the impact of a variant on expression level, we calculated the average Z-score of the expression level in cells from individuals carrying the variant compared to all samples.</p>
</sec><sec id="s4d">
<title>Derivation of sequence-based pathogenicity score</title>
<p>We used a naïve Bayes classification scheme in order to derive a probability of pathogenicity for a given variant using the following sequence-based features: maximum transcript length affected, maximum percentage of domain truncation, number of isoforms and ratio of isoforms affected, truncation of the principal isoform and localization in an NMD-target region. Solely for the purpose of the classifier, missing values were imputed to zero for percentage of domain truncation and to the longest isoform for principal isoform annotation. We defined a matrix <italic>X</italic><sub>N×K</sub> of <italic>K</italic> sequence-based features for <italic>N</italic> variants of a given type in the dataset and a binary vector <italic>c</italic><sub>N×1</sub> annotating variants as benign or pathogenic. A new variant, <italic>y</italic><sub>1×K</sub>, is evaluated using maximum likelihood estimates for class-specific means from the annotated data, and a common intra-class variance vector (except for binary features). We estimate the variance vector as <italic>ν</italic> = <italic>E</italic>[(<italic>x<sub>i</sub></italic> - μ<sub>ci</sub>)<sup>2</sup>], where <italic>x<sub>i</sub></italic> is the <italic>i<sup>th</sup></italic> row in matrix <italic>X</italic> and μ<sub>ci</sub> is the mean vector corresponding to the class indicated by <italic>c<sub>i</sub></italic>,. We assigned a pathogenic class with 1 and benign class with 0. Assuming a prior probability of pathogenicity, <italic>p<sub>1</sub></italic>, posterior probability of pathogenicity can be evaluated as:<disp-formula id="pcbi.1003757.e001"><graphic position="anchor" xlink:href="info:doi/10.1371/journal.pcbi.1003757.e001" xlink:type="simple"/></disp-formula>where <italic>p<sub>0</sub> = 1-p<sub>1</sub></italic> is the prior probability of being benign and <italic>θ = {μ<sub>1</sub>,μ<sub>2</sub>,ν}</italic> is the set of model parameter vectors. The conditional likelihood of <italic>y</italic> for a given class is assumed to factorize as product of <italic>K</italic> likelihoods corresponding to the <italic>K</italic> sequence features available (naïve Bayes assumption). We used normal, and Bernoulli likelihood functions to model continuous and binary features respectively. It is straightforward to show the ranking produced from this posterior probability does not depend on the prior probability <italic>p<sub>1</sub></italic> as long as it is larger than zero and it is equal for all the mutations under consideration.</p>
</sec><sec id="s4e">
<title>Evaluation of pathogenicity scores</title>
<p>As reference throughout the work, and as a learning set for the predictive scores (ROC analyses), we used a catalogue of pathogenic mutations from the Online Mendelian Inheritance in Man <xref ref-type="bibr" rid="pcbi.1003757-OMIM1">[37]</xref> database. Only genes with a cytogenetic location (genemap2.txt accessed 18/10/2013 at OMIM: <ext-link ext-link-type="uri" xlink:href="http://ftp.omim.org" xlink:type="simple">ftp.omim.org</ext-link>) and with a gene status of confirmed or provisional were kept. For each gene with an associated OMIM number, all allelic variants with a “live” status and a dbSNP identifier were obtained through the OMIM API server (<ext-link ext-link-type="uri" xlink:href="http://api.omim.org/" xlink:type="simple">http://api.omim.org/</ext-link>). We used Ensembl Variation <xref ref-type="bibr" rid="pcbi.1003757-Flicek1">[38]</xref> (Ensembl release 71, April 2013, dataset Homo sapiens Short Variation, SNPs and indels, GRCh37.p10, accessed 25/10/2013 at <ext-link ext-link-type="uri" xlink:href="http://apr2013.archive.ensembl.org/biomart/martview/" xlink:type="simple">http://apr2013.archive.ensembl.org/biomart/martview/</ext-link>) to obtain the genomic coordinates for each dbSNP identifier together with the clinical significance of each specific allele as reported by ClinVar and dbSNP following OMIM guidelines (<ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov/clinvar/docs/clinsig/" xlink:type="simple">http://www.ncbi.nlm.nih.gov/clinvar/docs/clinsig/</ext-link>). Only variants with a dbSNP identifier annotatted as “pathogenic” and mapping to a unique genomic location were kept for further analysis. SnpEff Variant Analysis was then used to re-annotate the selected pathogenic variants as described above.</p>
<p>We benchmarked three different pathogenicity scores using all stop-gain variants from the OMIM dataset as positive (pathogenic), and all common variants (MAF≥1%) not present in OMIM dataset as negative (benign) variants. The sequence-based score is the posterior probability calculated from the naïve Bayes classification scheme described in previous section using an empirical prior for pathogenicity. We used two different gene-based scores: first the probability provided by MacArthur et al <xref ref-type="bibr" rid="pcbi.1003757-MacArthur3">[19]</xref> for prioritization of variants derived from two gene-level features: conservation and protein interaction network proximity to genes associated with a recessive disease. And second the Residual Variation Intolerance Score (RVIS) <xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref> that provides a measure of the departure from the average number of common functional mutations in genes with a similar amount of mutational burden. RVIS pathogenic score was assessed as f(-RVIS), where f(.) is the logistic function. The joint score was defined as the product of the sequence-based score and one of the two previously defined gene-based scores. This joint score can be interpreted as the joint probability of a pathogenic mutation in a gene assuming conditional independence of the two probability scores. The receiver operating characteristic (ROC) curve was derived using random subsampling validation iterations. In each iteration, we use 75% of the data to train the classifier and use the remaining 25% for validation. This was done 10000 times, to minimize the Monte Carlo error, and validation set scores were combined to calculate the ROC curve. The same procedure was applied to frameshift variants.</p>
<p>Except for ROC analyses, sequence-based scores used throughout the work were derived from the learning set described above excluding both i) OMIM pathogenic variants reported in ESP and 1000 Genomes Project and ii) OMIM pathogenic variants affecting innate immunity genes and interferon stimulated genes.</p>
</sec><sec id="s4f">
<title>Assessment of dN/dS values</title>
<p>Genome-wide codon alignments of orthologous genes for nine primate species (human, chimpanzee, gorilla, orangutan, macaque, marmoset, tarsier, bushbaby, and mouse lemur) were collected from Ensembl v57. We assessed dN/dS estimates using both Ensembl Compara's protein-based alignments, and DNA-based alignments of primate sequences generated from genomic DNA alignments. Sitewise Likelihood Ratio test <xref ref-type="bibr" rid="pcbi.1003757-Massingham1">[39]</xref> was used to calculate the overall dN/dS for a given gene based on a one-ratio model where all sites have the same dN/dS value.</p>
</sec><sec id="s4g">
<title>Analysis of innate immunity genes and interferon stimulated genes (ISGs) with antiviral activity</title>
<p>A representative list of 1503 human innate immunity genes <xref ref-type="bibr" rid="pcbi.1003757-Rausell1">[20]</xref> was used. Within this list, we further analyzed 387 interferon stimulated genes (ISGs) <xref ref-type="bibr" rid="pcbi.1003757-Schoggins1">[26]</xref>. Additionally, we focused on those ISGs showing antiviral activity against 18 viruses (including important human pathogens such as HIV-1, hepatitis C virus, influenza virus and other respiratory viruses) upon overexpression in <italic>in vitro</italic> cellular assays <xref ref-type="bibr" rid="pcbi.1003757-Schoggins1">[26]</xref>, <xref ref-type="bibr" rid="pcbi.1003757-Schoggins2">[27]</xref>. We first identified all ISGs carrying gene truncating variants, and then characterized the subset of those genes associated with more than 50% viral inhibition in the cellular assays.</p>
</sec></sec><sec id="s5">
<title>Supporting Information</title>
<supplementary-material id="pcbi.1003757.s001" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s001" position="float" xlink:type="simple"><label>Figure S1</label><caption>
<p><bold>Distribution of variants along the gene sequence.</bold> The distribution is shown for synonymous (green), missense (blue), stop-gain (red) and frameshift (orange) variants binned by minor allele frequency (MAF) intervals: Singletons (panel A), MAF&lt;0.001 (<bold>Panel B</bold>), MAF <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pcbi.1003757.e002" xlink:type="simple"/></inline-formula> [0.001–0.01) (<bold>Panel C</bold>), MAF <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pcbi.1003757.e003" xlink:type="simple"/></inline-formula> [0.01–0.05) (<bold>Panel D</bold>) and MAF&gt;0.05 (<bold>Panel E</bold>). Numbers of variants in each category are reported in <bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s008">Table S1</xref></bold>. Data were combined across the sequence using intervals of 10%. The longest transcript for each gene was used as the reference sequence length.</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s002" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s002" position="float" xlink:type="simple"><label>Figure S2</label><caption>
<p><bold>Distribution of variants according to sequence features and allele frequency represented separately for the ESP and the 1000 Genomes datasets.</bold> The percentage of variants upstream of a functional domain (<bold>Panels A and E</bold>), in alternatively spliced sites (<bold>Panel B and F</bold>), in the principal isoform (<bold>panel C and G</bold>) and in regions targeted by NMD (<bold>Panel D and H</bold>). Panels A, B, C and D correspond to variants in the ESP dataset, and panels E, F, G and H to the 1000 Genomes dataset. The distribution is shown for synonymous (green), missense (blue), stop-gains (red) and frameshift (orange) variants according to minor allele frequency (MAF) intervals, where singletons (variants detected only in one individual) are represented separately. The pattern of OMIM disease variants and homozygous variants for each feature is shown. The corresponding coding genome background (measured as the percentage of nucleotides displaying the feature) is shown as a grey line (partly hidden by the distribution of synonymous variants in some panels). The y-axis represents the percentage of variants for the categories represented in the x-axis. Logistic regression was used to model the relationship between observing a given sequence feature in a given type of variant as a function of the logarithm of the minor allele frequency (MAF). In the ESP dataset, the odds ratio estimates for stop-gain variants were significantly different from those of synonymous variants in all panels (p-values&lt;1e-04, heterogeneity test <xref ref-type="bibr" rid="pcbi.1003757-QuintanaMurci1">[1]</xref>; for frameshifts, in panels B, C and D (p-values&lt;5e-02). In the 1000G dataset, the odds ratio estimates for stop-gain variants were significantly different from those of synonymous variants in all panels (p-values≪5e-02, heterogeneity test <xref ref-type="bibr" rid="pcbi.1003757-QuintanaMurci1">[1]</xref>; for frameshifts, in panel F (p-value&lt;5e-02). Distribution for frameshift variants from the 1000 Genomes dataset is noisy due to small sample size (<bold><xref ref-type="supplementary-material" rid="pcbi.1003757.s008">Table S1</xref></bold>).</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s003" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s003" position="float" xlink:type="simple"><label>Figure S3</label><caption>
<p><bold>Association of NMD-target variants with gene expression using standard RPKM normalization.</bold> Results in <xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2</bold></xref> are reproduced here using standard RPKM normalized expression values from Lappalainen et al. <xref ref-type="bibr" rid="pcbi.1003757-MacArthur1">[2]</xref>. <bold>Panel A</bold> shows the distribution of average expression z-scores for genes from individuals carrying different types of variants (synonymous, missense, frameshift and stop-gain). The black half represents the distribution of variants outside the NMD-target region and the colored half for those within the NMD-target region. As in <xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2</bold></xref>, statistically significant differences were observed for stop-gain variants predicted to trigger NMD (n = 756) compared to synonymous variants (one-sided Wilcoxon rank-sum test p-value&lt;2.2e-16). <bold>Panel B</bold> shows the distribution of average expression z-scores described in panel A for synonymous (grey) and stop-gain (dark and light purple) variants within the NMD-target region. The distribution of NMD-target stop-gains is represented separately for singletons (dark purple, n = 488) and non-singletons (n = 268). Distributions are statistically different (one-sided Wilcoxon rank-sum test = 4.4e-10). <bold>Panel C</bold> shows the distribution of average expression z-scores described in panel A for synonymous (grey) and stop-gain (dark and light pink) variants within the NMD-target region of genes with multiple isoforms described in CCDS. The distribution of NMD-target stop-gain is represented separately for those affecting all isoforms (dark pink, n = 216) and those affecting only a fraction of isoforms (light pink, n = 85). As in <xref ref-type="fig" rid="pcbi-1003757-g002"><bold>Figure 2</bold></xref>, distributions are statistically different (one-sided Wilcoxon rank-sum test = 1.5e-03).</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s004" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s004" position="float" xlink:type="simple"><label>Figure S4</label><caption>
<p><bold>Receiver operating characteristic of the performance of pathogenicity scores for stop and frameshift variants.</bold> Shown are the ROC curves corresponding to the sequence-based classifier (SB) developed in this work, a gene-based scores (GB) (<bold>Panels A–F</bold>: MacArthur 2012 <xref ref-type="bibr" rid="pcbi.1003757-Peterson1">[3]</xref>; <bold>Panels G–L</bold>: RVIS <xref ref-type="bibr" rid="pcbi.1003757-Adzhubei1">[4]</xref>), and the joint score combining the sequence-based and a gene-based score (SB×GB). Dashed curves correspond to a randomization test in which rows in sequence features are shuffled column-wise (denoted by SB<sup>(r)</sup> and GBxSB<sup>(r)</sup>). Classification power was evaluated on a set of pathogenic variants found in OMIM database (referred in the figure as Positives (Pos), and common variants not known to be pathogenic (referred in the figure as Negatives (Neg). Total number of Positive and Negative variants used is indicated above each panel. <bold>Panels A–C</bold> and <bold>G–I</bold> represent stop-gain variants while <bold>Panels D–F</bold> and <bold>J–L</bold> represent frameshif variants. Results are shown for both the ESP and 1000 Genomes datasets considered together (<bold>Panels A, D, G, J</bold>) or separately (<bold>Panels B, E, H, K</bold> for the ESP dataset and <bold>panels C, F, I, L</bold> for the 1000 Genomes dataset). Number of pathogenic and common variants used for benchmarking is shown on top of each panel. AUC values of ROC curves for each model are indicated. Incorporating sequence features led to an increased area under the ROC curve in all evaluated settings (<xref ref-type="fig" rid="pcbi-1003757-g003"><bold>Figure 3B</bold></xref>).</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s005" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s005" position="float" xlink:type="simple"><label>Figure S5</label><caption>
<p><bold>Correlation between sequence-based scores and gene-based scores for truncating variants.</bold> Figure shows the correlation between the sequence-based pathogenicity score developed in this work and two gene-based pathogenicity scores (<bold>Panels A</bold> and <bold>D:</bold> MacArthur 2012 <xref ref-type="bibr" rid="pcbi.1003757-Peterson1">[3]</xref>; <bold>Panels B</bold> and <bold>E</bold>: RVIS <xref ref-type="bibr" rid="pcbi.1003757-Adzhubei1">[4]</xref>). Correlation between the two gene-based scores is shown in <bold>Panels C</bold> and <bold>F</bold>. <bold>Panels A–C</bold> represent values for 17645 stop-gain variants reported by the ESP and the 1000 Genomes datasets (panels A–C). <bold>Panels D–F</bold> represent values for 155 disease stop-gain variants annotated as pathogenic by OMIM and reported by the ESP and the 1000 Genomes datasets (we note that OMIM variants used here were not considered for learning in the Bayesian classification; see <xref ref-type="sec" rid="s4">Methods</xref>). Upper <bold>Panels A–C</bold> display the distribution of the score on the y-axis in the form of boxplots conditioned to decile bins of the score on the x-axis. Lower <bold>Panels D–E</bold> represent the values for each individual OMIM variant (depicted with cross marks). For comparison across scores, they are represented as rank percentiles, where the value of a given variant accounts for the percentage of all stop-variants that had a score more pathogenic than the variant. Therefore, a rank percentile of “0” indicates a variant with the highest predicted probability of being pathogenic while a rank percentile of “100” indicates a variant with the lowest predicted severity. Grey triangles beside the panels represent the direction of increasing pathogenicity for the corresponding variable. Lines in <bold>Panels D–F</bold> divide variants in four regimes according to their belonging to the top 20% pathogenicity ranking of the corresponding scores, the top-right regime being the one where both scores agreed. Spearman rank correlation tests yielded significant p-values in panels A–C (p-value&lt;2.2e-16). Spearman correlations were &lt;0.13 (panels A and B; [0.107,0.125] and [0.115,0.123] 95% CI from 10,000 bootstrap samples, respectively), &lt;0.24 (panel C; [0.224,0.241] 95% CI). No significant p-values were found in panels D–E (Spearman correlation &lt;0.07). Similar figures were obtained for frameshift variants in analogous analyses to panels A–C. Analogous analyses to panels D–E on frameshifts variants were not possible due to lack of OMIM pathogenic frameshift variants in the ESP and 1000 Genomes datasets.</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s006" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s006" position="float" xlink:type="simple"><label>Figure S6</label><caption>
<p><bold>Pathogenicity score distributions for rare frameshift variants in innate immunity genes.</bold> Rank percentile distributions of pathogenicity scores for rare frameshift variants (MAF&lt;1%) are shown in different sets of genes: protein coding genome background (grey, “Genome”), innate immunity genes (light turquoise, “Inn Imm”) and their subset of interferon stimulated genes (dark turquoise, “ISGs”). In contrast with <xref ref-type="fig" rid="pcbi-1003757-g007"><bold>Figure 7</bold></xref>, the same categories for OMIM disease frameshifts are not shown due to low number or absence of variants. All variants are reported in ESP and 1000 Genomes Projects. Variants with the highest probability of being pathogenic have rank percentiles closer to zero (top of the panels). <bold>Panel A</bold> represents precomputed gene-based pathogenicity scores from <xref ref-type="bibr" rid="pcbi.1003757-Peterson1">[3]</xref>. <bold>Panel B</bold> represents sequence-based pathogenicity scores, i.e. posterior probabilities using the features described in the present work (see main text). Each box spans between 1st and 3rd quantile, and the median is denoted by a bold line in the middle. Total number of variants within each distribution is indicated. Differences in number of variants in equivalent categories between panel A and B originate from unavailability of the gene-based scores for some genes. Statistical differences against the genome reference (one-sided Wilcoxon rank sum tests) are indicated with asterisks according to Bonferroni corrected p-values: &lt;5e-02 (*), &lt;5e-03 (**) and &lt;5e-04 (**). The genome-wide median is denoted by a red line. Spearman correlation between the sequenced-based and gene-based pathogenicity scores was below 0.13 in all sets of genes analyzed.</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s007" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s007" position="float" xlink:type="simple"><label>Figure S7</label><caption>
<p><bold>Pipeline implemented to annotate genetic variants in reference human transcripts and protein sequences.</bold> Figure depicts the schematic pipeline followed for the annotation of variants (see <xref ref-type="sec" rid="s4">Methods</xref>). Analysis was restricted to variants affecting autosomal protein coding genes and transcripts annotated by the Consensus CDS (CCDS) project (<xref ref-type="bibr" rid="pcbi.1003757-Kumar1">[5]</xref>. Annotation of principal isoforms used APPRIS system (<xref ref-type="bibr" rid="pcbi.1003757-Petrovski1">[6]</xref>. Transcript-based information was related to protein-based information through UniProt <xref ref-type="bibr" rid="pcbi.1003757-Khurana1">[7]</xref>. InterPro database (<xref ref-type="bibr" rid="pcbi.1003757-Wang1">[8]</xref> was used to retrieve protein domain information.</p>
<p>(TIF)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s008" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s008" position="float" xlink:type="simple"><label>Table S1</label><caption>
<p><bold>Distribution of variants according to allele frequency and dataset.</bold> Number of variants included in the study is reported in total and according to their original dataset: the NHLBI GO Exome Sequencing Project (ESP) and the 1000 Genomes Project. Distribution is shown according to variant type and minor allele frequency intervals and the number of genes bearing each type of variants is reported.</p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s009" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s009" position="float" xlink:type="simple"><label>Table S2</label><caption>
<p><bold>Distribution of variants displaying different sequence features according to allele frequency.</bold> Table shows the absolute numbers and corresponding percentages of the distributions of variants shown in <xref ref-type="fig" rid="pcbi-1003757-g001"><bold>Figure 1</bold></xref>. Absolute number (1), reference number (2) and percentage (3) of variants upstream of a functional domain (A), in alternatively spliced sites (B), in the principal isoform (C) and in regions targeted by NMD (D) are shown according to minor allele frequency intervals. Corresponding figures are reported for OMIM disease variants and homozygous variants together with a coding genome background reference measured in nucleotides.</p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s010" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s010" position="float" xlink:type="simple"><label>Table S3</label><caption>
<p><bold>Parameters of the Naïve Bayesian classifier learned from the joint dataset.</bold></p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s011" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s011" position="float" xlink:type="simple"><label>Table S4</label><caption>
<p><bold>Sequence features and pathogenicity scores of gene truncating variants in antiviral interferon stimulated genes.</bold> Table shows figures for 15 stop-gain and 7 frameshift variants affecting 13 of 42 genes with anti-viral activity in cellular assays <xref ref-type="bibr" rid="pcbi.1003757-Ritchie1">[9]</xref> <xref ref-type="bibr" rid="pcbi.1003757-Kircher1">[10]</xref>. Analysis was restricted to variants identified in at least two individuals. Variants are ranked according to their sequence-based pathogenicity score. High-scoring variants affecting <italic>MX1</italic> and <italic>HPSE</italic> are highlighted in violet and green respectively and discussed in the main text.</p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pcbi.1003757.s012" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xlink:href="info:doi/10.1371/journal.pcbi.1003757.s012" position="float" xlink:type="simple"><label>Text S1</label><caption>
<p><bold>References in Supplementary Information legends.</bold></p>
<p>(DOCX)</p>
</caption></supplementary-material></sec></body>
<back>
<ack>
<p>The authors would like to thank Zoltan Kutalik for feedback on statistical analyses, the 1000 Genomes Project, the Geuvadis Consortium and the NHLBI GO Exome Sequencing Project. Some of the computations for this study were performed at the Vital-IT (<ext-link ext-link-type="uri" xlink:href="http://www.vital-it.ch" xlink:type="simple">http://www.vital-it.ch</ext-link>) center for high-performance computing of the Swiss Institute of Bioinformatics. The pipeline's code implemented in this work for the analysis of genome and exome variants is freely available for download from <ext-link ext-link-type="uri" xlink:href="http://nutvar.labtelenti.org/" xlink:type="simple">http://nutvar.labtelenti.org/</ext-link>.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pcbi.1003757-QuintanaMurci1"><label>1</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Quintana-Murci</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Alcais</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Abel</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Casanova</surname><given-names>JL</given-names></name> (<year>2007</year>) <article-title>Immunology in natura: clinical, epidemiological and evolutionary genetics of infectious diseases</article-title>. <source>Nat Immunol</source> <volume>8</volume>: <fpage>1165</fpage>–<lpage>1171</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-MacArthur1"><label>2</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>MacArthur</surname><given-names>DG</given-names></name>, <name name-style="western"><surname>Manolio</surname><given-names>TA</given-names></name>, <name name-style="western"><surname>Dimmock</surname><given-names>DP</given-names></name>, <name name-style="western"><surname>Rehm</surname><given-names>HL</given-names></name>, <name name-style="western"><surname>Shendure</surname><given-names>J</given-names></name>, <etal>et al</etal>. (<year>2014</year>) <article-title>Guidelines for investigating causality of sequence variants in human disease</article-title>. <source>Nature</source> <volume>508</volume>: <fpage>469</fpage>–<lpage>476</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Peterson1"><label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Peterson</surname><given-names>TA</given-names></name>, <name name-style="western"><surname>Doughty</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Kann</surname><given-names>MG</given-names></name> (<year>2013</year>) <article-title>Towards precision medicine: advances in computational approaches for the analysis of human variants</article-title>. <source>J Mol Biol</source> <volume>425</volume>: <fpage>4047</fpage>–<lpage>4063</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Adzhubei1"><label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Adzhubei</surname><given-names>IA</given-names></name>, <name name-style="western"><surname>Schmidt</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Peshkin</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Ramensky</surname><given-names>VE</given-names></name>, <name name-style="western"><surname>Gerasimova</surname><given-names>A</given-names></name>, <etal>et al</etal>. (<year>2010</year>) <article-title>A method and server for predicting damaging missense mutations</article-title>. <source>Nat Methods</source> <volume>7</volume>: <fpage>248</fpage>–<lpage>249</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Kumar1"><label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kumar</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Henikoff</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Ng</surname><given-names>PC</given-names></name> (<year>2009</year>) <article-title>Predicting the effects of coding non-synonymous variants on protein function using the SIFT algorithm</article-title>. <source>Nat Protoc</source> <volume>4</volume>: <fpage>1073</fpage>–<lpage>1081</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Petrovski1"><label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Petrovski</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Wang</surname><given-names>Q</given-names></name>, <name name-style="western"><surname>Heinzen</surname><given-names>EL</given-names></name>, <name name-style="western"><surname>Allen</surname><given-names>AS</given-names></name>, <name name-style="western"><surname>Goldstein</surname><given-names>DB</given-names></name> (<year>2013</year>) <article-title>Genic intolerance to functional variation and the interpretation of personal genomes</article-title>. <source>PLoS Genet</source> <volume>9</volume>: <fpage>e1003709</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Khurana1"><label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Khurana</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Fu</surname><given-names>Y</given-names></name>, <name name-style="western"><surname>Colonna</surname><given-names>V</given-names></name>, <name name-style="western"><surname>Mu</surname><given-names>XJ</given-names></name>, <name name-style="western"><surname>Kang</surname><given-names>HM</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Integrative annotation of variants from 1092 humans: application to cancer genomics</article-title>. <source>Science</source> <volume>342</volume>: <fpage>1235587</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Wang1"><label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wang</surname><given-names>K</given-names></name>, <name name-style="western"><surname>Li</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Hakonarson</surname><given-names>H</given-names></name> (<year>2010</year>) <article-title>ANNOVAR: functional annotation of genetic variants from high-throughput sequencing data</article-title>. <source>Nucleic Acids Res</source> <volume>38</volume>: <fpage>e164</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Ritchie1"><label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ritchie</surname><given-names>GR</given-names></name>, <name name-style="western"><surname>Dunham</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Zeggini</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Flicek</surname><given-names>P</given-names></name> (<year>2014</year>) <article-title>Functional annotation of noncoding sequence variants</article-title>. <source>Nat Methods</source> <volume>11</volume>: <fpage>294</fpage>–<lpage>296</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Kircher1"><label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kircher</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Witten</surname><given-names>DM</given-names></name>, <name name-style="western"><surname>Jain</surname><given-names>P</given-names></name>, <name name-style="western"><surname>O'Roak</surname><given-names>BJ</given-names></name>, <name name-style="western"><surname>Cooper</surname><given-names>GM</given-names></name>, <etal>et al</etal>. (<year>2014</year>) <article-title>A general framework for estimating the relative pathogenicity of human genetic variants</article-title>. <source>Nat Genet</source> <volume>46</volume>: <fpage>310</fpage>–<lpage>315</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Genomes1"><label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple"><collab xlink:type="simple">Genomes Project C</collab> (<year>2010</year>) <name name-style="western"><surname>Abecasis</surname><given-names>GR</given-names></name>, <name name-style="western"><surname>Altshuler</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Auton</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Brooks</surname><given-names>LD</given-names></name>, <etal>et al</etal>. (<year>2010</year>) <article-title>A map of human genome variation from population-scale sequencing</article-title>. <source>Nature</source> <volume>467</volume>: <fpage>1061</fpage>–<lpage>1073</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-MacArthur2"><label>12</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>MacArthur</surname><given-names>DG</given-names></name>, <name name-style="western"><surname>Tyler-Smith</surname><given-names>C</given-names></name> (<year>2010</year>) <article-title>Loss-of-function variants in the genomes of healthy humans</article-title>. <source>Hum Mol Genet</source> <volume>19</volume>: <fpage>R125</fpage>–<lpage>130</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Nagy1"><label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nagy</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Maquat</surname><given-names>LE</given-names></name> (<year>1998</year>) <article-title>A rule for termination-codon position within intron-containing genes: when nonsense affects RNA abundance</article-title>. <source>Trends Biochem Sci</source> <volume>23</volume>: <fpage>198</fpage>–<lpage>199</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Nelson1"><label>14</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nelson</surname><given-names>MR</given-names></name>, <name name-style="western"><surname>Wegmann</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Ehm</surname><given-names>MG</given-names></name>, <name name-style="western"><surname>Kessner</surname><given-names>D</given-names></name>, <name name-style="western"><surname>St Jean</surname><given-names>P</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>An abundance of rare functional variants in 202 drug target genes sequenced in 14,002 people</article-title>. <source>Science</source> <volume>337</volume>: <fpage>100</fpage>–<lpage>104</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Tennessen1"><label>15</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Tennessen</surname><given-names>JA</given-names></name>, <name name-style="western"><surname>Bigham</surname><given-names>AW</given-names></name>, <name name-style="western"><surname>O'Connor</surname><given-names>TD</given-names></name>, <name name-style="western"><surname>Fu</surname><given-names>W</given-names></name>, <name name-style="western"><surname>Kenny</surname><given-names>EE</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>Evolution and functional impact of rare coding variation from deep sequencing of human exomes</article-title>. <source>Science</source> <volume>337</volume>: <fpage>64</fpage>–<lpage>69</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Fu1"><label>16</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Fu</surname><given-names>W</given-names></name>, <name name-style="western"><surname>O'Connor</surname><given-names>TD</given-names></name>, <name name-style="western"><surname>Jun</surname><given-names>G</given-names></name>, <name name-style="western"><surname>Kang</surname><given-names>HM</given-names></name>, <name name-style="western"><surname>Abecasis</surname><given-names>G</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Analysis of 6,515 exomes reveals the recent origin of most human protein-coding variants</article-title>. <source>Nature</source> <volume>493</volume>: <fpage>216</fpage>–<lpage>220</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Jungreis1"><label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Jungreis</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Lin</surname><given-names>MF</given-names></name>, <name name-style="western"><surname>Spokony</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Chan</surname><given-names>CS</given-names></name>, <name name-style="western"><surname>Negre</surname><given-names>N</given-names></name>, <etal>et al</etal>. (<year>2011</year>) <article-title>Evidence of abundant stop codon readthrough in Drosophila and other metazoa</article-title>. <source>Genome Res</source> <volume>21</volume>: <fpage>2096</fpage>–<lpage>2113</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Wills1"><label>18</label>
<mixed-citation publication-type="book" xlink:type="simple">Wills N (2010) Translational bypassing—peptidyl-tRNA repairing at nonoverlapping sites. In: JF Atkins RG, editor. Recoding: Expansion of decoding rules enriches gene expression. New York: Springer. pp. 365–381.</mixed-citation>
</ref>
<ref id="pcbi.1003757-MacArthur3"><label>19</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>MacArthur</surname><given-names>DG</given-names></name>, <name name-style="western"><surname>Balasubramanian</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Frankish</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Huang</surname><given-names>N</given-names></name>, <name name-style="western"><surname>Morris</surname><given-names>J</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>A systematic survey of loss-of-function variants in human protein-coding genes</article-title>. <source>Science</source> <volume>335</volume>: <fpage>823</fpage>–<lpage>828</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Rausell1"><label>20</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rausell</surname><given-names>A</given-names></name>, <name name-style="western"><surname>McLaren</surname><given-names>PJ</given-names></name>, <name name-style="western"><surname>Telenti</surname><given-names>A</given-names></name> (<year>2013</year>) <article-title>HIV and innate immunity - a genomics perspective</article-title>. <source>F1000Prime Rep</source> <volume>5</volume>: <fpage>29</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Genomes2"><label>21</label>
<mixed-citation publication-type="journal" xlink:type="simple"><collab xlink:type="simple">Genomes Project C</collab> (<year>2012</year>) <name name-style="western"><surname>Abecasis</surname><given-names>GR</given-names></name>, <name name-style="western"><surname>Auton</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Brooks</surname><given-names>LD</given-names></name>, <name name-style="western"><surname>DePristo</surname><given-names>MA</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>An integrated map of genetic variation from 1,092 human genomes</article-title>. <source>Nature</source> <volume>491</volume>: <fpage>56</fpage>–<lpage>65</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Lappalainen1"><label>22</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Lappalainen</surname><given-names>T</given-names></name>, <name name-style="western"><surname>Sammeth</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Friedlander</surname><given-names>MR</given-names></name>, <name name-style="western"><surname>t Hoen</surname><given-names>PA</given-names></name>, <name name-style="western"><surname>Monlong</surname><given-names>J</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Transcriptome and genome sequencing uncovers functional variation in humans</article-title>. <source>Nature</source> <volume>501</volume>: <fpage>506</fpage>–<lpage>511</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Pruitt1"><label>23</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Pruitt</surname><given-names>KD</given-names></name>, <name name-style="western"><surname>Harrow</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Harte</surname><given-names>RA</given-names></name>, <name name-style="western"><surname>Wallin</surname><given-names>C</given-names></name>, <name name-style="western"><surname>Diekhans</surname><given-names>M</given-names></name>, <etal>et al</etal>. (<year>2009</year>) <article-title>The consensus coding sequence (CCDS) project: Identifying a common protein-coding gene set for the human and mouse genomes</article-title>. <source>Genome Res</source> <volume>19</volume>: <fpage>1316</fpage>–<lpage>1323</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Yngvadottir1"><label>24</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Yngvadottir</surname><given-names>B</given-names></name>, <name name-style="western"><surname>Xue</surname><given-names>Y</given-names></name>, <name name-style="western"><surname>Searle</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Hunt</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Delgado</surname><given-names>M</given-names></name>, <etal>et al</etal>. (<year>2009</year>) <article-title>A genome-wide survey of the prevalence and evolutionary forces acting on human nonsense SNPs</article-title>. <source>Am J Hum Genet</source> <volume>84</volume>: <fpage>224</fpage>–<lpage>234</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-GonzalezPorta1"><label>25</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gonzalez-Porta</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Frankish</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Rung</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Harrow</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Brazma</surname><given-names>A</given-names></name> (<year>2013</year>) <article-title>Transcriptome analysis of human tissues and cell lines reveals one dominant transcript per gene</article-title>. <source>Genome Biol</source> <volume>14</volume>: <fpage>R70</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Schoggins1"><label>26</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schoggins</surname><given-names>JW</given-names></name>, <name name-style="western"><surname>Wilson</surname><given-names>SJ</given-names></name>, <name name-style="western"><surname>Panis</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Murphy</surname><given-names>MY</given-names></name>, <name name-style="western"><surname>Jones</surname><given-names>CT</given-names></name>, <etal>et al</etal>. (<year>2011</year>) <article-title>A diverse range of gene products are effectors of the type I interferon antiviral response</article-title>. <source>Nature</source> <volume>472</volume>: <fpage>481</fpage>–<lpage>485</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Schoggins2"><label>27</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schoggins</surname><given-names>JW</given-names></name>, <name name-style="western"><surname>Macduff</surname><given-names>DA</given-names></name>, <name name-style="western"><surname>Imanaka</surname><given-names>N</given-names></name>, <name name-style="western"><surname>Gainey</surname><given-names>MD</given-names></name>, <name name-style="western"><surname>Shrestha</surname><given-names>B</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Pan-viral specificity of IFN-induced genes reveals new roles for cGAS in innate immunity</article-title>. <source>Nature</source> <volume>505</volume>: <fpage>691</fpage>–<lpage>695</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Montgomery1"><label>28</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Montgomery</surname><given-names>SB</given-names></name>, <name name-style="western"><surname>Goode</surname><given-names>DL</given-names></name>, <name name-style="western"><surname>Kvikstad</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Albers</surname><given-names>CA</given-names></name>, <name name-style="western"><surname>Zhang</surname><given-names>ZD</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>The origin, evolution, and functional impact of short insertion-deletion variants identified in 179 human genomes</article-title>. <source>Genome Res</source> <volume>23</volume>: <fpage>749</fpage>–<lpage>761</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Montgomery2"><label>29</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Montgomery</surname><given-names>SB</given-names></name>, <name name-style="western"><surname>Lappalainen</surname><given-names>T</given-names></name>, <name name-style="western"><surname>Gutierrez-Arcelus</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Dermitzakis</surname><given-names>ET</given-names></name> (<year>2011</year>) <article-title>Rare and common regulatory variation in population-scale sequenced human genomes</article-title>. <source>PLoS Genet</source> <volume>7</volume>: <fpage>e1002144</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Kukurba1"><label>30</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kukurba</surname><given-names>KR</given-names></name>, <name name-style="western"><surname>Zhang</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Li</surname><given-names>X</given-names></name>, <name name-style="western"><surname>Smith</surname><given-names>KS</given-names></name>, <name name-style="western"><surname>Knowles</surname><given-names>DA</given-names></name>, <etal>et al</etal>. (<year>2014</year>) <article-title>Allelic Expression of Deleterious Protein-Coding Variants across Human Tissues</article-title>. <source>PLoS Genet</source> <volume>10</volume>: <fpage>e1004304</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-White1"><label>31</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>White</surname><given-names>JK</given-names></name>, <name name-style="western"><surname>Gerdin</surname><given-names>AK</given-names></name>, <name name-style="western"><surname>Karp</surname><given-names>NA</given-names></name>, <name name-style="western"><surname>Ryder</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Buljan</surname><given-names>M</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Genome-wide generation and systematic phenotyping of knockout mice reveals new roles for many genes</article-title>. <source>Cell</source> <volume>154</volume>: <fpage>452</fpage>–<lpage>464</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Cingolani1"><label>32</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Cingolani</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Platts</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Wang le</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Coon</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Nguyen</surname><given-names>T</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>A program for annotating and predicting the effects of single nucleotide polymorphisms, SnpEff: SNPs in the genome of Drosophila melanogaster strain w1118; iso-2; iso-3</article-title>. <source>Fly (Austin)</source> <volume>6</volume>: <fpage>80</fpage>–<lpage>92</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Hunter1"><label>33</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Hunter</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Jones</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Mitchell</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Apweiler</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Attwood</surname><given-names>TK</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>InterPro in 2011: new developments in the family and domain prediction database</article-title>. <source>Nucleic Acids Res</source> <volume>40</volume>: <fpage>D306</fpage>–<lpage>312</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Guberman1"><label>34</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Guberman</surname><given-names>JM</given-names></name>, <name name-style="western"><surname>Ai</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Arnaiz</surname><given-names>O</given-names></name>, <name name-style="western"><surname>Baran</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Blake</surname><given-names>A</given-names></name>, <etal>et al</etal>. (<year>2011</year>) <article-title>BioMart Central Portal: an open database network for the biological community</article-title>. <source>Database (Oxford)</source> <volume>2011</volume>: <fpage>bar041</fpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-UniProt1"><label>35</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>UniProt</surname><given-names>C</given-names></name> (<year>2013</year>) <article-title>Update on activities at the Universal Protein Resource (UniProt) in 2013</article-title>. <source>Nucleic Acids Res</source> <volume>41</volume>: <fpage>D43</fpage>–<lpage>47</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Rodriguez1"><label>36</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rodriguez</surname><given-names>JM</given-names></name>, <name name-style="western"><surname>Maietta</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Ezkurdia</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Pietrelli</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Wesselink</surname><given-names>JJ</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>APPRIS: annotation of principal and alternative splice isoforms</article-title>. <source>Nucleic Acids Res</source> <volume>41</volume>: <fpage>D110</fpage>–<lpage>117</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-OMIM1"><label>37</label>
<mixed-citation publication-type="other" xlink:type="simple">OMIM (2013) Online Mendelian Inheritance in Man (<ext-link ext-link-type="uri" xlink:href="http://www.omim.org/" xlink:type="simple">http://www.omim.org/</ext-link>). McKusick-Nathans Institute of Genetic Medicine, Johns Hopkins University (Baltimore, MD).</mixed-citation>
</ref>
<ref id="pcbi.1003757-Flicek1"><label>38</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Flicek</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Ahmed</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Amode</surname><given-names>MR</given-names></name>, <name name-style="western"><surname>Barrell</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Beal</surname><given-names>K</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Ensembl 2013</article-title>. <source>Nucleic Acids Res</source> <volume>41</volume>: <fpage>D48</fpage>–<lpage>55</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Massingham1"><label>39</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Massingham</surname><given-names>T</given-names></name>, <name name-style="western"><surname>Goldman</surname><given-names>N</given-names></name> (<year>2005</year>) <article-title>Detecting amino acid sites under positive selection and purifying selection</article-title>. <source>Genetics</source> <volume>169</volume>: <fpage>1753</fpage>–<lpage>1762</lpage>.</mixed-citation>
</ref>
<ref id="pcbi.1003757-Randall1"><label>40</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Randall</surname><given-names>JC</given-names></name>, <name name-style="western"><surname>Winkler</surname><given-names>TW</given-names></name>, <name name-style="western"><surname>Kutalik</surname><given-names>Z</given-names></name>, <name name-style="western"><surname>Berndt</surname><given-names>SI</given-names></name>, <name name-style="western"><surname>Jackson</surname><given-names>AU</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Sex-stratified genome-wide association studies including 270,000 individuals show sexual dimorphism in genetic loci for anthropometric traits</article-title>. <source>PLoS Genet</source> <volume>9</volume>: <fpage>e1003500</fpage>.</mixed-citation>
</ref>
</ref-list></back>
</article>