<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
    <front>
        <journal-meta><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="nlm-ta">PLoS Genet</journal-id><journal-id journal-id-type="pmc">plosgen</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS Genetics</journal-title></journal-title-group><issn pub-type="ppub">1553-7390</issn><issn pub-type="epub">1553-7404</issn><publisher>
                <publisher-name>Public Library of Science</publisher-name>
                <publisher-loc>San Francisco, USA</publisher-loc>
            </publisher></journal-meta>
        <article-meta><article-id pub-id-type="publisher-id">09-PLGE-RA-0191R3</article-id><article-id pub-id-type="doi">10.1371/journal.pgen.1000668</article-id><article-categories>
                <subj-group subj-group-type="heading">
                    <subject>Research Article</subject>
                </subj-group>
                <subj-group subj-group-type="Discipline">
                    <subject>Computational Biology/Genomics</subject>
                    <subject>Genetics and Genomics</subject>
                    <subject>Genetics and Genomics/Bioinformatics</subject>
                    <subject>Genetics and Genomics/Genomics</subject>
                </subj-group>
            </article-categories><title-group><article-title>Needles in the Haystack: Identifying Individuals Present in Pooled
                    Genomic Data</article-title><alt-title alt-title-type="running-head">Identifying Individuals in Pooled Genomic
                    Data</alt-title></title-group><contrib-group>
                <contrib contrib-type="author" xlink:type="simple">
                    <name name-style="western">
                        <surname>Braun</surname>
                        <given-names>Rosemary</given-names>
                    </name>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                    <xref ref-type="corresp" rid="cor1">
                        <sup>*</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="author" xlink:type="simple">
                    <name name-style="western">
                        <surname>Rowe</surname>
                        <given-names>William</given-names>
                    </name>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="author" xlink:type="simple">
                    <name name-style="western">
                        <surname>Schaefer</surname>
                        <given-names>Carl</given-names>
                    </name>
                    <xref ref-type="aff" rid="aff2">
                        <sup>2</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="author" xlink:type="simple">
                    <name name-style="western">
                        <surname>Zhang</surname>
                        <given-names>Jinghui</given-names>
                    </name>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                </contrib>
                <contrib contrib-type="author" xlink:type="simple">
                    <name name-style="western">
                        <surname>Buetow</surname>
                        <given-names>Kenneth</given-names>
                    </name>
                    <xref ref-type="aff" rid="aff1">
                        <sup>1</sup>
                    </xref>
                    <xref ref-type="aff" rid="aff2">
                        <sup>2</sup>
                    </xref>
                </contrib>
            </contrib-group><aff id="aff1">
                <label>1</label>
                <addr-line>Laboratory of Population Genetics, National Cancer Institute, National
                    Institutes of Health, Bethesda, Maryland, United States of America</addr-line>
            </aff><aff id="aff2">
                <label>2</label>
                <addr-line>Center for Biomedical Informatics and Information Technology, National
                    Cancer Institute, National Institutes of Health, Bethesda, Maryland, United
                    States of America</addr-line>
            </aff><contrib-group>
                <contrib contrib-type="editor" xlink:type="simple">
                    <name name-style="western">
                        <surname>Gibson</surname>
                        <given-names>Greg</given-names>
                    </name>
                    <role>Editor</role>
                    <xref ref-type="aff" rid="edit1"/>
                </contrib>
            </contrib-group><aff id="edit1">The University of Queensland, Australia</aff><author-notes>
                <corresp id="cor1">* E-mail: <email xlink:type="simple">braunr@mail.nih.gov</email></corresp>
                <fn fn-type="con">
                    <p>Conceived and designed the experiments: RB WR CS JZ KB. Performed the
                        experiments: RB. Analyzed the data: RB. Contributed
                        reagents/materials/analysis tools: RB. Wrote the paper: RB.</p>
                </fn>
            <fn fn-type="conflict">
                <p>The authors have declared that no competing interests exist.</p>
            </fn></author-notes><pub-date pub-type="collection">
                <month>10</month>
                <year>2009</year>
            </pub-date><pub-date pub-type="epub">
                <day>2</day>
                <month>10</month>
                <year>2009</year>
            </pub-date><volume>5</volume><issue>10</issue><elocation-id>e1000668</elocation-id><history>
                <date date-type="received">
                    <day>6</day>
                    <month>2</month>
                    <year>2009</year>
                </date>
                <date date-type="accepted">
                    <day>31</day>
                    <month>8</month>
                    <year>2009</year>
                </date>
            </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2009</copyright-year><license><license-p>This is an open-access article distributed under the terms of the
                Creative Commons Public Domain declaration which stipulates that, once placed in the
                public domain, this work may be freely reproduced, distributed, transmitted,
                modified, built upon, or otherwise used by anyone for any lawful purpose.</license-p></license></permissions><related-article ext-link-type="uri" id="RA1" page="e1000665" related-article-type="companion" vol="5" xlink:href="info:doi/10.1371/journal.pgen.1000665" xlink:type="simple"> <article-title>Public Access to Genome-Wide Data: Five Views on Balancing Research with Privacy and Protection</article-title>
            </related-article><related-article ext-link-type="uri" id="RA2" page="e1000628" related-article-type="companion" vol="5" xlink:href="info:doi/10.1371/journal.pgen.1000628" xlink:type="simple"> <article-title>The Limits of Individual Identification from Sample Allele Frequencies: Theory and Statistical Analysis</article-title></related-article><related-article ext-link-type="uri" id="RA3" page="e1000167" related-article-type="companion" vol="4" xlink:href="info:doi/10.1371/journal.pgen.1000167" xlink:type="simple"> <article-title>Resolving Individuals Contributing Trace Amounts of DNA to Highly Complex Mixtures Using High-Density SNP Genotyping Microarrays</article-title></related-article><abstract>
                <p>Recent publications have described and applied a novel metric that quantifies the
                    genetic distance of an individual with respect to two population samples, and
                    have suggested that the metric makes it possible to infer the presence of an
                    individual of known genotype in a sample for which only the marginal allele
                    frequencies are known. However, the assumptions, limitations, and utility of
                    this metric remained incompletely characterized. Here we present empirical tests
                    of the method using publicly accessible genotypes, as well as analytical
                    investigations of the method's strengths and limitations. The results
                    reveal that the null distribution is sensitive to the underlying assumptions,
                    making it difficult to accurately calibrate thresholds for classifying an
                    individual as a member of the population samples. As a result, the
                    false-positive rates obtained in practice are considerably higher than
                    previously believed. However, despite the metric's inadequacies for
                    identifying the presence of an individual in a sample, our results suggest
                    potential avenues for future research on tuning this method to problems of
                    ancestry inference or disease prediction. By revealing both the strengths and
                    limitations of the proposed method, we hope to elucidate situations in which
                    this distance metric may be used in an appropriate manner. We also discuss the
                    implications of our findings in forensics applications and in the protection of
                    GWAS participant privacy.</p>
            </abstract><abstract abstract-type="summary">
                <title>Author Summary</title>
                <p>In this report, we evaluate a recently-published method for resolving whether
                    individuals are present in a complex genomic DNA mixture. Based on the intuition
                    that an individual will be genetically “closer” to a sample
                    containing him than to a sample not, the method investigated here uses a
                    distance metric to quantify the similarity of an individual relative to two
                    population samples. Although initial applications of this approach showed a
                    promising false-negative rate, the accuracy of the assumed null distribution
                    (and hence the true false-positive rate) remained uninvestigated; here, we
                    explore this question analytically and describe tests of this method to assess
                    the likelihood that an individual who is not in the mixture is mistakenly
                    classified as being a member. Our results show that the method has a high
                    false-positive rate in practice due to its sensitivity to underlying
                    assumptions, limiting its utility for inferring the presence of an individual in
                    a population. By revealing both the strengths and limitations of the proposed
                    method, we elucidate situations in which this distance metric may be used in an
                    appropriate manner in forensics and medical privacy policy.</p>
            </abstract><funding-group><funding-statement>This research was supported by the Intramural Research Program of the National
                    Cancer Institute, National Institutes of Health, Bethesda, MD. RB was supported
                    by the Cancer Prevention Fellowship Program, National Cancer Institute, National
                    Institutes of Health, Bethesda, MD. The funders had no role in study design,
                    data collection and analysis, decision to publish, or preparation of the
                    manuscript.</funding-statement></funding-group><counts>
                <page-count count="8"/>
            </counts></article-meta>
    </front>
    <body>
        <sec id="s1">
            <title>Introduction</title>
            <p>In the recently published article “Resolving Individuals Contributing Trace
                Amounts of DNA to Highly Complex Mixtures Using High-Density SNP Genotyping
                Microarrays” <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>, the authors describe a method by which the
                presence of a individual with a known genotype may be inferred as being part of a
                mixture of genetic material for which marginal minor allele frequencies (MAFs), but
                not sample genotypes, are known.</p>
            <p>The method <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> is motivated by the idea that the presence of a
                specific individual's genetic material will bias the MAFs of a sample of
                which they are part in a subtle but systematic manner, such that when considering
                multiple loci, the bias introduced by a specific individual can be detected even
                when his DNA comprises only a small fraction of the mixture. More generally, it is
                well known that samples of a population will exhibit slightly different MAFs due to
                sampling variance following a binomial distribution; the genotype of the individual
                in question contributes to this variation, and so may be
                “closer” to a sample containing him than to a sample which does
                not. Based on this intuition, the article <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> defines a genetic
                distance statistic to measure the distance of an individual relative to two samples,
                summarized as follows:</p>
            <p>Consider an underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e001" xlink:type="simple"/></inline-formula> from which two samples <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e002" xlink:type="simple"/></inline-formula> (of size <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e003" xlink:type="simple"/></inline-formula>) and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e004" xlink:type="simple"/></inline-formula> (of size <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e005" xlink:type="simple"/></inline-formula>) are drawn independently and identically distributed (i.i.d.)
                [in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>, these are referred to as
                “reference” and “mixture”
                respectively]. Consider now an additional sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e006" xlink:type="simple"/></inline-formula>; we wish to detect whether <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e007" xlink:type="simple"/></inline-formula> was drawn from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e008" xlink:type="simple"/></inline-formula>, versus the null hypothesis that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e009" xlink:type="simple"/></inline-formula> was drawn from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e010" xlink:type="simple"/></inline-formula> independent of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e011" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e012" xlink:type="simple"/></inline-formula>. Given the MAFs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e013" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e014" xlink:type="simple"/></inline-formula> at locus <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e015" xlink:type="simple"/></inline-formula> for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e016" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e017" xlink:type="simple"/></inline-formula>, respectively, and given the MAFs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e018" xlink:type="simple"/></inline-formula> for sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e019" xlink:type="simple"/></inline-formula> with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e020" xlink:type="simple"/></inline-formula> (corresponding to homozygous major, heterozygous, and homozygous
                minor alleles) at each locus <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e021" xlink:type="simple"/></inline-formula>, <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> defines the relative distance of sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e022" xlink:type="simple"/></inline-formula> from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e023" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e024" xlink:type="simple"/></inline-formula> at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e025" xlink:type="simple"/></inline-formula> as:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e026" xlink:type="simple"/><label>(1)</label></disp-formula>By assuming only independent loci are chosen and invoking the central
                limit theorem for the large number of loci genotyped in modern studies, the article
                    <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>
                asserts that the <italic>z</italic>-score of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e027" xlink:type="simple"/></inline-formula> across all loci will be normally distributed,<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e028" xlink:type="simple"/><label>(2)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e029" xlink:type="simple"/></inline-formula> denotes the average over all SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e030" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e031" xlink:type="simple"/></inline-formula> is the number of SNPs, and Equation 2 exploits the assumption
                    <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>
                that an individual who is in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e032" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e033" xlink:type="simple"/></inline-formula> will be on average equidistant to both under the null hypothesis,
                i.e., <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e034" xlink:type="simple"/></inline-formula>. Per Equation 2, the null hypothesis that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e035" xlink:type="simple"/></inline-formula> is in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e036" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e037" xlink:type="simple"/></inline-formula> is rejected for values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e038" xlink:type="simple"/></inline-formula> which exceed the quantiles of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e039" xlink:type="simple"/></inline-formula> at the chosen significance level.</p>
            <p>The article <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> proposes using this approach in a forensics context,
                in which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e040" xlink:type="simple"/></inline-formula> is a mixture of genetic material of unknown composition (e.g.,
                from a crime scene), and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e041" xlink:type="simple"/></inline-formula> is suspect's genotype; by choosing an appropriate
                reference sample for group <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e042" xlink:type="simple"/></inline-formula>, it is hypothesized that large, positive <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e043" xlink:type="simple"/></inline-formula> will be obtained for individuals whose genotypes are included in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e044" xlink:type="simple"/></inline-formula>, and hence bias <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e045" xlink:type="simple"/></inline-formula>, while individuals whose genotypes are not in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e046" xlink:type="simple"/></inline-formula> should have insignificant <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e047" xlink:type="simple"/></inline-formula> since they should intuitively be no more similar to the mixture
                sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e048" xlink:type="simple"/></inline-formula> than they are to the reference sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e049" xlink:type="simple"/></inline-formula>. In <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>, the authors applied this test to a multitude of
                individuals <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e050" xlink:type="simple"/></inline-formula>, each of which are present in the samples constructed by them for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e051" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e052" xlink:type="simple"/></inline-formula>, and report near-zero false negative rates. The article concludes
                that it is possible to identify the presence of DNA of specific individuals within a
                series of highly complex genomic mixtures, and that these “findings show a
                clear path for identifying whether specific individuals are within a study based on
                summary-level statistics.” In response, many GWAS data sources have
                retracted the publicly available frequency data pending further study of this method
                due to the concern that the privacy of study participants can be compromised.
                However, because no samples absent from both <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e053" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e054" xlink:type="simple"/></inline-formula> were used, false positive rates—significant <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e055" xlink:type="simple"/></inline-formula> for individuals neither in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e056" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e057" xlink:type="simple"/></inline-formula>—are not assessed in practice; rather, they are simply
                assumed (Equation 2) to follow the nominal false-positive rate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e058" xlink:type="simple"/></inline-formula> given by quantiles of the standard normal.</p>
            <p>The conclusion that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e059" xlink:type="simple"/></inline-formula> is comparable to a standard normal rests on several assumptions:</p>
            <list list-type="order">
                <list-item>
                    <p>that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e060" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e061" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e062" xlink:type="simple"/></inline-formula> are all samples of the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e063" xlink:type="simple"/></inline-formula>;</p>
                </list-item>
                <list-item>
                    <p>that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e064" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e065" xlink:type="simple"/></inline-formula> are similarly sized samples; and</p>
                </list-item>
                <list-item>
                    <p>that the SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e066" xlink:type="simple"/></inline-formula> used to compute <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e067" xlink:type="simple"/></inline-formula> are independent.</p>
                </list-item>
            </list>
            <p>Because these assumptions are difficult to control in practice, the effect of
                deviations from these assumptions is of interest. In this manuscript, we expand on
                    <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> by
                investigating these effects both analytically and by applying Equations 1, 2 to null
                samples (those present in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e068" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e069" xlink:type="simple"/></inline-formula>). We also consider the accuracy of the classification when a
                relative of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e070" xlink:type="simple"/></inline-formula> is present in sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e071" xlink:type="simple"/></inline-formula>.</p>
            <p>Our tests reveal a good separation of the distributions for positive (i.e., in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e072" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e073" xlink:type="simple"/></inline-formula>) and null (in neither) samples, suggesting that a suprising amount
                of information remains in pooled data. However, our results indicate that membership
                classification via Equation 2 is sensitive to the underlying assumptions such that
                the distribution for null samples does not follow <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e074" xlink:type="simple"/></inline-formula>, yielding misleadingly large <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e075" xlink:type="simple"/></inline-formula> for null samples. As a result, applying the method from <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> is tricky
                in practice since additional information is often necessary to set appropriate
                thresholds for significance. Finally, we conclude with a discussion of the
                implications of our findings, both in forensics as well as regarding identification
                of individuals contributing DNA in GWAS.</p>
        </sec>
        <sec id="s2">
            <title>Methods</title>
            <p>We explore the performance of the method described in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> both analytically and
                empirically. For the empirical studies, we attempt to classify sample genotypes
                derived from publicly available data sources in order to assess the chances that an
                individual is mistakenly classified into a group which does not contain his specific
                genotype.</p>
            <sec id="s2a">
                <title>Genotype data</title>
                <p>2287 genotypes were obtained from the Cancer Genomic Markers of Susceptibility
                    (CGEMS) breast cancer study. The samples were sourced as described in <xref ref-type="bibr" rid="pgen.1000668-Hunter1">[2]</xref>.
                    Briefly, the samples comprised 1145 breast cancer cases and a comparable number
                    (1142) of matched controls from the participants of the Nurses Health Study. All
                    the participants were American women of European descent. The samples were
                    genotyped against the Illumina 550K arrays, which assays over 550,000 SNPs
                    across the genome. To assess the genetic identity shared between samples, we
                    computed the fraction of SNPs with identical alleles for all possible pairs of
                    individuals; none exceeded <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e076" xlink:type="simple"/></inline-formula>.</p>
                <p>Additionally, 90 genotypes of American individuals of European descent (CEPH) and
                    90 genotypes of Yoruban individuals were obtained from the HapMap Project <xref ref-type="bibr" rid="pgen.1000668-The1">[3]</xref>. In
                    both cases, the 90 individuals were members of 30 family trios comprising two
                    unrelated parents and their offspring. SNPs in common with those assayed by the
                    CGEMS study and located on chromosomes 1–22 were kept in the analysis
                    (sex chromosomes were excluded since the CGEMS participants were uniformly
                    female); a total of 481,482 SNPs met these criteria.</p>
            </sec>
            <sec id="s2b">
                <title>Classification of genotypes</title>
                <p>The method as described in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> and summarized in the <xref ref-type="sec" rid="s1">Introduction</xref> was implemented using R <xref ref-type="bibr" rid="pgen.1000668-R1">[4]</xref>. Subsets of the data
                    described above were used to construct pools <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e077" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e078" xlink:type="simple"/></inline-formula>, using the remaining genotypes as test samples for which the
                    null hypothesis is true. A summary of the tests is provided in <xref ref-type="table" rid="pgen-1000668-t001">Table 1</xref>. In each test, SNPs
                    which did not achieve a minor allele frequency <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e079" xlink:type="simple"/></inline-formula> in both <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e080" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e081" xlink:type="simple"/></inline-formula> were excluded from the computation.</p>
                <table-wrap id="pgen-1000668-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pgen.1000668.t001</object-id><label>Table 1</label><caption>
                        <title>Summary of tests performed.</title>
                    </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pgen-1000668-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.t001" xlink:type="simple"/><table>
                        <colgroup span="1">
                            <col align="left" span="1"/>
                            <col align="center" span="1"/>
                            <col align="center" span="1"/>
                            <col align="center" span="1"/>
                        </colgroup>
                        <thead>
                            <tr>
                                <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e082" xlink:type="simple"/></inline-formula> individuals</td>
                                <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e083" xlink:type="simple"/></inline-formula> population</td>
                                <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e084" xlink:type="simple"/></inline-formula> population</td>
                                <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e085" xlink:type="simple"/></inline-formula> distribution</td>
                            </tr>
                        </thead>
                        <tbody>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">100 CGEMS cases not in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e086" xlink:type="simple"/></inline-formula></td>
                                <td align="left" colspan="1" rowspan="1">1042 CGEMS controls</td>
                                <td align="left" colspan="1" rowspan="1">1045 CGEMS cases</td>
                                <td align="left" colspan="1" rowspan="1">
                                    <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref>
                                </td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">100 CGEMS controls not in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e087" xlink:type="simple"/></inline-formula></td>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">90 HapMap CEPH</td>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">90 HapMap YRI</td>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">HapMap YRI mothers 16–30</td>
                                <td align="left" colspan="1" rowspan="1">HapMap YRI mothers 1–15 and fathers
                                    1–15</td>
                                <td align="left" colspan="1" rowspan="1">HapMap YRI children 1–15 and fathers
                                    16–30</td>
                                <td align="left" colspan="1" rowspan="1">
                                    <xref ref-type="fig" rid="pgen-1000668-g002">Figure 2</xref>
                                </td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">HapMap YRI children 16–30</td>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">HapMap CEPH mothers 16–30</td>
                                <td align="left" colspan="1" rowspan="1">HapMap CEPH mothers 1–15 and fathers
                                    1–15</td>
                                <td align="left" colspan="1" rowspan="1">HapMap CEPH children 1–15 and fathers
                                    16–30</td>
                                <td align="left" colspan="1" rowspan="1">
                                    <xref ref-type="fig" rid="pgen-1000668-g002">Figure 2</xref>
                                </td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">HapMap CEPH children 16–30</td>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1"/>
                            </tr>
                        </tbody>
                    </table></alternatives><table-wrap-foot>
                        <fn id="nt101">
                            <p>Summary of tests described. In the last four rows, the numbers refer
                                to the families in the HapMap YRI and CEPH populations, such that
                                child 1 is the offspring of mother 1 and father 1, et cetera.</p>
                        </fn>
                    </table-wrap-foot></table-wrap>
            </sec>
        </sec>
        <sec id="s3">
            <title>Results</title>
            <p>The assertion that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e088" xlink:type="simple"/></inline-formula> as given in Equation 2 follows a standard normal distribution
                under the null hypothesis that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e089" xlink:type="simple"/></inline-formula> is in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e090" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e091" xlink:type="simple"/></inline-formula> is based upon the assumptions that</p>
            <list list-type="order">
                <list-item>
                    <p><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e092" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e093" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e094" xlink:type="simple"/></inline-formula> are all samples of the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e095" xlink:type="simple"/></inline-formula>;</p>
                </list-item>
                <list-item>
                    <p><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e096" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e097" xlink:type="simple"/></inline-formula> are similarly sized samples; and</p>
                </list-item>
                <list-item>
                    <p>the SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e098" xlink:type="simple"/></inline-formula> used to compute <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e099" xlink:type="simple"/></inline-formula> are independent.</p>
                </list-item>
            </list>
            <p>We investigated the effect of deviation from these assumptions. A full treatment is
                presented in <xref ref-type="supplementary-material" rid="pgen.1000668.s001">Text
                S1</xref>, and we summarize the results briefly here. In the case where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e100" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e101" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e102" xlink:type="simple"/></inline-formula> are not samples of the same underlying population, the differences
                in the minor allele frequencies of the source populations dominate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e103" xlink:type="simple"/></inline-formula> such that deviations from zero are no longer attributable to the
                subtle influence of <italic>Y</italic>'s presence in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e104" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e105" xlink:type="simple"/></inline-formula>. In the case where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e106" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e107" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e108" xlink:type="simple"/></inline-formula> are samples of the same population but <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e109" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e110" xlink:type="simple"/></inline-formula> are of differing sizes, the larger one will be a more
                representative sample of the underlying population and hence closer, on average, to
                a future sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e111" xlink:type="simple"/></inline-formula>. Both violations of assumptions 1 and 2 above will lead to
                non-zero <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e112" xlink:type="simple"/></inline-formula> for null samples. Considering that the difference in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e113" xlink:type="simple"/></inline-formula> with and without the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e114" xlink:type="simple"/></inline-formula> assumption in Equation 2 is<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e115" xlink:type="simple"/><label>(3)</label></disp-formula>and that the number of SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e116" xlink:type="simple"/></inline-formula> is on the order of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e117" xlink:type="simple"/></inline-formula>, even slight deviations away from the assumed <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e118" xlink:type="simple"/></inline-formula> can have a pronounced effect when comparing <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e119" xlink:type="simple"/></inline-formula> against a standard normal as given by Equation 2. Equation 2 also
                presumes that the SNPs are independent, such that the variance of the mean of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e120" xlink:type="simple"/></inline-formula> can be estimated as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e121" xlink:type="simple"/></inline-formula> in the denominator of Equation 2; as shown in <xref ref-type="supplementary-material" rid="pgen.1000668.s001">Text S1</xref>, even a
                slight average correlation amongst the SNPs (due, for instance, to linkage
                disequilibrium) will cause the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e122" xlink:type="simple"/></inline-formula> in practice to be much wider than that assumed in Equation 2, once
                again owing to the large number of SNPs considered. Because it appears that slight
                deviations from the assumptions outlined above may have a strong effect on the
                obtained <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e123" xlink:type="simple"/></inline-formula> values, the false-positive rate of the method proposed in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> may in
                practice be considerably higher than the nominal false-positive rate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e124" xlink:type="simple"/></inline-formula> given by quantiles of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e125" xlink:type="simple"/></inline-formula>.</p>
            <sec id="s3a">
                <title>Empirical tests</title>
                <p>To explore the performance of the method in realistic situations, we carried out
                    the computations described by Equations 1,2 for various <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e126" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e127" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e128" xlink:type="simple"/></inline-formula> as described in <xref ref-type="table" rid="pgen-1000668-t001">Table 1</xref>. Distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e129" xlink:type="simple"/></inline-formula> for each of the tests described in <xref ref-type="table" rid="pgen-1000668-t001">Table 1</xref> are shown in the corresponding
                    figures listed in the table. We find that while the distributions of
                        in-<italic>F</italic>, in-<italic>G</italic> and in-neither values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e130" xlink:type="simple"/></inline-formula> are distinct, calibrating thresholds for classifying an
                    unknown sample is difficult without additional information. This is due to the
                    fact that the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e131" xlink:type="simple"/></inline-formula> for null samples deviates strongly from a standard normal in
                    practice.</p>
                <p>We begin first by considering a best-case situation in which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e132" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e133" xlink:type="simple"/></inline-formula> are both large samples of the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e134" xlink:type="simple"/></inline-formula>, and the samples to be classified are also from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e135" xlink:type="simple"/></inline-formula>. Here, we randomly select 100 cases and 100 controls from
                    CGEMS to form an out-of-pool test sample set comprising 200 individuals, using
                    the remaining 1045 CGEMS cases and 1042 CGEMS controls as pools <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e136" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e137" xlink:type="simple"/></inline-formula>, respectively. (Several such random subsets were created; the
                    results were consistent and hence we present a single representative one.) <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e138" xlink:type="simple"/></inline-formula> (Equation 1, 2) was computed for all the samples and compared
                    to a standard normal (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e139" xlink:type="simple"/></inline-formula> yields a nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e140" xlink:type="simple"/></inline-formula> (<italic>p</italic>-value) of 0.05 and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e141" xlink:type="simple"/></inline-formula> yields a nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e142" xlink:type="simple"/></inline-formula>). The sensitivity and specificities obtained are given in
                        <xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref>.</p>
                <table-wrap id="pgen-1000668-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pgen.1000668.t002</object-id><label>Table 2</label><caption>
                        <title>Empirical sensitivity and specificity for the tests shown in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref> assuming <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e143" xlink:type="simple"/></inline-formula>.</title>
                    </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pgen-1000668-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.t002" xlink:type="simple"/><table>
                        <colgroup span="1">
                            <col align="left" span="1"/>
                            <col align="center" span="1"/>
                            <col align="center" span="1"/>
                            <col align="center" span="1"/>
                            <col align="center" span="1"/>
                        </colgroup>
                        <thead>
                            <tr>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="2" rowspan="1">481,382 SNPs</td>
                                <td align="left" colspan="2" rowspan="1">50,000 SNPs</td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1"/>
                                <td align="left" colspan="1" rowspan="1">
                                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e144" xlink:type="simple"/></inline-formula>
                                </td>
                                <td align="left" colspan="1" rowspan="1">
                                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e145" xlink:type="simple"/></inline-formula>
                                </td>
                                <td align="left" colspan="1" rowspan="1">
                                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e146" xlink:type="simple"/></inline-formula>
                                </td>
                                <td align="left" colspan="1" rowspan="1">
                                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e147" xlink:type="simple"/></inline-formula>
                                </td>
                            </tr>
                        </thead>
                        <tbody>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">Sensitivity</td>
                                <td align="left" colspan="1" rowspan="1">99.8%</td>
                                <td align="left" colspan="1" rowspan="1">97.5%</td>
                                <td align="left" colspan="1" rowspan="1">96.3%</td>
                                <td align="left" colspan="1" rowspan="1">36.3%</td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">Specificity, 200 CGEMS</td>
                                <td align="left" colspan="1" rowspan="1">31.0%</td>
                                <td align="left" colspan="1" rowspan="1">70.5%</td>
                                <td align="left" colspan="1" rowspan="1">79.0%</td>
                                <td align="left" colspan="1" rowspan="1">99.5%</td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">Specificity, 90 HapMap CEPH</td>
                                <td align="left" colspan="1" rowspan="1">5.5%</td>
                                <td align="left" colspan="1" rowspan="1">27.7%</td>
                                <td align="left" colspan="1" rowspan="1">45.5%</td>
                                <td align="left" colspan="1" rowspan="1">100.0%</td>
                            </tr>
                            <tr>
                                <td align="left" colspan="1" rowspan="1">Specificity, 90 HapMap YRI</td>
                                <td align="left" colspan="1" rowspan="1">0.0%</td>
                                <td align="left" colspan="1" rowspan="1">0.0%</td>
                                <td align="left" colspan="1" rowspan="1">4.4%</td>
                                <td align="left" colspan="1" rowspan="1">97.7%</td>
                            </tr>
                        </tbody>
                    </table></alternatives><table-wrap-foot>
                        <fn id="nt102">
                            <p>Classification results are given for two different nominal false
                                positive rates <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e148" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e149" xlink:type="simple"/></inline-formula>.</p>
                        </fn>
                    </table-wrap-foot></table-wrap>
                <p>Distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e150" xlink:type="simple"/></inline-formula> values for all three groups of CGEMS samples are shown in
                        <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A</xref>. Notably, the
                    distributions of in-<italic>F</italic>, in-<italic>G</italic>, and in-neither
                    samples are all quite distinct. For the positive samples (those in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e151" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e152" xlink:type="simple"/></inline-formula>), the classifier performs fairly well, correctly classifying
                    2083 samples (and calling 4 as in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e153" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e154" xlink:type="simple"/></inline-formula>). However, of the 200 test samples which were in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e155" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e156" xlink:type="simple"/></inline-formula>, only 62 have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e157" xlink:type="simple"/></inline-formula>; the rate of false positives is thus 69% if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e158" xlink:type="simple"/></inline-formula> is used as an indicator of group membership under the
                    assumptions in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> at the nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e159" xlink:type="simple"/></inline-formula> (see <xref ref-type="table" rid="pgen-1000668-t002">Table
                    2</xref>).</p>
                <fig id="pgen-1000668-g001" position="float">
                    <object-id pub-id-type="doi">10.1371/journal.pgen.1000668.g001</object-id>
                    <label>Figure 1</label>
                    <caption>
                        <title>Comparison of <italic>T</italic> distributions.</title>
                        <p>Comparison of <italic>T</italic> distributions for true positive and null
                            samples versus putative null distribution, starting with 481,382 SNPs in
                            (A,B) and 50,000 SNPs in (C,D). In all plots, true positive <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e160" xlink:type="simple"/></inline-formula> (1042 CGEMS controls) is shown as a solid green curve,
                            true positive <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e161" xlink:type="simple"/></inline-formula> (1045 CGEMS cases) is shown as a solid red curve, and
                            the putative null <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e162" xlink:type="simple"/></inline-formula> is given as a thin grey curve. The dark and light grey
                            regions represent the areas for which the null hypothesis would be
                            accepted at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e163" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e164" xlink:type="simple"/></inline-formula>, respectively. In plots (A,C), CGEMS test samples in
                            neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e165" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e166" xlink:type="simple"/></inline-formula> (100 CGEMS cases and 100 CGEMS controls) are given by
                            a heavy black curve. The CGEMS case and CGEMS control distributions
                            within this group are shown as dashed red and green lines, respectively.
                            In plots (B,D), <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e167" xlink:type="simple"/></inline-formula> distributions are given for HapMap CEPHs (cyan) and
                            YRIs (blue). Vertical lines mark the 0.05 and 0.95 quantiles of the
                            negative CGEMS samples (black), HapMap CEPHs (cyan), and HapMap YRIs
                            (blue).</p>
                    </caption>
                    <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.g001" xlink:type="simple"/>
                </fig>
                <p>Next, we consider a less ideal, yet probable, case in which the null samples are
                    not from the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e168" xlink:type="simple"/></inline-formula>. Here, we leave <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e169" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e170" xlink:type="simple"/></inline-formula> as above, and apply Equation 1, 2 to 90 HapMap American
                    individuals of European descent (whom, one might assume, would be relatively
                    similar to the Americans of European descent comprising groups <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e171" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e172" xlink:type="simple"/></inline-formula>). A plot of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e173" xlink:type="simple"/></inline-formula> value distributions is given in cyan in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1B</xref>. Again, there is little overlap
                    with the true positive distributions, but when comparing the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e174" xlink:type="simple"/></inline-formula> values against <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e175" xlink:type="simple"/></inline-formula>, the sensitivity is quite low (see <xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref>). A yet more extreme case, in which
                    90 HapMap Yoruban individuals were classified with respect to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e176" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e177" xlink:type="simple"/></inline-formula>, results in a distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e178" xlink:type="simple"/></inline-formula> values that overlaps with the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e179" xlink:type="simple"/></inline-formula> values from group <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e180" xlink:type="simple"/></inline-formula> (<xref ref-type="fig" rid="pgen-1000668-g001">Figure
                    1B</xref>, blue curve) and exceedingly low specificity (<xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref>). We thus see in practice a strong
                    dependence of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e181" xlink:type="simple"/></inline-formula> upon the assumption that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e182" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e183" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e184" xlink:type="simple"/></inline-formula> are samples of the same population.</p>
                <p>The reason for the high false-positive rates in practice despite the stringent
                    nominal false positive rate is clear from the plots <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1B:</xref> namely, it can be seen that
                    the putative null distribution (light grey line, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e185" xlink:type="simple"/></inline-formula>, cf Equation 2) does not correspond to the observed
                    distribution for samples for which the null hypothesis is correct, with
                    differences in both the location and width.</p>
                <p>The overall shift in the location of the distributions is a result of violations
                    of the assumption that each sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e186" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e187" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e188" xlink:type="simple"/></inline-formula> are drawn on from the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e189" xlink:type="simple"/></inline-formula>. The magnitude of this effect is derived in <xref ref-type="supplementary-material" rid="pgen.1000668.s001">Text S1</xref> as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e190" xlink:type="simple"/></inline-formula>, where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e191" xlink:type="simple"/></inline-formula> are the MAFs of the population from which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e192" xlink:type="simple"/></inline-formula> is drawn (hence the different rightward shifts of the CGEMS,
                    CEPH, and YRI distributions). Because of the large number of SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e193" xlink:type="simple"/></inline-formula> in Equation 2, small deviations from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e194" xlink:type="simple"/></inline-formula> are magnified; even ancestrally similar populations, such as
                    the 200 CGEMS test samples and the HapMap CEPHS, have different distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e195" xlink:type="simple"/></inline-formula>.</p>
                <p>The broadening of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e196" xlink:type="simple"/></inline-formula> distribution is a result of correlation between SNPs. In
                    Equation 2, it is assumed that the variance of the mean of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e197" xlink:type="simple"/></inline-formula> be estimable by the mean of the variance, ie, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e198" xlink:type="simple"/></inline-formula>, which is true for independent SNPs. However, if there exists
                    average correlation <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e199" xlink:type="simple"/></inline-formula> amongst the SNPs (due to linkage disequilibrium),<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e200" xlink:type="simple"/><label>(4)</label></disp-formula>which can be quite large even for small average correlation <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e201" xlink:type="simple"/></inline-formula> due to the high number <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e202" xlink:type="simple"/></inline-formula> of SNPs. The result of increased LD is a broader distribution
                    of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e203" xlink:type="simple"/></inline-formula> values, as observed in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1B:</xref> we observe a narrower
                    distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e204" xlink:type="simple"/></inline-formula> for the HapMap YRI samples versus the Caucasian CGEMS
                    participants and HapMap CEPHs (the Yoruban individuals, who come from an older
                    population, have lower average LD).</p>
                <p>The effect of LD on the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e205" xlink:type="simple"/></inline-formula> may be countered by selecting fewer SNPs; the results of this
                    approach can be seen in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1C
                        and 1D</xref> and in <xref ref-type="table" rid="pgen-1000668-t002">Table
                    2</xref>. Here, 50,000 SNPs were selected, uniformly distributed across of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e206" xlink:type="simple"/></inline-formula> SNPs used in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1B</xref>. 50,000 SNPs was shown in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> to be a reasonable
                    lower bound to detect at nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e207" xlink:type="simple"/></inline-formula> one individual amongst 1000, which is the concentration of
                    true positive individuals in this test. As is clear from <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref>, reducing the number of SNPs narrows
                    the distributions considerably, yet at the same time brings them closer together
                    such that the crisp separation previously obtained is reduced. Using this
                    method, we see that the 200 CGEMS samples now have a distribution closer to that
                    of the putative null <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e208" xlink:type="simple"/></inline-formula> such that using a threshold of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e209" xlink:type="simple"/></inline-formula> yields an improved—yet still larger than
                    nominal—21% false-positive rate while maintaining a high
                    96.3% true positive rate. However, the misclassification rate is
                    still over 50% for both HapMap samples, and improving these values
                    requires compromising the sensitivity, a direct result of the overlapping <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e210" xlink:type="simple"/></inline-formula> distributions for the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e211" xlink:type="simple"/></inline-formula> and HapMap samples.</p>
                <p>Despite the low sensitivities obtained in our tests, it is apparent from <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref> that the true
                    positive individuals have a significantly different distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e212" xlink:type="simple"/></inline-formula> values than do the null samples, such that if appropriate
                    thresholds were selected the classification could be improved (note that in
                    practice, the distributions of the true positive individuals are unknown, since
                    reconstructing them requires full genotypes, not just the MAFs, of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e213" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e214" xlink:type="simple"/></inline-formula>). One simple apprach, motivated by the observed separation of
                    distributions in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref>,
                    would be to collect a set of presumed-null genotypes from which to estimate the
                    null <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e215" xlink:type="simple"/></inline-formula> distribution. Consider a situation in which we have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e216" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e217" xlink:type="simple"/></inline-formula>, along with an individual <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e218" xlink:type="simple"/></inline-formula> who is one of the 200 CGEMS samples not in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e219" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e220" xlink:type="simple"/></inline-formula>, but no other genotypes. We might reasonably turn to publicly
                    available HapMap genotypes as a group from which to construct an empirical null
                    distribution for setting thresholds. The lines in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1C</xref> depict this case. Using the
                    0.05 and 0.95 quantiles obtained from the HapMap CEPH <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e221" xlink:type="simple"/></inline-formula> distribution (cyan bars) as thresholds improves the accuracy
                    relative to using <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e222" xlink:type="simple"/></inline-formula> quantiles, but still incorrectly classifies half of the 200
                    CGEMS samples; the false positive rate is yet greater (and the true-positive
                    rate smaller) when using the HapMap YRI quantiles (blue bars). Likewise, roughly
                    a quarter of the HapMap CEPHs and the majority of HapMap YRIs lie outside the
                    thresholds set from the 200 CGEMS samples in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1B and 1D</xref>.</p>
                <p>These examples, as well as the analytical results described in <xref ref-type="supplementary-material" rid="pgen.1000668.s001">Text S1</xref>,
                    show that deviations from the assumptions that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e223" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e224" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e225" xlink:type="simple"/></inline-formula> are i.i.d. samples of the same population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e226" xlink:type="simple"/></inline-formula> can produce misleadingly large values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e227" xlink:type="simple"/></inline-formula>. While Equations 1, 2 produce good separation of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e228" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e229" xlink:type="simple"/></inline-formula> and null sample distributions, appropriately calibrating the
                    thresholds for classification is difficult in practice.</p>
                <sec id="s3a1">
                    <title>Classification of relatives</title>
                    <p>We briefly consider the classification of individuals who are relatives of
                        true positives. This can be investigated by using HapMap trios, since we can
                        reasonably expect that the children will bear a greater resemblance to their
                        parents than their parents do to one another. Recalling that the HapMap
                        pools consist of thirty individual mother-father-offspring pedigrees, we
                        construct pools as follows:</p>
                    <list list-type="bullet">
                        <list-item>
                            <p><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e230" xlink:type="simple"/></inline-formula> = Mothers from
                                pedigrees 1–15 and fathers from pedigrees
                            1–15</p>
                        </list-item>
                        <list-item>
                            <p><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e231" xlink:type="simple"/></inline-formula> = Children from
                                pedigrees 1–15 and fathers from pedigrees
                            16–30</p>
                        </list-item>
                    </list>
                    <p>and then compute <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e232" xlink:type="simple"/></inline-formula> for mothers and children from pedigrees 16–30
                        using the same SNP criteria as before. The results of these tests for both
                        the CEPH and YRI pedigrees, given in <xref ref-type="fig" rid="pgen-1000668-g002">Figure 2</xref>, are as expected, with the
                        children having a significantly higher distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e233" xlink:type="simple"/></inline-formula> than the mothers; the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e234" xlink:type="simple"/></inline-formula> values for all the children were so large that
                        <italic>p</italic>-values <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e235" xlink:type="simple"/></inline-formula> were obtained when comparing to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e236" xlink:type="simple"/></inline-formula>. By contrast, 5/15 of the YRI mothers from pedigrees
                        16–30 and 10/15 of the CEPH mothers from pedigrees 16–30
                        yielded <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e237" xlink:type="simple"/></inline-formula> (with distributions roughly centered about <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e238" xlink:type="simple"/></inline-formula>). The wider distribution amongst the CEPHS again reflects
                        the effect of LD. In <xref ref-type="fig" rid="pgen-1000668-g002">Figure
                        2</xref> we can see that the method has the power to resolve three groups:
                        those in a group, those related to members of a group, and those who are
                        neither. Note, however, that without having the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e239" xlink:type="simple"/></inline-formula> for true positives (which necessitates knowing the
                        genotypes of true positives), it is not clear that setting a threshold to
                        distinguish between true positives and their relatives is possible.</p>
                    <fig id="pgen-1000668-g002" position="float">
                        <object-id pub-id-type="doi">10.1371/journal.pgen.1000668.g002</object-id>
                        <label>Figure 2</label>
                        <caption>
                            <title>Distribution of <italic>T</italic>.</title>
                            <p>Distributions of <italic>T</italic> for out-of-group samples who are
                                related (red line) and unrelated (blue line) to individuals in
                                    <italic>G</italic> for HapMap YRI (A) and HapMap CEPH (B)
                                populations. (C) and (D) show the same distributions as (A) and (B)
                                respectively, with the addition (green line) of individuals who are
                                in <italic>G</italic> and unrelated to <italic>F</italic> (i.e.,
                                true positives). Dashed black lines indicate the <italic>T</italic>
                                significance thresholds of ±1.64 at nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e240" xlink:type="simple"/></inline-formula>.</p>
                        </caption>
                        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.g002" xlink:type="simple"/>
                    </fig>
                </sec>
                <sec id="s3a2">
                    <title>Positive predictive value of the method</title>
                    <p>The effect of the modest specificity—even in the best of cases
                        described above—on the posterior probability that the individual <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e241" xlink:type="simple"/></inline-formula> is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e242" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e243" xlink:type="simple"/></inline-formula> is considerable, given that the prior probability is
                        likely to be relatively small in most applications of this method. Let us
                        consider the positive predictive value (PPV), which quantifies the post-test
                        probability that an individual <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e244" xlink:type="simple"/></inline-formula> with a positive result (i.e., significant <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e245" xlink:type="simple"/></inline-formula>) is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e246" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e247" xlink:type="simple"/></inline-formula>. This probability depends on the prior probability that
                        the individual is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e248" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e249" xlink:type="simple"/></inline-formula>, i.e., on the prevalence of being a member of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e250" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e251" xlink:type="simple"/></inline-formula>. PPV follows directly from Bayes' theorem, and is
                        defined as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e252" xlink:type="simple"/><label>(5)</label></disp-formula>where the PPV is the posterior probability that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e253" xlink:type="simple"/></inline-formula> is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e254" xlink:type="simple"/></inline-formula> given a prior probability of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e255" xlink:type="simple"/></inline-formula>. We can write this equivalently in terms of the positive
                        likelihood ratio <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e256" xlink:type="simple"/></inline-formula>,<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e257" xlink:type="simple"/><label>(6)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.e258" xlink:type="simple"/><label>(7)</label></disp-formula>A plot of PPV vs. prevalence is given in <xref ref-type="fig" rid="pgen-1000668-g003">Figure 3</xref>. Even with the best sensitivity
                        (96.3%) and specificity (79%) obtained in <xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref>—that
                        in which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e259" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e260" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e261" xlink:type="simple"/></inline-formula> were strictly drawn on the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e262" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e263" xlink:type="simple"/></inline-formula> SNPs were used, and a nominal <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e264" xlink:type="simple"/></inline-formula> was used as a threshold—the prior probability
                        (prevalence) of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e265" xlink:type="simple"/></inline-formula> being in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e266" xlink:type="simple"/></inline-formula> needs to exceed 66% in order to achieve a
                        90% post-test probability that the subject is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e267" xlink:type="simple"/></inline-formula>. For a PPV of 99%, the prior probability needs
                        to exceed 72% for any specificity under 95%, assuming
                        the observed sensitivity of 99%. The low specificities obtained
                        in practice thus require a strong prior belief that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e268" xlink:type="simple"/></inline-formula> is in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e269" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e270" xlink:type="simple"/></inline-formula>.</p>
                    <fig id="pgen-1000668-g003" position="float">
                        <object-id pub-id-type="doi">10.1371/journal.pgen.1000668.g003</object-id>
                        <label>Figure 3</label>
                        <caption>
                            <title>Positive predictive value (PPV) as a function of prevalence and
                                specificity given 99% sensitivity.</title>
                            <p>In (A), PPV is shown on the <italic>y</italic> axis and color
                                corresponds to specificity. The black curve depicts the
                                87% sensitivity line—the best sensitivity
                                obtained in the empirical tests in <xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref>. In (B), PPV is shown by
                                color, and the <italic>y</italic> axis corresponds to
                            specificity.</p>
                        </caption>
                        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.g003" xlink:type="simple"/>
                    </fig>
                    <p>The difference between the empirical false-positive rate and the nominal
                        false-positive rate based on the standard normal has a strong effect on the
                        posterior probabilities. Consider that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e271" xlink:type="simple"/></inline-formula> at 87% specificity and 99%
                        sensitivity is 7.6, versus 990000 if the nominal false-positive rate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e272" xlink:type="simple"/></inline-formula> were correct. For prior probability of 1/1000, the first
                        case yields a posterior probability of 1.1/1000, while the second yields a
                        posterior probability of 998/1000. These differences, which are difficult to
                        measure without additional, well-matched null sample genotypes and which
                        depend strongly on the degree to which the assumptions underlying the method
                        are met (consider the differences between the CGEMS and HapMap CEPH
                        specificities in <xref ref-type="table" rid="pgen-1000668-t002">Table
                        2</xref>), pose a severe limitation on the utility of using Equations 1,2 to
                        resolve <italic>Y</italic>'s membership in samples <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e273" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e274" xlink:type="simple"/></inline-formula>.</p>
                </sec>
            </sec>
        </sec>
        <sec id="s4">
            <title>Discussion</title>
            <p>In this work, we have further characterized and tested the genetic distance metric
                initially proposed in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>. This metric, summarized here by Equations 1,2,
                quantifies the distance of an individual genotype <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e275" xlink:type="simple"/></inline-formula> with respect to two samples <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e276" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e277" xlink:type="simple"/></inline-formula> using the marginal minor allele frequencies <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e278" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e279" xlink:type="simple"/></inline-formula> of the two samples and the genotype <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e280" xlink:type="simple"/></inline-formula>. The article <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> proposes to use this metric to infer the presence
                of the individual in one of the two samples, and the authors demonstrate the utility
                of their classifier on known positive samples (i.e., samples which are in either <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e281" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e282" xlink:type="simple"/></inline-formula>) showing that in this situation their method yields
                classifications of high sensitivity. Our investigations confirm that the sensitivity
                is quite high (correctly classifying true positives into groups <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e283" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e284" xlink:type="simple"/></inline-formula>) and that in-<italic>F</italic>, in-<italic>G</italic>, and null
                samples have distinct distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e285" xlink:type="simple"/></inline-formula> values. However, we also find that the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e286" xlink:type="simple"/></inline-formula> for null samples does not follow the presumed standard normal, and
                thus the specificity is considerably less than that predicted by the quantiles of
                the putative null distribution <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e287" xlink:type="simple"/></inline-formula>. Calibrating a more accurate set of thresholds is difficult in
                practice, limiting the utility of Equations 1, 2 to positively identify
                <italic>Y</italic>'s presence in samples <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e288" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e289" xlink:type="simple"/></inline-formula>.</p>
            <p>In this work we have shown that high <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e290" xlink:type="simple"/></inline-formula> values, significant when compared against <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e291" xlink:type="simple"/></inline-formula>, may be obtained for samples that are in neither of the pools due
                to violations of the assumptions that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e292" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e293" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e294" xlink:type="simple"/></inline-formula> are all samples of the same underlying population; that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e295" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e296" xlink:type="simple"/></inline-formula> are similarly sized samples; and that the SNPs <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e297" xlink:type="simple"/></inline-formula> used to compute <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e298" xlink:type="simple"/></inline-formula> are independent. The high false positive rates in <xref ref-type="table" rid="pgen-1000668-t002">Table 2</xref> result from deviations
                of the first and third assumptions. These assumptions are difficult to meet; for
                instance, HapMap CEPH and CGEMS samples are sufficiently dissimilar that the HapMap
                CEPH samples exhibit greater deviation from violations of the first assumption,
                despite the fact that both samples are Americans of European descent. Additionally,
                the conclusion that high <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e299" xlink:type="simple"/></inline-formula> values result from <italic>Y</italic>'s presence in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e300" xlink:type="simple"/></inline-formula> relies upon the questionable assumption that individuals in
                neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e301" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e302" xlink:type="simple"/></inline-formula> will be equidistant from both, resulting in false positives for
                relatives of true positive individuals, even when the other assumptions are met.</p>
            <p>The low false positive rate in practice, resulting from the difficulty in accurately
                calibrating the significance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e303" xlink:type="simple"/></inline-formula>, results in a likelihood ratio (and hence post-test probability)
                that is also low. When the prior probability of <italic>Y</italic>'s
                presence in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e304" xlink:type="simple"/></inline-formula> or <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e305" xlink:type="simple"/></inline-formula> is modest, strong evidence (i.e., high specificity) is needed to
                outweigh this prior, which was not achieved in our tests. On the other hand, when
                samples were known <italic>a priori</italic> to be in one of the groups
                    <italic>F</italic>/<italic>G</italic>, Equations 1,2 correctly identify the
                sample of which the individual is part.</p>
            <p>These findings have implications both in forensics (for which the method <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> was
                proposed) and GWAS privacy (which has become a topic of considerable interest in
                light of <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>). We briefly consider each:</p>
            <sec id="s4a">
                <title>Forensics implications</title>
                <p>The stated purpose <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> of the method—namely, to positively
                    identify the presence of a particular individual in a mixed pool of genetic data
                    of unknown size and composition—is difficult to achieve. In this
                    scenario, we have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e306" xlink:type="simple"/></inline-formula> (from forensic evidence) and a suspect genotype <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e307" xlink:type="simple"/></inline-formula>. To apply the method, we would need 1) to assume that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e308" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e309" xlink:type="simple"/></inline-formula> are indeed i.i.d. samples of the same population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e310" xlink:type="simple"/></inline-formula>; 2) to obtain a sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e311" xlink:type="simple"/></inline-formula> which is <italic>also</italic> a sample of the underlying
                    population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e312" xlink:type="simple"/></inline-formula>, well-matched in size and composition to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e313" xlink:type="simple"/></inline-formula>; 3) to obtain an estimate of the sample size of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e314" xlink:type="simple"/></inline-formula> such that sample-size effects can be appropriately discounted
                    (see <xref ref-type="supplementary-material" rid="pgen.1000668.s001">Text
                    S1</xref>); and 4) to assume that the <italic>p</italic>-values at the selected
                    classification thresholds are accurate. The high false-positive rates which
                    result from even small violations of these criteria make it exceedingly likely
                    that an innocent party will be wrongly identified as suspicious; it is even more
                    likely for a relative of an individual whose DNA is present in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e315" xlink:type="simple"/></inline-formula>.</p>
            </sec>
            <sec id="s4b">
                <title>GWAS privacy implications</title>
                <p>Here the scenario of concern is that of a malefactor with the genotype of one (or
                    many) individuals, and access to the case and control MAFs from published
                    studies; could the malefactor use this method to discern whether one of the
                    genotypes in his possession belongs to a GWAS subject? In this case, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e316" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e317" xlink:type="simple"/></inline-formula> are known to be samples of the same underlying population <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e318" xlink:type="simple"/></inline-formula> (due to the careful matching in GWAS), and their sample sizes
                    are large and known. However, the malefactor still needs 1) to assure that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e319" xlink:type="simple"/></inline-formula> is a member of this population as well (as shown by the poor
                    results when HapMap samples were classified using CGEMS MAFs) and 2) to assume
                    that the <italic>p</italic>-values at the selected classification thresholds are
                    accurate. Additionally, the prior probability that any of the genotypes in the
                    malefactor's possession comes from a GWAS subject is likely to be quite
                    small, since GWAS samples are a tiny fraction of the population from which they
                    are drawn. Even if the malefactor were able to narrow down the prior probability
                    to one in three, a sensitivity of 99% and a specificity of
                    95% is needed to obtain a 90% posterior probability that
                    the individual is truly a participant.</p>
                <p>On the other hand, if the malefactor <italic>does</italic> have prior knowledge
                    that the individual <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e320" xlink:type="simple"/></inline-formula> participated in a certain GWAS but does not know
                    <italic>Y</italic>'s case status, Equations 1, 2 permit the malefactor
                    to discover with high accuracy which group <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e321" xlink:type="simple"/></inline-formula> was in. Additionally, in the case of <italic>a priori</italic>
                    knowledge, the participant's genotype is not strictly necessary, since
                    a relative's DNA will yield a large <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e322" xlink:type="simple"/></inline-formula> score that falls on the appropriate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e323" xlink:type="simple"/></inline-formula> side of null.</p>
                <p>Despite these limitations, we observe that the distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e324" xlink:type="simple"/></inline-formula> values for in-<italic>F</italic>, in-<italic>G</italic>, and
                    null samples separate strongly, suggesting that each individual contributes a
                    pattern of allele frequencies that remains in the pooled data. While calibrating
                    thresholds to distinguish these distributions without additional information is
                    not presently possible, the potential for more sophisticated methods to overcome
                    these barriers cannot be discounted and presents an avenue for future work.</p>
                <p>Moreover, we believe that the distance metric (Equations 1, 2) as presented may
                    still have forensic and research utility. It is clear from both our studies and
                    the original paper <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref> that the sensitivity is quite high; in the
                    (rare) case that a sample has an insignificant <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e325" xlink:type="simple"/></inline-formula>, it is very likely that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e326" xlink:type="simple"/></inline-formula> is in neither <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e327" xlink:type="simple"/></inline-formula> nor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e328" xlink:type="simple"/></inline-formula>. We can also see that genetically distinct groups have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e329" xlink:type="simple"/></inline-formula> distributions with little overlap (<xref ref-type="fig" rid="pgen-1000668-g001">Figure 1</xref>), and so it may be worth
                    investigating the utility of Equations 1,2 for ancestry inference.</p>
                <p>On this note, let us once more consider the quantity which Equation 1 measures,
                    namely the distance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e330" xlink:type="simple"/></inline-formula> from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e331" xlink:type="simple"/></inline-formula> relative to the distance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e332" xlink:type="simple"/></inline-formula> from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e333" xlink:type="simple"/></inline-formula>. Referring to <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1C</xref>, we can see that samples <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e334" xlink:type="simple"/></inline-formula> which are more like those in sample <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e335" xlink:type="simple"/></inline-formula> have a distribution that lies to the right of samples which
                    are more similar to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e336" xlink:type="simple"/></inline-formula>, as expected; that is, in <xref ref-type="fig" rid="pgen-1000668-g001">Figure 1A and 1C</xref>, the distribution of null
                    (not in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e337" xlink:type="simple"/></inline-formula>) CGEMS cases (dashed red line) is shifted to the right with
                    respect to the distribution of null CGEMS controls, as might be expected from
                    Equation 1, i.e., the CGEMS case <italic>Y</italic>s are closer to CGEMS case
                        <italic>G</italic>s than are the CGEMS control <italic>Y</italic>s. Although
                    this difference is not statistically significant, one could imagine that it may
                    be possible to select SNPs for which the shift is significant, i.e., a selection
                    of SNPs for which unknown cases are statistically more likely to be closer (via
                    Equation 1) to the cases in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e338" xlink:type="simple"/></inline-formula> and unknown controls are statistically more likely to be
                    closer to the controls in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e339" xlink:type="simple"/></inline-formula>. In this case, a subset of SNPs known to be associated with
                    disease may potentially be used with Equations 1, 2 to predict the case status
                    of new individuals; conversely, finding a subset of SNPs which produce
                    significant separations of the test samples may be indicative of a group of SNPs
                    which play a role in disease. Because this type of application would use fewer
                    SNPs and would involve the comparison of two distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e340" xlink:type="simple"/></inline-formula> (cases <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e341" xlink:type="simple"/></inline-formula> vs. controls <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pgen.1000668.e342" xlink:type="simple"/></inline-formula>), it may be possible to circumvent some of the problems
                    stemming from the unknown width and location of the null distribution described
                    above; still, much work is needed to investigate this possible application. If
                    successful, the metric proposed in <xref ref-type="bibr" rid="pgen.1000668-Homer1">[1]</xref>, while failing to
                    function as a tool to positively identify the presence of a specific
                    individual's DNA in a finite genetic sample, may if refined be a useful
                    tool in the analysis of GWAS data.</p>
            </sec>
        </sec>
        <sec id="s5">
            <title>Supporting Information</title>
            <supplementary-material id="pgen.1000668.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pgen.1000668.s001" xlink:type="simple">
                <label>Text S1</label>
                <caption>
                    <p><italic>D<sub>i</sub></italic> and <italic>T</italic> under the null
                        hypothesis.</p>
                    <p>(0.22 MB PDF)</p>
                </caption>
            </supplementary-material>
        </sec>
    </body>
    <back>
        <ref-list>
            <title>References</title>
            <ref id="pgen.1000668-Homer1">
                <label>1</label>
                <element-citation publication-type="journal" xlink:type="simple">
                    <person-group person-group-type="author">
                        <name name-style="western">
                            <surname>Homer</surname>
                            <given-names>N</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Szelinger</surname>
                            <given-names>S</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Redman</surname>
                            <given-names>M</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Duggan</surname>
                            <given-names>D</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Tembe</surname>
                            <given-names>W</given-names>
                        </name>
                        <etal/>
                    </person-group>
                    <year>2008</year>
                    <article-title>Resolving individuals contributing trace amounts of DNA to highly
                        complex mixtures using high-density SNP genotyping microarrays.</article-title>
                    <source>PLoS Genet</source>
                    <volume>4</volume>
                    <fpage>e1000167</fpage>
                    <comment>doi:<ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1371/journal.pgen.1000167" xlink:type="simple">10.1371/journal.pgen.1000167</ext-link></comment>
                </element-citation>
            </ref>
            <ref id="pgen.1000668-Hunter1">
                <label>2</label>
                <element-citation publication-type="journal" xlink:type="simple">
                    <person-group person-group-type="author">
                        <name name-style="western">
                            <surname>Hunter</surname>
                            <given-names>DJ</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Kraft</surname>
                            <given-names>P</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Jacobs</surname>
                            <given-names>KB</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Cox</surname>
                            <given-names>DG</given-names>
                        </name>
                        <name name-style="western">
                            <surname>Yeager</surname>
                            <given-names>M</given-names>
                        </name>
                        <etal/>
                    </person-group>
                    <article-title>A genome-wide association study identifies alleles in FGFR2
                        associated with risk of sporadic postmenopausal breast cancer.</article-title>
                    <source>Nat Genet</source>
                    <volume>39</volume>
                    <fpage>870</fpage>
                    <lpage>874</lpage>
                </element-citation>
            </ref>
            <ref id="pgen.1000668-The1">
                <label>3</label>
                <element-citation publication-type="journal" xlink:type="simple">
                    <collab xlink:type="simple">The International HapMap Consortium</collab>
                    <article-title>The International HapMap Project.</article-title>
                    <source>Nature</source>
                    <volume>426</volume>
                    <fpage>789</fpage>
                    <lpage>796</lpage>
                </element-citation>
            </ref>
            <ref id="pgen.1000668-R1">
                <label>4</label>
                <element-citation publication-type="other" xlink:type="simple">
                    <collab xlink:type="simple">R Development Core Team</collab>
                    <year>2004</year>
                    <source>A language and environment for statistical computing</source>
                    <publisher-loc>Vienna, Austria</publisher-loc>
                </element-citation>
            </ref>
        </ref-list>
        
    </back>
</article>