<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
  <front>
    <journal-meta><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id><journal-id journal-id-type="pmc">ploscomp</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS Computational Biology</journal-title></journal-title-group><issn pub-type="ppub">1553-734X</issn><issn pub-type="epub">1553-7358</issn><publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher></journal-meta>
    <article-meta><article-id pub-id-type="publisher-id">PCOMPBIOL-D-11-00526</article-id><article-id pub-id-type="doi">10.1371/journal.pcbi.1002255</article-id><article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Evolutionary biology</subject>
            <subj-group>
              <subject>Evolutionary genetics</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Genetics</subject>
            <subj-group>
              <subject>Population genetics</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Genomics</subject>
            <subj-group>
              <subject>Genome sequencing</subject>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Mathematics</subject>
          <subj-group>
            <subject>Statistics</subject>
            <subj-group>
              <subject>Statistical methods</subject>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Genetics and Genomics</subject>
          <subject>Evolutionary Biology</subject>
          <subject>Mathematics</subject>
        </subj-group>
      </article-categories><title-group><article-title>The Statistics of Bulk Segregant Analysis Using Next Generation Sequencing</article-title><alt-title alt-title-type="running-head">Next-Generation BSA</alt-title></title-group><contrib-group>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Magwene</surname>
            <given-names>Paul M.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Willis</surname>
            <given-names>John H.</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Kelly</surname>
            <given-names>John K.</given-names>
          </name>
          <xref ref-type="aff" rid="aff3">
            <sup>3</sup>
          </xref>
        </contrib>
      </contrib-group><aff id="aff1"><label>1</label><addr-line>Department of Biology and IGSP Center for Systems Biology, Duke University, Durham, North Carolina, United States of America</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>Department of Biology, Duke University, Durham, North Carolina, United States of America</addr-line>       </aff><aff id="aff3"><label>3</label><addr-line>Department of Ecology and Evolutionary Biology, University of Kansas, Lawrence, Kansas, United States of America</addr-line>       </aff><contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Siepel</surname>
            <given-names>Adam</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group><aff id="edit1">Cornell University, United States of America</aff><author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">paul.magwene@duke.edu</email></corresp>
        <fn fn-type="con">
          <p>Conceived and designed the experiments: PMM JHW JKK. Performed the experiments: PMM JKK. Analyzed the data: PMM JKK. Contributed reagents/materials/analysis tools: PMM JKK. Wrote the paper: PMM JHW JKK.</p>
        </fn>
      <fn fn-type="conflict">
        <p>The authors have declared that no competing interests exist.</p>
      </fn></author-notes><pub-date pub-type="collection">
        <month>11</month>
        <year>2011</year>
      </pub-date><pub-date pub-type="epub">
        <day>3</day>
        <month>11</month>
        <year>2011</year>
      </pub-date><volume>7</volume><issue>11</issue><elocation-id>e1002255</elocation-id><history>
        <date date-type="received">
          <day>15</day>
          <month>4</month>
          <year>2011</year>
        </date>
        <date date-type="accepted">
          <day>13</day>
          <month>9</month>
          <year>2011</year>
        </date>
      </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2011</copyright-year><copyright-holder>Magwene et al</copyright-holder><license><license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions><abstract>
        <p>We describe a statistical framework for QTL mapping using bulk segregant analysis (BSA) based on high throughput, short-read sequencing. Our proposed approach is based on a smoothed version of the standard <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e001" xlink:type="simple"/></inline-formula> statistic, and takes into account variation in allele frequency estimates due to sampling of segregants to form bulks as well as variation introduced during the sequencing of bulks. Using simulation, we explore the impact of key experimental variables such as bulk size and sequencing coverage on the ability to detect QTLs. Counterintuitively, we find that relatively large bulks maximize the power to detect QTLs even though this implies weaker selection and less extreme allele frequency differences. Our simulation studies suggest that with large bulks and sufficient sequencing depth, the methods we propose can be used to detect even weak effect QTLs and we demonstrate the utility of this framework by application to a BSA experiment in the budding yeast <italic>Saccharomyces cerevisiae</italic>.</p>
      </abstract><abstract abstract-type="summary">
        <title>Author Summary</title>
        <p>Quantitative or complex phenotypes are traits that are under the control of multiple genes and environmental factors. Identifying the parts of the genome that contribute to variation in complex traits (Quantitative Trait Loci or QTLs), and ultimately the genes and alleles that are mechanistically responsible for trait variation, is a primary challenge in animal and plant breeding, population studies of human health and disease, and evolutionary genetics. In this study we describe an analytical framework that allows investigators to marry a QTL mapping approach called “bulk segregant analysis” (BSA) with high-throughput genome sequencing methodologies in order to map traits quickly, efficiently, and in a relatively inexpensive manner. This framework provides a statistical basis for analyzing BSA experiments that use next-generation sequencing and will help to accelerate the identification of QTLs in both model and non-model organisms.</p>
      </abstract><funding-group><funding-statement>This research was supported by NIH grant P50GM081883-04 (to PMM), NIH grant R01-GM073990 (to JKK and JHW), NSF grant DEB-10-19753 (to PMM) and NSF grant IOS-10-24966 (to JHW). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts>
        <page-count count="9"/>
      </counts></article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>Bulk segregant analysis (BSA; <xref ref-type="bibr" rid="pcbi.1002255-Michelmore1">[1]</xref>) is a QTL mapping technique for identifying genomic regions containing genetic loci affecting a trait of interest. Starting with a segregating population from a genetic cross, individuals are assayed for the focal trait and two pools (bulks) of segregants are created by selecting individuals from the tails of the phenotypic distribution (other sampling designs can also be used as discussed below). Genotype frequencies are estimated for the two bulks, either via genotyping of individuals or via the creation of pooled DNA samples from which allele frequencies are estimated. Allele frequencies should be approximately equal between the two bulks in genomic regions without loci affecting the trait. Regions of the genome containing causal loci should exhibit allele frequency differences between bulks. BSA is most effective with high marker density and accurate allele frequency estimation within bulks <xref ref-type="bibr" rid="pcbi.1002255-Ehrenreich1">[2]</xref>. The former was effectively addressed with the application of microarray based genotyping to BSA <xref ref-type="bibr" rid="pcbi.1002255-Winzeler1">[3]</xref>–<xref ref-type="bibr" rid="pcbi.1002255-Demogines1">[8]</xref>. More recently, investigators have begun to use massively parallel sequencing methods to estimate allele frequencies for BSA studies <xref ref-type="bibr" rid="pcbi.1002255-Ehrenreich2">[9]</xref>–<xref ref-type="bibr" rid="pcbi.1002255-Parts1">[11]</xref>, which has a number of advantages. For organisms with moderately sized genomes, next generation sequencing can provide essentially single base-pair resolution. In such cases rather than simply observing markers in linkage with causal loci the BSA-sequencing approach should allow one to observe allelic biases at the causal loci themselves. For larger genomes where high coverage of the entire genome is less practical, BSA-sequencing still has many potential advantages. For example, it does not require the design of new genotyping arrays for new crosses and may provide greater resolution than array based genotyping. Furthermore, sequencing data yields counts of alleles at polymorphic loci and thus provides a simple and intuitive way of estimating allele frequencies.</p>
      <p>In bulk segregant studies based on high-throughput sequencing there are two sources of variation that affect allele frequency estimates. The first is variation due to the sampling of segregants that constitute the bulks themselves. This source of variation can be minimized by increasing both the size of the segregant population and the size of the bulk samples. The second source of variation is a consequence of the measurement technique used to estimate allele frequencies in the bulks. In the case of sequencing of pooled DNA samples, the sources of variation of this second type include, but are not limited to, library preparation, sequencing chemistry, sequencing coverage, post-sequencing alignment of reads, and base/allele calling algorithms. Here again, some of these sources of variation can be minimized by standardization of experimental protocols and analysis pipelines. However some of these sources of variation, particularly stochasticity in sequencing coverage, are an inherent property of short-read sequencing methods.</p>
      <p>In this paper, we develop explicit statistical models to describe the sources of variation that should be considered in the analysis of BSA-sequencing data. We first develop test statistics based on the classic <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e002" xlink:type="simple"/></inline-formula> -statistic accounting for the two phase sampling inherent to BSA. We then propose an analysis pipeline for whole-genome studies and present a proof-of-concept example with data from yeast. A combination of simulation and empirical application demonstrate the utility of this analytical framework.</p>
    </sec>
    <sec id="s2">
      <title>Results</title>
      <sec id="s2a">
        <title>Theory and Analytical Framework</title>
        <sec id="s2a1">
          <title>Expected distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e003" xlink:type="simple"/></inline-formula> for BSA-sequencing data</title>
          <p>Consider the experimental design with an F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e004" xlink:type="simple"/></inline-formula> population consisting of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e005" xlink:type="simple"/></inline-formula> individuals, each of which is measured for a phenotype of interest. A set of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e006" xlink:type="simple"/></inline-formula> individuals from each of the tails of the distribution (low and high) are collected. DNA bulks are prepared by combining equal amounts of tissue/cells from individuals within each bulk followed by DNA extraction, or by extracting DNA from each individual and combining equal amounts. Following preparation of DNA bulks, genomic libraries are prepared and sequenced at average coverage <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e007" xlink:type="simple"/></inline-formula> per SNP. Thus for each SNP the data is four allele counts that can be summarized in a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e008" xlink:type="simple"/></inline-formula> table, where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e009" xlink:type="simple"/></inline-formula> is the allele from the high parent (<xref ref-type="table" rid="pcbi-1002255-t001">Table 1</xref>). The <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e010" xlink:type="simple"/></inline-formula>-values in the table are counts of alleles not individuals. The observed allele frequency of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e011" xlink:type="simple"/></inline-formula> in the low bulk is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e012" xlink:type="simple"/></inline-formula>; that in the high bulk is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e013" xlink:type="simple"/></inline-formula>. If the SNP is close to a QTL with effects in the expected direction (i.e. the ‘high allele’ increases trait values), then we expect <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e014" xlink:type="simple"/></inline-formula>.</p>
          <table-wrap id="pcbi-1002255-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.t001</object-id><label>Table 1</label><caption>
              <title>The summary of data from a single variable site.</title>
            </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002255-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.t001" xlink:type="simple"/><table>
              <colgroup span="1">
                <col align="left" span="1"/>
                <col align="center" span="1"/>
                <col align="center" span="1"/>
                <col align="center" span="1"/>
              </colgroup>
              <thead>
                <tr>
                  <td align="left" colspan="1" rowspan="1"/>
                  <td align="left" colspan="1" rowspan="1">Low bulk</td>
                  <td align="left" colspan="1" rowspan="1">High bulk</td>
                  <td align="left" colspan="1" rowspan="1">Total</td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e015" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e016" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e017" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e018" xlink:type="simple"/></inline-formula>
                  </td>
                </tr>
                <tr>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e019" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e020" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e021" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e022" xlink:type="simple"/></inline-formula>
                  </td>
                </tr>
                <tr>
                  <td align="left" colspan="1" rowspan="1">Total</td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e023" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1">
                    <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e024" xlink:type="simple"/></inline-formula>
                  </td>
                  <td align="left" colspan="1" rowspan="1"/>
                </tr>
              </tbody>
            </table></alternatives><table-wrap-foot>
              <fn id="nt101">
                <label/>
                <p>The <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e025" xlink:type="simple"/></inline-formula> represent counts of alleles <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e026" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e027" xlink:type="simple"/></inline-formula> generated from sequencing of the segregant bulks.</p>
              </fn>
            </table-wrap-foot></table-wrap>
          <p>The counts in <xref ref-type="table" rid="pcbi-1002255-t001">Table 1</xref> are determined by two levels of hierarchical of sampling. The first sample is the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e028" xlink:type="simple"/></inline-formula> chromosomes that constitute each bulk (assuming diploid inheritance). Second, there is random variation in the number of reads per allele within each bulk due to the stochastic nature of next-generation sequencing. Let <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e029" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e030" xlink:type="simple"/></inline-formula> be the expected (‘true’) frequency of the high allele in each bulk. The realized frequencies (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e031" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e032" xlink:type="simple"/></inline-formula>) differ from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e033" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e034" xlink:type="simple"/></inline-formula> in each bulk due to binomial sampling:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e035" xlink:type="simple"/><label>(1)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e036" xlink:type="simple"/><label>(2)</label></disp-formula>If we assume that sequencing coverage is approximately Poisson, then the conditional distributions of the observed allele counts are:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e037" xlink:type="simple"/><label>(3)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e038" xlink:type="simple"/><label>(4)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e039" xlink:type="simple"/><label>(5)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e040" xlink:type="simple"/><label>(6)</label></disp-formula></p>
          <p>A natural statistic to characterize the data at each SNP is the standard <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e041" xlink:type="simple"/></inline-formula> -statistic:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e042" xlink:type="simple"/><label>(7)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e043" xlink:type="simple"/></inline-formula> is the ‘expected value’ for count <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e044" xlink:type="simple"/></inline-formula>. The null hypothesis is that there is no QTL close to the focal SNP. This implies the standard expected counts for a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e045" xlink:type="simple"/></inline-formula> contingency table, e.g. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e046" xlink:type="simple"/></inline-formula>. If the null hypothesis is correct, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e047" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e048" xlink:type="simple"/></inline-formula>. If we further assume no segregation distortion and equal (average) sequencing coverage of each bulk, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e049" xlink:type="simple"/></inline-formula>. See the supplementary materials (<xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref>) for a generalization that includes segregation distortion.</p>
          <p>However, due to the hierarchical sampling scheme, the usual expectation that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e050" xlink:type="simple"/></inline-formula> follows a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e051" xlink:type="simple"/></inline-formula> distribution (chi-square with 1 d.f.; <xref ref-type="bibr" rid="pcbi.1002255-Sokal1">[12]</xref>) does not hold in the present situation. The mean and variance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e052" xlink:type="simple"/></inline-formula> are inflated relative the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e053" xlink:type="simple"/></inline-formula> even when the null hypothesis is true (i.e. there is no QTL). Based on the arguments in <xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref> we approximate the mean and variance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e054" xlink:type="simple"/></inline-formula> as:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e055" xlink:type="simple"/><label>(8)</label></disp-formula><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e056" xlink:type="simple"/><label>(9)</label></disp-formula>These equations predict convergence on <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e057" xlink:type="simple"/></inline-formula> under certain parameter sets. In particular, if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e058" xlink:type="simple"/></inline-formula>, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e059" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e060" xlink:type="simple"/></inline-formula>, as expected from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e061" xlink:type="simple"/></inline-formula>.</p>
          <p>A simulation model was used to test the accuracy of approximate equations (8) and (9). We simulated genetic data for a chromosomal region of 10 cM in recombinational length. Informative markers were uniformly distributed along this chromosome with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e062" xlink:type="simple"/></inline-formula> SNPs per cM. The causal locus (QTL) was located at the center of the chromosome and was thus flanked by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e063" xlink:type="simple"/></inline-formula> SNPs on each side. Alternative homozygotes at the QTL differ by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e064" xlink:type="simple"/></inline-formula> phenotypic units on average (additive gene action) and simulations of the null hypothesis (no QTL) were done with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e065" xlink:type="simple"/></inline-formula>. In each simulation run, we first established the genotypes and phenotypes of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e066" xlink:type="simple"/></inline-formula> distinct F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e067" xlink:type="simple"/></inline-formula> segregants. Each individual was assigned a QTL genotype according to Mendelian probabilities (0.25, 0.5, 0.25) and the phenotype was assigned as the genotypic value plus a normal deviate. Individuals were then ranked by phenotype and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e068" xlink:type="simple"/></inline-formula> were selected from each tail. The full haplotype of these individuals was then established by working out from each allele at the QTL and allowing recombination to occur probabilistically according to the linkage map. Given the haplotypes in each bulk, we simulated an independent Poisson number for each count of <xref ref-type="table" rid="pcbi-1002255-t001">Table 1</xref> for each SNP. These data were used to calculate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e069" xlink:type="simple"/></inline-formula> at each SNP, and also <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e070" xlink:type="simple"/></inline-formula> as described below, within windows around each SNP. For the latter we needed to specify a window size in centimorgans. For each parameter set, this entire procedure was repeated 10,000 times. <xref ref-type="table" rid="pcbi-1002255-t001">Table 1</xref> in <xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref> reports simulation results for the null hypothesis (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e071" xlink:type="simple"/></inline-formula>) for a range of reasonable combinations of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e072" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e073" xlink:type="simple"/></inline-formula>. There is a close correspondence of observed means and variances of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e074" xlink:type="simple"/></inline-formula> with the values predicted by equations (8) and (9). As expected, in these simulations the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e075" xlink:type="simple"/></inline-formula> is right skewed with a mean and variance exceeding the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e076" xlink:type="simple"/></inline-formula> expectations.</p>
          <p>The full distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e077" xlink:type="simple"/></inline-formula> values is depicted for one parameter set (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e078" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e079" xlink:type="simple"/></inline-formula>) in <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1a</xref>. The gray histogram shows the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e080" xlink:type="simple"/></inline-formula> under the null hypothesis (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e081" xlink:type="simple"/></inline-formula>) while the overlapping red histogram shows the corresponding distribution in the case of a weak QTL (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e082" xlink:type="simple"/></inline-formula>). Focusing first on the null distribution, because the distribution is right skewed (mean = 1.19, variance = 2.93), if we compare this distribution to critical values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e083" xlink:type="simple"/></inline-formula> the observed false positive rate is somewhat elevated (6.98% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e084" xlink:type="simple"/></inline-formula>; 1.98% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e085" xlink:type="simple"/></inline-formula>). However when <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e086" xlink:type="simple"/></inline-formula> approaches <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e087" xlink:type="simple"/></inline-formula> the mean and variance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e088" xlink:type="simple"/></inline-formula> far exceed the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e089" xlink:type="simple"/></inline-formula> expectation and type I error rates increase dramatically. Perhaps even more problematic is the inability of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e090" xlink:type="simple"/></inline-formula> to detect a QTL based on the naïve <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e091" xlink:type="simple"/></inline-formula> expectation. For the weak QTL case, where the QTL explains 2% of the phenotypic variance, the causal SNP is significant at a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e092" xlink:type="simple"/></inline-formula> in only 34.9% of the simulations, and in only 16.8% of simulations at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e093" xlink:type="simple"/></inline-formula>. The application of the naïve <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e094" xlink:type="simple"/></inline-formula> thus suffers from a lack of power.</p>
          <fig id="pcbi-1002255-g001" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.g001</object-id>
            <label>Figure 1</label>
            <caption>
              <title>The distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e095" xlink:type="simple"/></inline-formula> (A) and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e096" xlink:type="simple"/></inline-formula> values (B) from 10,000 simulations.</title>
              <p>The gray histograms depict the observed distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e097" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e098" xlink:type="simple"/></inline-formula> for the null case (no QTL), while the red distributions depict the distributions in the case of a weak QTL that explains 2% of the phenotypic variance.</p>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.g001" xlink:type="simple"/>
          </fig>
        </sec>
        <sec id="s2a2">
          <title><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e099" xlink:type="simple"/></inline-formula>, A Smoothed Version of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e100" xlink:type="simple"/></inline-formula></title>
          <p>A substantial source of variation in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e101" xlink:type="simple"/></inline-formula> is the random margin in <xref ref-type="table" rid="pcbi-1002255-t001">Table 1</xref>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e102" xlink:type="simple"/></inline-formula>. To deal with this variation we propose the use of a weighted average of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e103" xlink:type="simple"/></inline-formula> across neighboring SNPs. Averaging <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e104" xlink:type="simple"/></inline-formula> values across SNPs is sensible because the real signal of divergence in allele frequency between bulks is conserved between closely linked sites but random noise due to variable sequencing read coverage is not. We suggest the following average test statistic for each SNP:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e105" xlink:type="simple"/><label>(10)</label></disp-formula>where the sum includes all SNPs within the window <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e106" xlink:type="simple"/></inline-formula> bracketing the SNP. This type of weighted moving average, where the weights are given by a kernel function, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e107" xlink:type="simple"/></inline-formula>, is also known as Nadaraya-Watson kernel regression <xref ref-type="bibr" rid="pcbi.1002255-Nadaraya1">[13]</xref>, <xref ref-type="bibr" rid="pcbi.1002255-Watson1">[14]</xref>. Nadaraya-Watson kernel regression acts as a smoothing function, with the amount of smoothing increasing with larger window size <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e108" xlink:type="simple"/></inline-formula> <xref ref-type="bibr" rid="pcbi.1002255-Schucany1">[15]</xref>. The simplest scheme for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e109" xlink:type="simple"/></inline-formula> would be to give equal weight to all SNPs within <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e110" xlink:type="simple"/></inline-formula> (a rectangular kernel). We opt instead to apply the tri-cube kernel fuction:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e111" xlink:type="simple"/><label>(11)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e112" xlink:type="simple"/></inline-formula> is standardized distance, with value 0 at the focal position and value 1 at the edge of the window. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e113" xlink:type="simple"/></inline-formula> is the sum of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e114" xlink:type="simple"/></inline-formula> for all SNPs in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e115" xlink:type="simple"/></inline-formula>. The tri-cube kernel is commonly used in local polynomial regression methods like LOESS <xref ref-type="bibr" rid="pcbi.1002255-Cleveland1">[16]</xref> and gives greater weight to observations that are close to the focal SNP. Any other weighting kernel that decreases smoothly to 0 as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e116" xlink:type="simple"/></inline-formula> goes to 1 could be used as well. We discuss the choice of the kernel window size, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e117" xlink:type="simple"/></inline-formula>, below.</p>
          <p>A methodological issue arises when kernel smoothing is used – at the beginning or end of a data series it can produce a biased estimate because the data included in the kernel bandwidth is asymmetric. The simplest way to deal with this is to append a reflected version of the values that fall within the right half-bandwith (at the beginning of the series) and left half-bandwidth (at the end of the series), run the kernel smoother as normal, and then trim the appended values from the output.</p>
        </sec>
        <sec id="s2a3">
          <title>Expected distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e118" xlink:type="simple"/></inline-formula> for BSA-sequencing data</title>
          <p>The null expectation of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e119" xlink:type="simple"/></inline-formula> is given by equation (8). The variance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e120" xlink:type="simple"/></inline-formula> depends on the variance of individual <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e121" xlink:type="simple"/></inline-formula> values (equation 9) and the covariance between SNPs within a window. In <xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref> we show that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e122" xlink:type="simple"/></inline-formula> can be approximated as:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e123" xlink:type="simple"/><label>(12)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e124" xlink:type="simple"/></inline-formula> indexes all SNPs other than <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e125" xlink:type="simple"/></inline-formula> contained within the window.</p>
          <p><xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1b</xref> illustrates the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e126" xlink:type="simple"/></inline-formula> for the same parameters as <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1a</xref> (plus window size <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e127" xlink:type="simple"/></inline-formula> cM and SNP density <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e128" xlink:type="simple"/></inline-formula> per cM). The difference between the null distributions in <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1a and 1b</xref> is due to the normalizing effect of averaging. The predicted mean and variance of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e129" xlink:type="simple"/></inline-formula> (1.17 and 0.066) are reasonably close to the observed moments (1.18 and 0.056). The distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e130" xlink:type="simple"/></inline-formula> is still right skewed but the right tail can reasonably predicted from log-normal densities with parameters derived from <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e131" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e132" xlink:type="simple"/></inline-formula> (<xref ref-type="supplementary-material" rid="pcbi.1002255.s001">Figure S1</xref> and <xref ref-type="supplementary-material" rid="pcbi.1002255.s003">Text S2</xref>). The observed false-positive rates (using a log-normal density estimation) are: 5.14% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e133" xlink:type="simple"/></inline-formula> and 1.86% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e134" xlink:type="simple"/></inline-formula>). Unlike the use of the naive <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e135" xlink:type="simple"/></inline-formula> -test based on <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e136" xlink:type="simple"/></inline-formula>, the type I error does not increase dramatically as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e137" xlink:type="simple"/></inline-formula> approaches <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e138" xlink:type="simple"/></inline-formula>. Furthermore, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e139" xlink:type="simple"/></inline-formula> has good power to detect QTLs. For the example illustrated in <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1b</xref> the causal SNP is significant in 94.3% of the simulations at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e140" xlink:type="simple"/></inline-formula>, and in 88.0% and 77.2% of simulations at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e141" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e142" xlink:type="simple"/></inline-formula> respectively.</p>
        </sec>
        <sec id="s2a4">
          <title>Non-parametric estimation of the null distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e143" xlink:type="simple"/></inline-formula></title>
          <p>In addition to the theoretical expectations discussed above, an empirical estimate of the null distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e144" xlink:type="simple"/></inline-formula> can be derived from the observed data itself. We assume that the observed data, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e145" xlink:type="simple"/></inline-formula>, is a mixture of the null distribution (non-QTL regions) and several contaminating distributions (QTLs). As discussed above, the null distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e146" xlink:type="simple"/></inline-formula> (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e147" xlink:type="simple"/></inline-formula>) is right-skewed with a tail density reasonably predicted from a log-normal distribution, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e148" xlink:type="simple"/></inline-formula>. We also assume the contaminating distributions have higher means than the null distribution. Our goal is to estimate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e149" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e150" xlink:type="simple"/></inline-formula> in a manner that is not unduly influenced by the contaminating distributions.</p>
          <p>Recall that for a log-normal distribution: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e151" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e152" xlink:type="simple"/></inline-formula> <xref ref-type="bibr" rid="pcbi.1002255-Mohn1">[17]</xref>. Thus if we can estimate the median and mode of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e153" xlink:type="simple"/></inline-formula> can use those to estimate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e154" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e155" xlink:type="simple"/></inline-formula>. To do so we propose the folowing steps:</p>
          <list list-type="order">
            <list-item>
              <p>Let <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e156" xlink:type="simple"/></inline-formula></p>
            </list-item>
            <list-item>
              <p>Let <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e157" xlink:type="simple"/></inline-formula>, the left median absolute deviation (MAD) of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e158" xlink:type="simple"/></inline-formula> where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e159" xlink:type="simple"/></inline-formula> is defined as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e160" xlink:type="simple"/></disp-formula></p>
            </list-item>
            <list-item>
              <p>Use Hampel's rule <xref ref-type="bibr" rid="pcbi.1002255-Davies1">[18]</xref> to identify outliers, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e161" xlink:type="simple"/></inline-formula>, as all <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e162" xlink:type="simple"/></inline-formula> in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e163" xlink:type="simple"/></inline-formula> that satisfy:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e164" xlink:type="simple"/></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e165" xlink:type="simple"/></inline-formula> defines the limits of the outlier regions <xref ref-type="bibr" rid="pcbi.1002255-Davies1">[18]</xref> and is usually taken to be 5.2 for normally distributed data.</p>
            </list-item>
            <list-item>
              <p>Construct a trimmed data set <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e166" xlink:type="simple"/></inline-formula> for all <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e167" xlink:type="simple"/></inline-formula> such that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e168" xlink:type="simple"/></inline-formula></p>
            </list-item>
            <list-item>
              <p>Let <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e169" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e170" xlink:type="simple"/></inline-formula> where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e171" xlink:type="simple"/></inline-formula> is a robust estimator of the mode for continuous variables (see <xref ref-type="bibr" rid="pcbi.1002255-Bickel1">[19]</xref> for several such estimators)</p>
            </list-item>
          </list>
          <p>The logic of this procedure is as follows. The median and MAD are robust estimators of location and spread respectively <xref ref-type="bibr" rid="pcbi.1002255-Rousseeuw1">[20]</xref>. In the absence of contaminating distributions <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e172" xlink:type="simple"/></inline-formula> should be approximately normally distributed, and hence the median and MAD of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e173" xlink:type="simple"/></inline-formula> can be used as robust estimates of the mean and spread of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e174" xlink:type="simple"/></inline-formula> (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e175" xlink:type="simple"/></inline-formula> for a symmetric distribution). Hampel's rule is a commonly used procedure to identify likely outliers in a set of data based on the median and MAD; if the underlying distribution is normally distributed and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e176" xlink:type="simple"/></inline-formula> this is approximately equivalent to identifying outliers as those observations with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e177" xlink:type="simple"/></inline-formula>-values <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e178" xlink:type="simple"/></inline-formula> (we use a one-sided test in the procedure above). When contaminating distributions (QTLs) are present, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e179" xlink:type="simple"/></inline-formula> lies to the right of the true mean of the null distribution. Thus, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e180" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e181" xlink:type="simple"/></inline-formula> are conservative estimators of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e182" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e183" xlink:type="simple"/></inline-formula>. We then use Hampel's procedure to identify observations likely to be drawn from the contaminating distributions and create a trimmed data set, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e184" xlink:type="simple"/></inline-formula>, with those outlying observations removed. From the trimmed data set we estimate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e185" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e186" xlink:type="simple"/></inline-formula>.</p>
          <p>For the null simulations in <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1b</xref> the observed false-positive rate estimated using this non-parametric approach are 3.18% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e187" xlink:type="simple"/></inline-formula> and 0.76% at <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e188" xlink:type="simple"/></inline-formula>. In general, the non-parameteric procedure tends to be slightly more conservative than our proposed parametric estimators but not greatly so. Because this non-parametric approach makes few distributional assumptions (other than approximate log-normality of the null distribution) it might be preferred in cases where one suspects the sampling (either of segregants or alleles) grossly violates the hierarchical model described above.</p>
        </sec>
        <sec id="s2a5">
          <title>Choosing <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e189" xlink:type="simple"/></inline-formula></title>
          <p>A weighted moving average is a type of low-pass filter; the larger the window size the lower the frequncy of signals that are rejected by the filter. The choice of smoothing width, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e190" xlink:type="simple"/></inline-formula>, is therefore a tradeoff between filtering out high-frequency deviations in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e191" xlink:type="simple"/></inline-formula> due to variable sequence coverage and SNP density and attenuating the signal of real QTLs. We want to pick a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e192" xlink:type="simple"/></inline-formula> that minimizes noise while maximizing the underlying signal. The matched filter theorem <xref ref-type="bibr" rid="pcbi.1002255-Turin1">[21]</xref> suggests that the filter that maximizes the signal-to-noise ratio of a symmetric signal is one which matches the shape of the signal. A simple measure of the shape of a symmetric signal is the full-width at half maximum (FWHM). The ratio of the width of the kernel to the peak FHWM (‘smoothing ratio’) is a useful metric for quantifying the effects of smoothing <xref ref-type="bibr" rid="pcbi.1002255-Enke1">[22]</xref>. As a rule of thumb, using a smoothing kernel with a smoothing ratio of approximately two provides a good signal-to-noise ratio <xref ref-type="bibr" rid="pcbi.1002255-Enke1">[22]</xref>. However, the matched filter may fail to distinguish multiple peaks when there are two or more signals in the input <xref ref-type="bibr" rid="pcbi.1002255-Gu1">[23]</xref> as we would expect in cases of multiple QTLs with overlapping regions of elevated <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e193" xlink:type="simple"/></inline-formula>. Specifically, peaks separated by less than twice the FWHM of the filter will be merged <xref ref-type="bibr" rid="pcbi.1002255-Mikl1">[24]</xref>. Therefore, to distinguish overlapping signals requires filters with smoothing ratios significantly smaller, perhaps as small as 0.7.</p>
          <p>In <xref ref-type="supplementary-material" rid="pcbi.1002255.s003">Text S2</xref> we derive the expected shape of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e194" xlink:type="simple"/></inline-formula> around a single causal SNP. For the case in which the causal allele is fixed in one bulk and has a frequency of 0.5 in the other bulk, the half-bandwidth (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e195" xlink:type="simple"/></inline-formula>) at half-maximum corresponds to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e196" xlink:type="simple"/></inline-formula>12.42 cM (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e197" xlink:type="simple"/></inline-formula>). More extreme allelic biases between the bulks favor slightly smaller bandwidths, while less extreme differences favor larger bandwidths. SNP density also affects the optimal kernel bandwidth, with higher SNP density favoring narrower bandwidths. In simulations and applied to real data we have found that kernels with smoothing ratios in the range 1–1.5 produce smoothed estimators with good signal-to-noise ratios and which are neither strongly over- or undersmoothed. In terms of mapping distances this corresponds to kernels with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e198" xlink:type="simple"/></inline-formula> in the range <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e199" xlink:type="simple"/></inline-formula>24.8–37.25 cM.</p>
          <p>Since recombination rates vary across genomes, a given genetic distance will correspond to a range of physical distances. In terms of the choice of smoothing width, higher recombination rates favor smaller window sizes (in physical distance). If regional recombination rates are known this can be incorporated into the analysis; however the use of average chromosomal or genomic recombination rates to choose a single physical size for the smoothing window should not be problematic unless recombination rates vary widely. In such cases, one can calculate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e200" xlink:type="simple"/></inline-formula> using a range of smoothing widths to explore whether peak estimates are strongly affected by over- or undersmoothing.</p>
        </sec>
      </sec>
      <sec id="s2b">
        <title>Proposed Analytical Pipeline</title>
        <p>Based on the arguments developed above, we propose the following analytical pipeline for the analysis of BSA-sequencing data sets. We assume that sequencing reads have been aligned to a reference genome where physical distances between polymorphic sites and (approximate) rates of recombination are known. We assume that all sites are biallelic. Following alignment of reads to a reference genome, per site counts of each allele are generated from the reads. Our recommended analysis pipeline for estimating QTLs is as follows:</p>
        <list list-type="order">
          <list-item>
            <p>For each variable site, calculate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e201" xlink:type="simple"/></inline-formula> based on the observed number of reads for each allele in each of the two pools</p>
          </list-item>
          <list-item>
            <p>At each site calculate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e202" xlink:type="simple"/></inline-formula> using a smoothing kernel with bandwidth <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e203" xlink:type="simple"/></inline-formula> bases where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e204" xlink:type="simple"/></inline-formula> is chosen based on known or estimated rates of recombination. Bandwidths should typically correspond to genetic map distances in the range 25–40 cM.</p>
          </list-item>
          <list-item>
            <p>Estimate parameters of the log-normal null-distribution (i.e. no QTL) of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e205" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e206" xlink:type="simple"/></inline-formula>, based on either theoretical expectations (equations (8) and (12 and <xref ref-type="supplementary-material" rid="pcbi.1002255.s003">Text S2</xref>) or using the robust empirical estimator of the null distribution inferred from the observed <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e207" xlink:type="simple"/></inline-formula>.</p>
          </list-item>
          <list-item>
            <p>Using <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e208" xlink:type="simple"/></inline-formula> estimate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e209" xlink:type="simple"/></inline-formula>-values directly using the log-normal CDF. Alternately log-transform <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e210" xlink:type="simple"/></inline-formula> and calculate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e211" xlink:type="simple"/></inline-formula> scores <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e212" xlink:type="simple"/></inline-formula> and corresponding <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e213" xlink:type="simple"/></inline-formula>-values at each site.</p>
          </list-item>
          <list-item>
            <p>Use a false discovery rate approach (FDR; <xref ref-type="bibr" rid="pcbi.1002255-Benjamini1">[25]</xref>, <xref ref-type="bibr" rid="pcbi.1002255-Benjamini2">[26]</xref>) to account for multiple comparisons and estimate an appropriate p-value threshold (or the corresponding <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e214" xlink:type="simple"/></inline-formula> threshold) to determine sites that deviate significantly from the background null distribution</p>
          </list-item>
          <list-item>
            <p>Define candidate QTL regions as continuous runs of significant sites</p>
          </list-item>
        </list>
      </sec>
      <sec id="s2c">
        <title>Power Analysis</title>
        <p>We used simulations to conduct a simple power analysis of our proposed methodology. In this analysis we used the mean <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e215" xlink:type="simple"/></inline-formula> at a causal site as measure of power for given values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e216" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e217" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e218" xlink:type="simple"/></inline-formula>, window size (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e219" xlink:type="simple"/></inline-formula>), SNP density, and for different magnitudes of QTL effect on phenotype. <xref ref-type="fig" rid="pcbi-1002255-g002">Figure 2</xref> summarizes results for two different values of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e220" xlink:type="simple"/></inline-formula>, corresponding to large (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e221" xlink:type="simple"/></inline-formula>) and very large (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e222" xlink:type="simple"/></inline-formula>) F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e223" xlink:type="simple"/></inline-formula> populations. We find that increasing coverage, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e224" xlink:type="simple"/></inline-formula>, is advantageous until <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e225" xlink:type="simple"/></inline-formula>, but has minimal effect beyond that. A somewhat counterintuitive result is that larger bulk size, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e226" xlink:type="simple"/></inline-formula>, is generally beneficial as long as sequencing coverage is modest to high. This is despite the fact that larger bulks imply weaker selection for a given <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e227" xlink:type="simple"/></inline-formula> (and hence a smaller allele frequency divergence among bulks). Based on these findings we recommend bulks consisting of at least 10% and as perhaps as high as 20% of the F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e228" xlink:type="simple"/></inline-formula> segregant population in order to maximize power to detect QTLs.</p>
        <fig id="pcbi-1002255-g002" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.g002</object-id>
          <label>Figure 2</label>
          <caption>
            <title>Power analysis.</title>
            <p>Average <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e229" xlink:type="simple"/></inline-formula> at a causal site as a function of sequencing coverage, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e230" xlink:type="simple"/></inline-formula>, and bulk size, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e231" xlink:type="simple"/></inline-formula>, for two different F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e232" xlink:type="simple"/></inline-formula> population sizes (left, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e233" xlink:type="simple"/></inline-formula>; right, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e234" xlink:type="simple"/></inline-formula>). Note the difference in scales between the two figures.</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.g002" xlink:type="simple"/>
        </fig>
      </sec>
      <sec id="s2d">
        <title>An Application to Yeast</title>
        <p>To demonstrate the correspondence between theory and data we here draw on a BSA-sequencing data set generated to identify loci that contribute to variation in colony morphology in the budding yeast <italic>Saccharomyces cerevisiae</italic> <xref ref-type="bibr" rid="pcbi.1002255-Granek1">[27]</xref>. A full description and analysis of these data will appear elsewhere (Granek et al., in prep). Here, these data serve to illustrate the utility of both our theoretical framework and the associated robust estimators for data analysis.</p>
        <p>The yeast data consist of a low and high bulk, each composed of 288 homozygous diploid segregants drawn from an F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e235" xlink:type="simple"/></inline-formula> population of size <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e236" xlink:type="simple"/></inline-formula> generated by sporulating a naturally heterozygous diploid strain <xref ref-type="bibr" rid="pcbi.1002255-Magwene1">[28]</xref>. The low bulk consists of segregants with simple colony morphology, while the high bulk consists of segregants with complex colony morphology (see <xref ref-type="bibr" rid="pcbi.1002255-Granek1">[27]</xref> for a description of morphology scoring). Creation of DNA pools, sequencing, and mapping of reads is described in the Methods section. Because each segregant is homozygous, the effective number of alleles sampled for each bulk is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e237" xlink:type="simple"/></inline-formula> instead of 2 <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e238" xlink:type="simple"/></inline-formula>. In total 44,066 polymorphic sites were analyzed with a mean interval between sites of approximately 280 bp. Below we refer to the two sequencing runs for the low bulks as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e239" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e240" xlink:type="simple"/></inline-formula>, and those for the high bulks as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e241" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e242" xlink:type="simple"/></inline-formula>. The coverage per SNP (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e243" xlink:type="simple"/></inline-formula>) for each sequencing run was as follows: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e244" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e245" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e246" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e247" xlink:type="simple"/></inline-formula>. For each of the analyses below, we used a smoothing window width of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e248" xlink:type="simple"/></inline-formula> (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e249" xlink:type="simple"/></inline-formula>30 cM), and took the average coverage of each bulk being compared as the estimate of coverage, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e250" xlink:type="simple"/></inline-formula>.</p>
        <p>Because there are two sequencing runs per DNA pool, variation in allele frequency estimates between sequencing runs from the same segregant bulk should be exclusively due to stochastic aspects of the sequencing reaction and primary bioinformatics analyses (base calling, read alignment). The structure of this data set is thus useful for dissecting the impact of sequencing variation on estimates of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e251" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e252" xlink:type="simple"/></inline-formula>, and the subsequent impact of this variability on the inference of QTL regions and peaks. We use these data to explore both the null model (no QTL; by analyzing the low-vs-low and high-vs-high comparisons) as well as the case where QTLs are expected (comparing low-vs-high bulks). In the null case, the differences in allele frequencies are subject to only one source of variation because the bulks are fixed but sequencing is variable. The non-null analyses are individually affected by both sources of variation (bulking and sequencing), but when comparing the results from comparable analyses (e.g. comparing QTL peak locations between the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e253" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e254" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e255" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e256" xlink:type="simple"/></inline-formula> analyses), the differences are again simply a function of sequencing variation.</p>
        <sec id="s2d1">
          <title>Null comparisons: Variation in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e257" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e258" xlink:type="simple"/></inline-formula> due to sequencing</title>
          <p>The two low samples (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e259" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e260" xlink:type="simple"/></inline-formula>) and the two high samples (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e261" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e262" xlink:type="simple"/></inline-formula>) represent independent sequencing runs of the same low and high segregant bulks respectively. Using <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e263" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e264" xlink:type="simple"/></inline-formula> from a comparison of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e265" xlink:type="simple"/></inline-formula> vs. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e266" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e267" xlink:type="simple"/></inline-formula> vs. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e268" xlink:type="simple"/></inline-formula> we can estimate the impact of sequencing on the variation of these statistics. When the two bulks differ only due to read number variation, there is only one source of variation, and the statistics of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e269" xlink:type="simple"/></inline-formula> should should be approximately <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e270" xlink:type="simple"/></inline-formula> with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e271" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e272" xlink:type="simple"/></inline-formula>. By invoking a weighted version of the central limit theorem <xref ref-type="bibr" rid="pcbi.1002255-Weber1">[29]</xref>, we find the distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e273" xlink:type="simple"/></inline-formula> should be approximately normal with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e274" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e275" xlink:type="simple"/></inline-formula> where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e276" xlink:type="simple"/></inline-formula>, the sum of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e277" xlink:type="simple"/></inline-formula> squared kernel weights in the smoothing window (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e278" xlink:type="simple"/></inline-formula> converges to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e279" xlink:type="simple"/></inline-formula> in the case of a square kernel). As illustrated in <xref ref-type="table" rid="pcbi-1002255-t002">Table 2</xref> the observed data for the null-comparisons conform well to the asymptotic expectations.</p>
          <table-wrap id="pcbi-1002255-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.t002</object-id><label>Table 2</label><caption>
              <title>Null comparisons for the yeast data set.</title>
            </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002255-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.t002" xlink:type="simple"/><table>
              <colgroup span="1">
                <col align="left" span="1"/>
                <col align="center" span="1"/>
                <col align="center" span="1"/>
                <col align="center" span="1"/>
                <col align="center" span="1"/>
              </colgroup>
              <thead>
                <tr>
                  <td align="left" colspan="1" rowspan="1">Comparison</td>
                  <td align="left" colspan="1" rowspan="1">Theoretical <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e280" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e281" xlink:type="simple"/></inline-formula></td>
                  <td align="left" colspan="1" rowspan="1">Observed <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e282" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e283" xlink:type="simple"/></inline-formula></td>
                  <td align="left" colspan="1" rowspan="1">Theoretical <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e284" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e285" xlink:type="simple"/></inline-formula></td>
                  <td align="left" colspan="1" rowspan="1">Observed <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e286" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e287" xlink:type="simple"/></inline-formula></td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e288" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e289" xlink:type="simple"/></inline-formula></td>
                  <td align="left" colspan="1" rowspan="1">1.000, 2.000</td>
                  <td align="left" colspan="1" rowspan="1">1.018, 2.050</td>
                  <td align="left" colspan="1" rowspan="1">1.000, 0.0124</td>
                  <td align="left" colspan="1" rowspan="1">1.020, 0.0115</td>
                </tr>
                <tr>
                  <td align="left" colspan="1" rowspan="1"><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e290" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e291" xlink:type="simple"/></inline-formula></td>
                  <td align="left" colspan="1" rowspan="1">1.000, 2.000</td>
                  <td align="left" colspan="1" rowspan="1">1.015, 2.077</td>
                  <td align="left" colspan="1" rowspan="1">1.000, 0.0124</td>
                  <td align="left" colspan="1" rowspan="1">1.014, 0.0117</td>
                </tr>
              </tbody>
            </table></alternatives><table-wrap-foot>
              <fn id="nt102">
                <label/>
                <p>Theoretical and observed means and variances of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e292" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e293" xlink:type="simple"/></inline-formula> for the null comparisons in the yeast data set.</p>
              </fn>
            </table-wrap-foot></table-wrap>
        </sec>
        <sec id="s2d2">
          <title>Between replicate comparisons of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e294" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e295" xlink:type="simple"/></inline-formula> in the presence of a QTL</title>
          <p>In addition to tests of the null model, the design of the yeast experiment facilitates a between replicate comparison of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e296" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e297" xlink:type="simple"/></inline-formula> in the presence of QTLs. There are four possible low-vs-high comparisons; here we focus on two of those, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e298" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e299" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e300" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e301" xlink:type="simple"/></inline-formula>. <xref ref-type="fig" rid="pcbi-1002255-g003">Figure 3</xref> illustrates the relationships for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e302" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e303" xlink:type="simple"/></inline-formula> at each SNP for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e304" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e305" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e306" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e307" xlink:type="simple"/></inline-formula>. The between replicate correlation for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e308" xlink:type="simple"/></inline-formula> is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e309" xlink:type="simple"/></inline-formula>0.677, while that between <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e310" xlink:type="simple"/></inline-formula> is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e311" xlink:type="simple"/></inline-formula>0.996. This illustrates the ability of the smoothing kernel to act as a low-pass filter on the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e312" xlink:type="simple"/></inline-formula> -statistic, filtering out the high-frequency noise associated with variation in read counts, while preserving the underlying signal of QTLs and increasing the repeatability of the analysis.</p>
          <fig id="pcbi-1002255-g003" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.g003</object-id>
            <label>Figure 3</label>
            <caption>
              <title>Comparison of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e313" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e314" xlink:type="simple"/></inline-formula> between technical replicates.</title>
              <p>The correspondence of raw <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e315" xlink:type="simple"/></inline-formula> (black) and smoothed <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e316" xlink:type="simple"/></inline-formula> values (red) for different sequencing runs of the same low-vs-high bulks from the yeast data set.</p>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.g003" xlink:type="simple"/>
          </fig>
          <p>Using the false discovery rate approach outline above, we estimated cutoff values for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e317" xlink:type="simple"/></inline-formula> using a FDR of 0.01 based on both our theoretical results (equations 8 and 12) and the corresponding non-parametric estimators. For the parametric estimate we used the following parameter values: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e318" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e319" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e320" xlink:type="simple"/></inline-formula>. The estimated <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e321" xlink:type="simple"/></inline-formula> cutoff values are as follows: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e322" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e323" xlink:type="simple"/></inline-formula> : 2.59 [parametric], 3.51 [non-parameteric]; <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e324" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e325" xlink:type="simple"/></inline-formula> : 2.58 [parametric], 3.91 [non-parametric].</p>
          <p>Using the theoretical <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e326" xlink:type="simple"/></inline-formula> cutoff of 2.59 we find 7,845 SNPs have significant <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e327" xlink:type="simple"/></inline-formula> values for the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e328" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e329" xlink:type="simple"/></inline-formula> comparison, and 8,011 significant SNPs for the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e330" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e331" xlink:type="simple"/></inline-formula> comparison, representing approximately 17% of the polymorphic sites. Nearly 38% of the significant sites are on chromosome XIII which appears to have multiple overlapping peaks leading to elevated <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e332" xlink:type="simple"/></inline-formula> values across much of the chromosome. The number of significant sites shared between the replicates is 7,330. We identified 12 significant regions (QTLs) in the two replicates (<xref ref-type="fig" rid="pcbi-1002255-g004">Figure 4</xref>). The QTLs are nearly identical between the replicates except for a marginal QTL on chromosome 7, where one of the replicates is significant but the other is just short of significance. To assess the variability in QTL location we compared the distance between peaks (using the single largest peak in cases of multiple peaks per chromosome). The mean and median absolute distances between nine comparable QTL peaks from the two comparisons are 5.08 Kb and 4.97 Kb respectively. The root mean square deviation (RMSD) between comparable QTL peaks is 6.7 Kb. Using the RMSD as a measure of spread and applying the 3<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e333" xlink:type="simple"/></inline-formula> rule of thumb, a conservative confidence interval for QTL peak is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e334" xlink:type="simple"/></inline-formula>20 Kb (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e335" xlink:type="simple"/></inline-formula>7.4 cM) around the observed peak. The size of this confidence interval is a function of read depth and SNP density, and <italic>is a measure of variability in peak estimation due to sequencing only</italic>. This confidence interval doesn't include variation that would arise from the bulking of segregants.</p>
          <fig id="pcbi-1002255-g004" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pcbi.1002255.g004</object-id>
            <label>Figure 4</label>
            <caption>
              <title>Yeast QTL Peaks.</title>
              <p>Chromosomal distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e336" xlink:type="simple"/></inline-formula> for the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e337" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e338" xlink:type="simple"/></inline-formula> (dark blue) and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e339" xlink:type="simple"/></inline-formula>-vs-<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e340" xlink:type="simple"/></inline-formula> (light blue) data sets. The dashed red line indicates the estimated <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e341" xlink:type="simple"/></inline-formula> threshold corresponding to a FDR of 0.01. Regions above the red line are QTL regions; the highest point in each QTL region was called as the QTL peak.</p>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.g004" xlink:type="simple"/>
          </fig>
          <p>As will be described elsewhere, candidate genes corresponding to several of the major peaks in this analysis have been functionally validated to affect yeast colony morphology (J. Granek and P. Magwene, unpublished data).</p>
        </sec>
      </sec>
    </sec>
    <sec id="s3">
      <title>Discussion</title>
      <p>The use of a test based on the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e342" xlink:type="simple"/></inline-formula> -statistic provides a straightforward framework for analyzing BSA-sequencing data. The <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e343" xlink:type="simple"/></inline-formula> -statistic has several advantages over the use of allele frequency differences as the basis for QTL estimation (e.g. <xref ref-type="bibr" rid="pcbi.1002255-Parts1">[11]</xref>). For example, as shown in the supporting information (<xref ref-type="supplementary-material" rid="pcbi.1002255.s003">Text S2</xref>), <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e344" xlink:type="simple"/></inline-formula> is expected to decrease much more rapidly around the causal site than bias in allele frequencies, implying narrower intervals of support around QTLs. Also in contrast to statistics based on the divergence of allele frequencies, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e345" xlink:type="simple"/></inline-formula> takes into account the strength of evidence related to sample size. This feature of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e346" xlink:type="simple"/></inline-formula> -statistic can also potentially complicate analyses, as variance in read depth contributes to variance in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e347" xlink:type="simple"/></inline-formula> over relatively small spatial scales. However, as we show above, weighted averaging of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e348" xlink:type="simple"/></inline-formula> effectively smooths out ‘high frequency’ noise associated with sequencing variation.</p>
      <sec id="s3a">
        <title>Bulk Size and Sequencing Considerations</title>
        <p>Our simulations suggest that for the experimental design considered here using bulk sizes as large as 15–20% of the phenotyped segregant population increases power to detect causal QTLs despite the fact that this means relatively smaller allele frequency differences between bulks. This is due to tradeoffs between bulk-size, selection intensity, and the variance of allele frequencies under the hierarchical sampling. Consider, for example, a single locus with alleles <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e349" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e350" xlink:type="simple"/></inline-formula>, where the effect of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e351" xlink:type="simple"/></inline-formula> is additive and the two homozygotes differ by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e352" xlink:type="simple"/></inline-formula> units on average. Assuming no segregation distortion, and an <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e353" xlink:type="simple"/></inline-formula> population generated from inbred lines, the change in the allele frequency of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e354" xlink:type="simple"/></inline-formula> in the high bulk after truncation selection is approximately <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e355" xlink:type="simple"/></inline-formula> <xref ref-type="bibr" rid="pcbi.1002255-Kimura1">[30]</xref>, <xref ref-type="bibr" rid="pcbi.1002255-Falconer1">[31]</xref> where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e356" xlink:type="simple"/></inline-formula> is the intensity of selection, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e357" xlink:type="simple"/></inline-formula> is the ‘standardized effect of the locus’ (these quantities can be related to the selection coefficient, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e358" xlink:type="simple"/></inline-formula>, by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e359" xlink:type="simple"/></inline-formula>). Given truncation selection on a normal distribution, the intensity of selection is given by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e360" xlink:type="simple"/></inline-formula> where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e361" xlink:type="simple"/></inline-formula> is the proportion of selected individuals and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e362" xlink:type="simple"/></inline-formula> is the probability density function at the truncation point <xref ref-type="bibr" rid="pcbi.1002255-Falconer1">[31]</xref>. Since the intensity of selection increases at a rate much less than <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e363" xlink:type="simple"/></inline-formula> (e.g. see <xref ref-type="bibr" rid="pcbi.1002255-Falconer1">[31]</xref>, Fig. 11.3), an <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e364" xlink:type="simple"/></inline-formula>-fold decrease in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e365" xlink:type="simple"/></inline-formula> results in a much less than <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e366" xlink:type="simple"/></inline-formula>-fold change in the intensity of selection. For example, let <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e367" xlink:type="simple"/></inline-formula> and consider truncation on the upper 20%, 10%, and 1%, of the phenotypic distribution. The increase in the frequency of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e368" xlink:type="simple"/></inline-formula> in the high bulk given these truncation points is approximately 3.5%, 4.4%, and 6.7% respectively (translating to allele frequency differences of 7%, 8.8%, and 13.4% in the two-bulk case). On the other hand, the variance of the realized frequencies of the alleles in each bulk is inversely proportional to bulk size (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e369" xlink:type="simple"/></inline-formula>). Thus, a twenty-fold decrease in bulk size translates to less than a two-fold increase in allele frequency divergence, but a twenty-fold increase in the variance of allele frequencies. As long as average coverage, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e370" xlink:type="simple"/></inline-formula>, is moderate to large, the benefit of increasing <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e371" xlink:type="simple"/></inline-formula> offsets the relatively smaller penalty resulting from a decrease in selection intensity. However, there is little benefit to increasing sequencing coverage beyond the size of the bulks.</p>
        <p>Sequencing can introduce complications such as biases toward particular nucleotide calls; however in general this should effect both segregant bulks in the same direction. Due to the averaging affect of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e372" xlink:type="simple"/></inline-formula>, unless such biased sites are common over very large map distances they are unlikely to have substantial affects on results derived under our proposed framework. Similarly, a low percentage of mismapped reads or miscalled SNP calling are unlikely to be problematic for our framework, again because of the averaging affect of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e373" xlink:type="simple"/></inline-formula>. However caution should be exercised in genomic regions that are particularly problematic in this regard, such as repeat rich regions.</p>
      </sec>
      <sec id="s3b">
        <title>Other Experimental Designs</title>
        <p>In this paper we have focused on QTL mapping with an F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e374" xlink:type="simple"/></inline-formula> experimental design, but clearly our framework can be extended to other designs. Common alternatives include mapping populations produced by imposing one or more generations of inbreeding on an F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e375" xlink:type="simple"/></inline-formula>, such as Recombinant Inbred Lines (RILs). The increased homozygosity of such populations should also be taken into consideration, as it increases the expected change in allele frequency due to selection but it also decreases the number of independent chromosomes that are sampled for a given number of selected individuals. Chromosomes in such RILs experience as much as twice the number of crossovers as do F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e376" xlink:type="simple"/></inline-formula> populations so the physical size of the smoothing window <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e377" xlink:type="simple"/></inline-formula> should be reduced to take this reduced linkage disequilibrium into account. Even greater reductions of linkage disequilibrium can be accomplished by an alternative design that imposes additional generations of random mating, rather than inbreeding, on an F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e378" xlink:type="simple"/></inline-formula>, resulting in more precise localization of QTLs. Additional generations of outcrossing (beyond the F<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e379" xlink:type="simple"/></inline-formula>) will likely magnify deviations of the null allele frequency from 0.5 owing to segregation distortion and/or inadvertent selection. This can be accommodated by application of formulas in <xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref> with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e380" xlink:type="simple"/></inline-formula> estimated from all sites within a genomic window.</p>
        <p>Other experimental designs, such as backcrosses, will not have allele frequencies of 0.5. For these situations the null expected distributions of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e381" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e382" xlink:type="simple"/></inline-formula> can be approximated using the equations presented in <xref ref-type="supplementary-material" rid="pcbi.1002255.s002">Text S1</xref>, although in this case it will be necessary to know the parental origin of the SNP alleles. Similarly, since <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e383" xlink:type="simple"/></inline-formula> can be generalized to an arbitrary number of classes <xref ref-type="bibr" rid="pcbi.1002255-Sokal1">[12]</xref>, one-tailed scenarios (e.g. <xref ref-type="bibr" rid="pcbi.1002255-Ehrenreich2">[9]</xref>) involving comparison to either a theoeretical population or a random sampling of segregants can be addressed in this framework.</p>
      </sec>
    </sec>
    <sec id="s4" sec-type="methods">
      <title>Methods</title>
      <sec id="s4a">
        <title>Sequencing of Yeast Bulks</title>
        <p>To create the bulked DNA pools each segregant was grown overnight in liquid medium to saturation (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e384" xlink:type="simple"/></inline-formula> cells/ml) and equal volumes of each culture were mixed to form cell bulks. Genomic DNA was isolated from the cell bulks and single Illumina DNA sequencing libraries were prepared from each bulk, using standard protocols as described in <xref ref-type="bibr" rid="pcbi.1002255-Magwene1">[28]</xref>. Each bulk DNA pool was sequenced twice using 50 bp reads on an Illumina GAII sequencing instrument. Approximately 15 M reads were generated in each sequencing run. Reads were aligned to the yeast reference genome (obtained from the Saccharomyces Genome Database, January 2010) using the program BWA <xref ref-type="bibr" rid="pcbi.1002255-Li1">[32]</xref> and polymorphic sites were called using SAMtools <xref ref-type="bibr" rid="pcbi.1002255-Li2">[33]</xref>. For each sequencing run, SAMtools was used to create a pileup file giving the alleles at each polymorphic site, from which allele counts were derived using scripts written in Python.</p>
      </sec>
    </sec>
    <sec id="s5">
      <title>Supporting Information</title>
      <supplementary-material id="pcbi.1002255.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.s001" xlink:type="simple">
        <label>Figure S1</label>
        <caption>
          <p><bold>Simulations results for the null distribution of </bold><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e385" xlink:type="simple"/></inline-formula><bold> based on 10,000 simulations with (</bold><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e386" xlink:type="simple"/></inline-formula><bold>, </bold><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e387" xlink:type="simple"/></inline-formula><bold>, </bold><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e388" xlink:type="simple"/></inline-formula><bold>).</bold> The gray histogram represents the observed distribution of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e389" xlink:type="simple"/></inline-formula>, corresponding to <xref ref-type="fig" rid="pcbi-1002255-g001">Figure 1b</xref>. The dashed lines represent log-normal distributions estimated from theoretical expectation (red line) or via the non-parametric approach described in the text (black line). Both the parametric and non-parametric approaches provide good control of type I error (right tail of the distribution).</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002255.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.s002" xlink:type="simple">
        <label>Text S1</label>
        <caption>
          <p>
            <bold>Generalization of theoretical results to include segregation distortion.</bold>
          </p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002255.s003" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002255.s003" xlink:type="simple">
        <label>Text S2</label>
        <caption>
          <p><bold>Miscellaneous information.</bold> This file includes information on: 1) estimation of the parameters of a log-normal distribution from the expected mean and variance of a variable of interest; 2) the expected shape of the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e390" xlink:type="simple"/></inline-formula> around at a QTL; and 3) A summary table of expected and observed means and variances of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002255.e391" xlink:type="simple"/></inline-formula> based on simulations of the null hypothesis (no QTL).</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>We thank Joshua Granek and Debra Murray who helped to generate the yeast BSA data set. We thank Stuart McDonald for conversations and feedback. We thank the Duke University Institute for Genome Sciences &amp; Policy Sequencing Facility for the sequencing of genomic libraries.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pcbi.1002255-Michelmore1">
        <label>1</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Michelmore</surname><given-names>RW</given-names></name><name name-style="western"><surname>Paran</surname><given-names>I</given-names></name><name name-style="western"><surname>Kesseli</surname><given-names>RV</given-names></name></person-group>             <year>1991</year>             <article-title>Identification of markers linked to disease-resistance genes by bulked segregant analysis: a rapid method to detect markers in specific genomic regions by using segregating populations.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>88</volume>             <fpage>9828</fpage>             <lpage>9832</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Ehrenreich1">
        <label>2</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ehrenreich</surname><given-names>IM</given-names></name><name name-style="western"><surname>Gerke</surname><given-names>JP</given-names></name><name name-style="western"><surname>Kruglyak</surname><given-names>L</given-names></name></person-group>             <year>2009</year>             <article-title>Genetic dissection of complex traits in yeast: insights from studies of gene expression and other phenotypes in the byxrm cross.</article-title>             <source>Cold Spring Harb Symp Quant Biol</source>             <volume>74</volume>             <fpage>145</fpage>             <lpage>153</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Winzeler1">
        <label>3</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Winzeler</surname><given-names>EA</given-names></name><name name-style="western"><surname>Richards</surname><given-names>DR</given-names></name><name name-style="western"><surname>Conway</surname><given-names>AR</given-names></name><name name-style="western"><surname>Goldstein</surname><given-names>AL</given-names></name><name name-style="western"><surname>Kalman</surname><given-names>S</given-names></name><etal/></person-group>             <year>1998</year>             <article-title>Direct allelic variation scanning of the yeast genome.</article-title>             <source>Science</source>             <volume>281</volume>             <fpage>1194</fpage>             <lpage>1197</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Borevitz1">
        <label>4</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Borevitz</surname><given-names>JO</given-names></name><name name-style="western"><surname>Liang</surname><given-names>D</given-names></name><name name-style="western"><surname>Plouffe</surname><given-names>D</given-names></name><name name-style="western"><surname>Chang</surname><given-names>HS</given-names></name><name name-style="western"><surname>Zhu</surname><given-names>T</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>Large-scale identification of single-feature polymorphisms in complex genomes.</article-title>             <source>Genome Res</source>             <volume>13</volume>             <fpage>513</fpage>             <lpage>523</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Brauer1">
        <label>5</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Brauer</surname><given-names>MJ</given-names></name><name name-style="western"><surname>Christianson</surname><given-names>CM</given-names></name><name name-style="western"><surname>Pai</surname><given-names>DA</given-names></name><name name-style="western"><surname>Dunham</surname><given-names>MJ</given-names></name></person-group>             <year>2006</year>             <article-title>Mapping novel traits by array-assisted bulk segregant analysis in saccharomyces cerevisiae.</article-title>             <source>Genetics</source>             <volume>173</volume>             <fpage>1813</fpage>             <lpage>1816</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Segr1">
        <label>6</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Segr</surname><given-names>AV</given-names></name><name name-style="western"><surname>Murray</surname><given-names>AW</given-names></name><name name-style="western"><surname>Leu</surname><given-names>JY</given-names></name></person-group>             <year>2006</year>             <article-title>High-resolution mutation mapping reveals parallel experimental evolution in yeast.</article-title>             <source>PLoS Biol</source>             <volume>4</volume>             <fpage>e256</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Boer1">
        <label>7</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Boer</surname><given-names>VM</given-names></name><name name-style="western"><surname>Amini</surname><given-names>S</given-names></name><name name-style="western"><surname>Botstein</surname><given-names>D</given-names></name></person-group>             <year>2008</year>             <article-title>Inuence of genotype and nutrition on survival and metabolism of starving yeast.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>105</volume>             <fpage>6930</fpage>             <lpage>6935</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Demogines1">
        <label>8</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Demogines</surname><given-names>A</given-names></name><name name-style="western"><surname>Smith</surname><given-names>E</given-names></name><name name-style="western"><surname>Kruglyak</surname><given-names>L</given-names></name><name name-style="western"><surname>Alani</surname><given-names>E</given-names></name></person-group>             <year>2008</year>             <article-title>Identification and dissection of a complex dna repair sensitivity phenotype in baker's yeast.</article-title>             <source>PLoS Genet</source>             <volume>4</volume>             <fpage>e1000123</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Ehrenreich2">
        <label>9</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ehrenreich</surname><given-names>IM</given-names></name><name name-style="western"><surname>Torabi</surname><given-names>N</given-names></name><name name-style="western"><surname>Jia</surname><given-names>Y</given-names></name><name name-style="western"><surname>Kent</surname><given-names>J</given-names></name><name name-style="western"><surname>Martis</surname><given-names>S</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>Dissection of genetically complex traits with extremely large pools of yeast segregants.</article-title>             <source>Nature</source>             <volume>464</volume>             <fpage>1039</fpage>             <lpage>1042</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Wenger1">
        <label>10</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Wenger</surname><given-names>JW</given-names></name><name name-style="western"><surname>Schwartz</surname><given-names>K</given-names></name><name name-style="western"><surname>Sherlock</surname><given-names>G</given-names></name></person-group>             <year>2010</year>             <article-title>Bulk segregant analysis by high-throughput sequencing reveals a novel xylose utilization gene from saccharomyces cerevisiae.</article-title>             <source>PLoS Genet</source>             <volume>6</volume>             <fpage>e1000942</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Parts1">
        <label>11</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Parts</surname><given-names>L</given-names></name><name name-style="western"><surname>Cubillos</surname><given-names>FA</given-names></name><name name-style="western"><surname>Warringer</surname><given-names>J</given-names></name><name name-style="western"><surname>Jain</surname><given-names>K</given-names></name><name name-style="western"><surname>Salinas</surname><given-names>F</given-names></name><etal/></person-group>             <year>2011</year>             <article-title>Revealing the genetic structure of a trait by sequencing a population under selection.</article-title>             <source>Genome Res</source>             <volume>21</volume>             <fpage>1131</fpage>             <lpage>1138</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Sokal1">
        <label>12</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Sokal</surname><given-names>RR</given-names></name><name name-style="western"><surname>Rohlf</surname><given-names>FJ</given-names></name></person-group>             <year>1994</year>             <source>Biometry</source>             <publisher-name>W. H. Freeman</publisher-name>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Nadaraya1">
        <label>13</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nadaraya</surname><given-names>EA</given-names></name></person-group>             <year>1964</year>             <article-title>On estimating regression.</article-title>             <source>Theor Probab Appl</source>             <volume>9</volume>             <fpage>141</fpage>             <lpage>142</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Watson1">
        <label>14</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Watson</surname><given-names>GS</given-names></name></person-group>             <year>1964</year>             <article-title>Smooth regression analysis.</article-title>             <source>Sankhaya</source>             <volume>26</volume>             <fpage>175</fpage>             <lpage>184</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Schucany1">
        <label>15</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Schucany</surname><given-names>WR</given-names></name></person-group>             <year>2004</year>             <article-title>Kernel smoothers: An overview of curve estimators for the first graduate course in nonparametric statistics.</article-title>             <source>Statist Sci</source>             <volume>19</volume>             <fpage>663</fpage>             <lpage>675</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Cleveland1">
        <label>16</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Cleveland</surname><given-names>WS</given-names></name></person-group>             <year>1979</year>             <article-title>Robust locally weighted regression and smoothing scatterplots.</article-title>             <source>J Amer Stat Assoc</source>             <volume>74</volume>             <fpage>829</fpage>             <lpage>826</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Mohn1">
        <label>17</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Mohn</surname><given-names>E</given-names></name></person-group>             <year>1979</year>             <article-title>Confidence estimation of measures of location in the log normal distribution.</article-title>             <source>Biometrika</source>             <volume>66</volume>             <fpage>567</fpage>             <lpage>575</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Davies1">
        <label>18</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Davies</surname><given-names>L</given-names></name><name name-style="western"><surname>Gather</surname><given-names>U</given-names></name></person-group>             <year>1993</year>             <article-title>The identification of multiple outliers.</article-title>             <source>J Amer Stat Assoc</source>             <volume>88</volume>             <fpage>782</fpage>             <lpage>792</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Bickel1">
        <label>19</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Bickel</surname><given-names>DR</given-names></name><name name-style="western"><surname>Frühwirth</surname><given-names>R</given-names></name></person-group>             <year>2006</year>             <article-title>On a fast, robust estimator of the mode: Comparisons to other robust estimators with applications.</article-title>             <source>Comput Stat Data An</source>             <volume>50</volume>             <fpage>3500</fpage>             <lpage>3530</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Rousseeuw1">
        <label>20</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Rousseeuw</surname><given-names>PJ</given-names></name><name name-style="western"><surname>Croux</surname><given-names>C</given-names></name></person-group>             <year>1993</year>             <article-title>Alternatives to the median absolute deviation.</article-title>             <source>J Amer Stat Assoc</source>             <volume>88</volume>             <fpage>1273</fpage>             <lpage>1283</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Turin1">
        <label>21</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Turin</surname><given-names>GL</given-names></name></person-group>             <year>1960</year>             <article-title>An introduction to matched filters.</article-title>             <source>IEEE Trans Inform Theory</source>             <volume>6</volume>             <fpage>311</fpage>             <lpage>329</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Enke1">
        <label>22</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Enke</surname><given-names>CG</given-names></name><name name-style="western"><surname>Nieman</surname><given-names>TA</given-names></name></person-group>             <year>1976</year>             <article-title>Signal-to-noise ratio enhancement by least-squares polynomial smoothing.</article-title>             <source>Anal Chem</source>             <volume>48</volume>             <fpage>705</fpage>             <lpage>712A</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Gu1">
        <label>23</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>H</given-names></name><name name-style="western"><surname>Gao</surname><given-names>R</given-names></name></person-group>             <year>1997</year>             <article-title>Resolution of overlapping echoes and constrained matched filter.</article-title>             <source>IEEE Trans Signal Proc</source>             <volume>45</volume>             <fpage>1854</fpage>             <lpage>1857</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Mikl1">
        <label>24</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Mikl</surname><given-names>M</given-names></name><name name-style="western"><surname>Marecek</surname><given-names>R</given-names></name><name name-style="western"><surname>Hlustk</surname><given-names>P</given-names></name><name name-style="western"><surname>Pavlicov</surname><given-names>M</given-names></name><name name-style="western"><surname>Drastich</surname><given-names>A</given-names></name><etal/></person-group>             <year>2008</year>             <article-title>Effects of spatial smoothing on fmri group inferences.</article-title>             <source>Magn Reson Imaging</source>             <volume>26</volume>             <fpage>490</fpage>             <lpage>503</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Benjamini1">
        <label>25</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Benjamini</surname><given-names>Y</given-names></name><name name-style="western"><surname>Hochberg</surname><given-names>Y</given-names></name></person-group>             <year>1995</year>             <article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing.</article-title>             <source>J Roy Statist Sci, B</source>             <volume>57</volume>             <fpage>289</fpage>             <lpage>300</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Benjamini2">
        <label>26</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Benjamini</surname><given-names>Y</given-names></name><name name-style="western"><surname>Yekutieli</surname><given-names>D</given-names></name></person-group>             <year>2001</year>             <article-title>The control of the false discovery rate in multiple testing under dependency.</article-title>             <source>Ann Stat</source>             <volume>29</volume>             <fpage>1165</fpage>             <lpage>1188</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Granek1">
        <label>27</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Granek</surname><given-names>JA</given-names></name><name name-style="western"><surname>Magwene</surname><given-names>PM</given-names></name></person-group>             <year>2010</year>             <article-title>Environmental and genetic determinants of colony morphology in yeast.</article-title>             <source>PLoS Genet</source>             <volume>6</volume>             <fpage>e1000823</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Magwene1">
        <label>28</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Magwene</surname><given-names>PM</given-names></name><name name-style="western"><surname>Kayıkçı</surname><given-names>Ömür</given-names></name><name name-style="western"><surname>Granek</surname><given-names>JA</given-names></name><name name-style="western"><surname>Reininga</surname><given-names>JM</given-names></name><name name-style="western"><surname>Scholl</surname><given-names>Z</given-names></name><etal/></person-group>             <year>2011</year>             <article-title>Outcrossing, mitotic recombination, and life-history trade-offs shape genome evolution in saccharomyces cerevisiae.</article-title>             <source>Proc Natl Acad Sci USA</source>             <volume>108</volume>             <fpage>1987</fpage>             <lpage>1992</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Weber1">
        <label>29</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Weber</surname><given-names>M</given-names></name></person-group>             <year>2006</year>             <article-title>A weighted central limit theorem.</article-title>             <source>Stat Probabil Lett</source>             <volume>76</volume>             <fpage>1482</fpage>             <lpage>1487</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Kimura1">
        <label>30</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kimura</surname><given-names>M</given-names></name><name name-style="western"><surname>Crow</surname><given-names>JF</given-names></name></person-group>             <year>1978</year>             <article-title>Effect of overall phenotypic selection on genetic change at individual loci.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>75</volume>             <fpage>6168</fpage>             <lpage>6171</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Falconer1">
        <label>31</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Falconer</surname><given-names>DS</given-names></name><name name-style="western"><surname>Mackay</surname><given-names>TFC</given-names></name></person-group>             <year>1996</year>             <source>Introduction to quantitative genetics, 4<sup>th</sup> edition</source>             <publisher-name>Longman</publisher-name>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Li1">
        <label>32</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names></name><name name-style="western"><surname>Durbin</surname><given-names>R</given-names></name></person-group>             <year>2009</year>             <article-title>Fast and accurate short read alignment with burrows-wheeler transform.</article-title>             <source>Bioinformatics</source>             <volume>25</volume>             <fpage>1754</fpage>             <lpage>1760</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002255-Li2">
        <label>33</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names></name><name name-style="western"><surname>Handsaker</surname><given-names>B</given-names></name><name name-style="western"><surname>Wysoker</surname><given-names>A</given-names></name><name name-style="western"><surname>Fennell</surname><given-names>T</given-names></name><name name-style="western"><surname>Ruan</surname><given-names>J</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>The sequence alignment/map format and samtools.</article-title>             <source>Bioinformatics</source>             <volume>25</volume>             <fpage>2078</fpage>             <lpage>2079</lpage>          </element-citation>
      </ref>
    </ref-list>
    
  </back>
</article>