<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
  <front>
    <journal-meta><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id><journal-id journal-id-type="pmc">ploscomp</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS Computational Biology</journal-title></journal-title-group><issn pub-type="ppub">1553-734X</issn><issn pub-type="epub">1553-7358</issn><publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher></journal-meta>
    <article-meta><article-id pub-id-type="publisher-id">PCOMPBIOL-D-11-00215</article-id><article-id pub-id-type="doi">10.1371/journal.pcbi.1002284</article-id><article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Computational biology</subject>
            <subj-group>
              <subject>Genomics</subject>
              <subj-group>
                <subject>Genome analysis tools</subject>
                <subj-group>
                  <subject>Gene prediction</subject>
                </subj-group>
              </subj-group>
              <subj-group>
                <subject>Comparative genomics</subject>
              </subj-group>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Computational Biology</subject>
        </subj-group>
      </article-categories><title-group><article-title>Genome Majority Vote Improves Gene Predictions</article-title><alt-title alt-title-type="running-head">Genome Majority Vote Improves Gene Predictions</alt-title></title-group><contrib-group>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Wall</surname>
            <given-names>Michael E.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
          <xref ref-type="aff" rid="aff3">
            <sup>3</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Raghavan</surname>
            <given-names>Sindhu</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="aff4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Cohn</surname>
            <given-names>Judith D.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Dunbar</surname>
            <given-names>John</given-names>
          </name>
          <xref ref-type="aff" rid="aff5">
            <sup>5</sup>
          </xref>
        </contrib>
      </contrib-group><aff id="aff1"><label>1</label><addr-line>Computer, Computational, and Statistical Sciences Division, Los Alamos National Laboratory, Los Alamos, New Mexico, United States of America</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>Center for Nonlinear Studies, Los Alamos National Laboratory, Los Alamos, New Mexico, United States of America</addr-line>       </aff><aff id="aff3"><label>3</label><addr-line>Theoretical Division, Los Alamos National Laboratory, Los Alamos, New Mexico, United States of America</addr-line>       </aff><aff id="aff4"><label>4</label><addr-line>Department of Computer Science, The University of Texas at Austin, Austin, Texas, United States of America</addr-line>       </aff><aff id="aff5"><label>5</label><addr-line>Bioscience Division, Los Alamos National Laboratory, Los Alamos, New Mexico, United States of America</addr-line>       </aff><contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Ouzounis</surname>
            <given-names>Christos A.</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group><aff id="edit1">The Centre for Research and Technology, Hellas, Greece</aff><author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">mewall@lanl.gov</email></corresp>
        <fn fn-type="con">
          <p>Conceived and designed the experiments: JD MEW. Performed the experiments: SR JDC MEW. Analyzed the data: SR JDC MEW. Contributed reagents/materials/analysis tools: SR JDC JD MEW. Wrote the paper: MEW SR JDC JD.</p>
        </fn>
      <fn fn-type="conflict">
        <p>The authors have declared that no competing interests exist.</p>
      </fn></author-notes><pub-date pub-type="collection">
        <month>11</month>
        <year>2011</year>
      </pub-date><pub-date pub-type="epub">
        <day>17</day>
        <month>11</month>
        <year>2011</year>
      </pub-date><volume>7</volume><issue>11</issue><elocation-id>e1002284</elocation-id><history>
        <date date-type="received">
          <day>15</day>
          <month>2</month>
          <year>2011</year>
        </date>
        <date date-type="accepted">
          <day>6</day>
          <month>10</month>
          <year>2011</year>
        </date>
      </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2011</copyright-year><license><license-p>This is an open-access article, free of all copyright, and may be freely reproduced, distributed, transmitted, modified, built upon, or otherwise used by anyone for any lawful purpose. The work is made available under the Creative Commons CC0 public domain dedication.</license-p></license></permissions><abstract>
        <p>Recent studies have noted extensive inconsistencies in gene start sites among orthologous genes in related microbial genomes. Here we provide the first documented evidence that imposing gene start consistency improves the accuracy of gene start-site prediction. We applied an algorithm using a genome majority vote (GMV) scheme to increase the consistency of gene starts among orthologs. We used a set of validated <italic>Escherichia coli</italic> genes as a standard to quantify accuracy. Results showed that the GMV algorithm can correct hundreds of gene prediction errors in sets of five or ten genomes while introducing few errors. Using a conservative calculation, we project that GMV would resolve many inconsistencies and errors in publicly available microbial gene maps. Our simple and logical solution provides a notable advance toward accurate gene maps.</p>
      </abstract><abstract abstract-type="summary">
        <title>Author Summary</title>
        <p>The genetic code tells us precisely how a DNA sequence will be translated into a protein. However, it is more difficult to identify where translation will start and stop in the entire length of an organism's genome sequence. Computer software can predict where the start sites are, and this is successful most of the time; however, errors do occur. We hypothesized that some errors might be corrected by comparing predictions for the genome sequences of closely related organisms. This correction scheme seems especially appropriate for bacterial genomes: not only is protein production in bacteria simpler than in higher organisms, but hundreds of bacterial DNA sequences are now available, and many of these are closely related. To test the hypothesis, we developed a method to detect whether a gene's start site is inconsistent with the majority of equivalent genes in a set of related bacterial genomes. The method then modifies the start if it can be made consistent with the majority of genomes. Our tests show this majority vote method improves the accuracy of gene start sites. Application of the method to existing bacterial genomes should eliminate many inconsistencies and correct a large number of errors.</p>
      </abstract><funding-group><funding-statement>This work was primarily funded by Los Alamos National Laboratory Directed Research and Development program (LDRD) grant 20080138DR. MEW and JDC received additional support from NIH/National Library of Medicine grant R01LM010120, and MEW received additional support from LDRD grant 20110435DR. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts>
        <page-count count="11"/>
      </counts></article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>All of genomics depends on accurate identification of coding regions. Most gene boundaries are predicted using computational methods, and only a tiny fraction have been verified experimentally. Unfortunately, the accuracy of current gene-finding algorithms is not perfect. Error rates for the most common algorithms—Glimmer3 <xref ref-type="bibr" rid="pcbi.1002284-Delcher1">[1]</xref>, GeneMark <xref ref-type="bibr" rid="pcbi.1002284-Besemer1">[2]</xref>, and Prodigal <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>—currently range from 1.5%–17.6% <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>. Gene prediction errors alter protein sequences and intergenic regions (IGRs). Changes in protein sequence influence calculations of similarities, phylogenetic analyses, and can lead to errors in function annotation. Changes in IGRs affect a suite of other predictions, such as operon structure, regulatory motifs, and comparison of regulatory regions among genomes. Changes in gene boundaries also affect microarray design and interpretation of microarray data <xref ref-type="bibr" rid="pcbi.1002284-Dai1">[4]</xref>.</p>
      <p>Gene-prediction error is a well-recognized problem <xref ref-type="bibr" rid="pcbi.1002284-Poptsova1">[5]</xref> but the full extent of gene prediction errors from current computational methods is unknown. Recent studies yielded insight into the problem for bacterial genomes. Pallejà, Harrington, &amp; Bork <xref ref-type="bibr" rid="pcbi.1002284-Pallej1">[6]</xref> found nearly a thousand examples of spurious gene overlaps (a gene stop being downstream of the following gene's start) in 338 bacterial genomes. Recently, we noted inconsistencies in gene start sites among 53% of the orthologous gene sets across the <italic>Burkholderia</italic> genus <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>. Although we expected real biological variation to yield some inconsistencies in gene starts, many inconsistencies for the <italic>Burkholderia</italic> genus included predictions of alternative starts in regions of nearly identical sequence and likely represented errors. We found most of these start site inconsistencies could be resolved by choosing alternative start sites for one or more of the orthologous genes, improving comparisons of IGRs across the genus. We and others have speculated (either implicitly or explicitly) that efforts to improve consistency of gene boundaries among orthologs can also improve the accuracy of gene predictions <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>, <xref ref-type="bibr" rid="pcbi.1002284-Aziz1">[8]</xref>, <xref ref-type="bibr" rid="pcbi.1002284-Pati1">[9]</xref>. However, this hypothesis has not yet been tested.</p>
      <p>Here, we test this idea using a set of validated <italic>Escherichia coli</italic> genes. We provide, for the first time, quantitative evidence showing that consistency increases accuracy. We discuss the significance of our results in the context of gene prediction methods that make use of multiple genomes, and find that our method is distinguished both by its effective use of larger numbers of genomes, by its simplicity and modularity, and by its use of contemporary (not older and error-ridden) gene-predictions. To our knowledge, the method is the only tool available for non-specialists to solve the routine problem of refining the accuracy of extant gene maps in public databases.</p>
    </sec>
    <sec id="s2">
      <title>Results/Discussion</title>
      <sec id="s2a">
        <title>Motivation for improving gene predictions using a majority vote</title>
        <p>Gene finding programs need to evaluate several possible start sites for each gene. The programs occasionally make mistakes and pick the wrong start site. If mistakes for orthologous genes in different genomes are uncorrelated (an unrealistic assumption, but useful as a reference point for algorithm development), then if less than half of the predicted starts are wrong, they might be corrected by a majority vote.</p>
        <p>To formulate the majority vote idea mathematically, consider a set of <italic>N</italic> orthologous genes with experimentally verified start sites that have consistent positions in a multiple sequence alignment. If the probability of a gene finder predicting the wrong start site for any of the orthologs is <italic>e</italic>, and if the predictions for different genomes are independent, then the probability <italic>p<sub>i</sub></italic> of finding <italic>i</italic> errors among all of the orthologs is given by a binomial distribution,<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e001" xlink:type="simple"/><label>(1)</label></disp-formula></p>
        <p>The probability of finding at least one error among all of the orthologs is<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e002" xlink:type="simple"/><label>(2)</label></disp-formula>and the probability of the majority of the orthologs containing an error is<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e003" xlink:type="simple"/><label>(3)</label></disp-formula></p>
        <p>For example, let the probability of predicting an incorrect start site be <italic>e</italic> = 0.05, at the low end of the range of error rates for common gene finders <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>. If the number of orthologous genes <italic>(N)</italic> is 5, the chance of at least one error, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e004" xlink:type="simple"/></inline-formula>, in the ortholog set is 22.6%, but the chance that the majority of orthologs are erroneous (<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e005" xlink:type="simple"/></inline-formula>) is only 0.12%. The mean error rate <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e006" xlink:type="simple"/></inline-formula> is 0.25 across all number of errors for five orthologs, and is 0.0035 across <italic>i</italic>&gt;2. In this scenario, choosing a globally consistent site where a majority of the original predictions coincide is expected to correct <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e007" xlink:type="simple"/></inline-formula> of the inconsistent ortholog sets, and to correct 1−0.0035/0.25 = 98.6% of the individual genes that have prediction errors.</p>
        <p>The above model illustrates that typical gene prediction error rates can lead to double-digit inconsistencies in ortholog sets (e.g. in the above case, a 5% error rate led to a 22.6% inconsistency rate). It also illustrates the ability of a majority vote to decrease errors and thereby increase accuracy through increasing consistency. The increase in accuracy requires that the error rate for a single gene start be less than 50%. This prerequisite is satisfied by modern gene calling software, for which reported error rates range from 1.5% to 17.6% <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>.</p>
      </sec>
      <sec id="s2b">
        <title>Genome Majority Vote algorithm</title>
        <p>Although the above model gives clear and quantitative insight into how comparative genomics might improve the accuracy of gene maps, it is merely a reference point and does not consider the important effects of real biological variation and correlated errors. In our previous study of gene start site consistency in the <italic>Burkholderia</italic> genus <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>, we noted that ortholog sets in which a majority of the start sites did not coincide were likely to represent biological variation, whereas ortholog sets in which a majority of the start sites coincided were likely to represent errors. To determine whether a majority vote scheme might decrease start site errors in real gene maps, we developed a Genome Majority Vote (GMV) algorithm and applied it to a conservative test case: gene maps from <italic>E.coli</italic> and close relatives.</p>
        <p>The GMV algorithm works as follows. For a given set of orthologous genes, if the positions of the start sites already coincide in a multiple sequence alignment, they are accepted. If they do not coincide, a start position is sought which is consistent for the majority of the genes and for which there is a reasonable alternative start site for the remaining genes in the set. If such a position is found, it is accepted, and the predictions are changed for the outlying genes. Otherwise, no start site prediction is made for the ortholog set.</p>
        <p>We implemented GMV in the pipeline illustrated in <xref ref-type="fig" rid="pcbi-1002284-g001">Fig. 1</xref>. The input of the pipeline is a set of genome FASTA files. The output is a set of gene predictions for each genome after enforcing consistency using a genome majority vote (GMV) algorithm. A typical GMV correction is illustrated in <xref ref-type="fig" rid="pcbi-1002284-g002">Fig. 2</xref>. Details of the pipeline are described in the <xref ref-type="sec" rid="s3">Methods</xref> section.</p>
        <fig id="pcbi-1002284-g001" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.g001</object-id>
          <label>Figure 1</label>
          <caption>
            <title>Flow diagram for the pipeline implementing the Genome Majority Vote algorithm.</title>
            <p>Individual steps A–E are explained in the text (<xref ref-type="sec" rid="s3">Methods</xref>).</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.g001" xlink:type="simple"/>
        </fig>
        <fig id="pcbi-1002284-g002" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.g002</object-id>
          <label>Figure 2</label>
          <caption>
            <title>Example of a GMV modification of gene starts that is typical in terms of ortholog sequence identity, change in the length of the gene, and the start codon before and after the change.</title>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.g002" xlink:type="simple"/>
        </fig>
      </sec>
      <sec id="s2c">
        <title>Conservative approach to evaluation of the GMV algorithm</title>
        <p>Any gene-calling software can in principle be used as a front end to provide input gene calls to GMV. Here we used Prodigal <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref> predicted gene maps for <italic>E. coli</italic> and close relatives as a starting point for GMV evaluation. This choice solved two problems. First, Prodigal conveniently provided a list of reasonable alternative start sites for each gene, simplifying the comparison and reassignment of possible start sites among genomes. Second, Prodigal provided gene maps with fewer prediction errors compared to existing GenBank annotations that were obtained from older, more error prone versions of gene finders like Glimmer2. Prodigal is reportedly the most robust gene finder for diverse genomes <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>. Therefore, our use of Prodigal gene maps instead of Glimmer3 <xref ref-type="bibr" rid="pcbi.1002284-Delcher1">[1]</xref>, GeneMark <xref ref-type="bibr" rid="pcbi.1002284-Besemer1">[2]</xref>, or older annotations appeared to be the most logical and conservative approach.</p>
        <p>Because Prodigal gene maps are likely to be more accurate than most of the gene maps currently in GenBank <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>, using Prodigal gene maps to test the performance of GMV in correcting errors should provide conservative estimates of performance. We reported previously that Genbank maps for <italic>Burkholderia</italic> species were more inconsistent than Prodigal maps <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>; we show similar results in a later section of this paper for a set of <italic>E. coli</italic> genomes of comparable diversity. Use of more error-prone gene maps—either from other gene finders or from genomes that are more problematic for gene prediction—would be expected to inflate the number of observed inconsistencies among orthologs and the projected impact of applying the GMV algorithm.</p>
        <p>To evaluate the performance of the algorithm, we created eight genome test sets that varied in size and diversity (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s010">Table S1</xref>, Supplementary Figs. S<sub>1</sub>–S<sub>8</sub>). Each set contained either 5 or 10 genomes and included <italic>E. coli</italic> K-12 MG1655 as the reference genome. The sets represented low, medium, high, or very high diversity. We used a set of 871 experimentally validated <italic>Escherichia coli</italic> K12 MG1655 genes downloaded from the EcoGene web site <xref ref-type="bibr" rid="pcbi.1002284-Rudd1">[10]</xref> (<ext-link ext-link-type="uri" xlink:href="http://ecogene.org" xlink:type="simple">http://ecogene.org</ext-link>) as a standard to determine error rates. A gene prediction was classified as erroneous if the translational start site differed from that of the validated gene; no errors were detected in translational stop sites.</p>
      </sec>
      <sec id="s2d">
        <title>GMV increases gene start site consistency</title>
        <p>Among the eight test sets, 5.9% to 61.8% of the ortholog sets had inconsistencies. The majority vote rule improved consistency for 13.2% to 51.9% of the ortholog sets (<xref ref-type="table" rid="pcbi-1002284-t001">Table 1</xref>). The impact varied depending on the number and diversity of genomes in the test sets. The impact was highest for the medium diversity sets.</p>
        <table-wrap id="pcbi-1002284-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t001</object-id><label>Table 1</label><caption>
            <title>Consistency statistics for ortholog sets.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t001" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="4" rowspan="1">5 genomes<xref ref-type="table-fn" rid="nt101">a</xref></td>
                <td align="left" colspan="4" rowspan="1">10 genomes<xref ref-type="table-fn" rid="nt101">a</xref></td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total # of ortholog sets generated in the pipeline</td>
                <td align="left" colspan="1" rowspan="1">3633</td>
                <td align="left" colspan="1" rowspan="1">2446</td>
                <td align="left" colspan="1" rowspan="1">1414</td>
                <td align="left" colspan="1" rowspan="1">988</td>
                <td align="left" colspan="1" rowspan="1">3271</td>
                <td align="left" colspan="1" rowspan="1">2133</td>
                <td align="left" colspan="1" rowspan="1">1317</td>
                <td align="left" colspan="1" rowspan="1">380</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which Prodigal starts were initially inconsistent<xref ref-type="table-fn" rid="nt102">b</xref></td>
                <td align="left" colspan="1" rowspan="1">213 (5.9%)</td>
                <td align="left" colspan="1" rowspan="1">536 (21.9%)</td>
                <td align="left" colspan="1" rowspan="1">574 (40.6%)</td>
                <td align="left" colspan="1" rowspan="1">547 (55.4%)</td>
                <td align="left" colspan="1" rowspan="1">251 (7.7%)</td>
                <td align="left" colspan="1" rowspan="1">614 (28.8%)</td>
                <td align="left" colspan="1" rowspan="1">634 (48.1%)</td>
                <td align="left" colspan="1" rowspan="1">235 (61.8%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which Prodigal starts were already consistent<xref ref-type="table-fn" rid="nt102">b</xref></td>
                <td align="left" colspan="1" rowspan="1">3420 (94.1%)</td>
                <td align="left" colspan="1" rowspan="1">1910 (78.1%)</td>
                <td align="left" colspan="1" rowspan="1">840 (59.4%)</td>
                <td align="left" colspan="1" rowspan="1">441 (44.6%)</td>
                <td align="left" colspan="1" rowspan="1">3020 (92.3%)</td>
                <td align="left" colspan="1" rowspan="1">1519 (71.2%)</td>
                <td align="left" colspan="1" rowspan="1">683 (51.9%)</td>
                <td align="left" colspan="1" rowspan="1">145 (38.2%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of inconsistent ortholog sets that were made consistent by GMV<xref ref-type="table-fn" rid="nt103">c</xref></td>
                <td align="left" colspan="1" rowspan="1">74 (34.7%)</td>
                <td align="left" colspan="1" rowspan="1">278 (51.9%)</td>
                <td align="left" colspan="1" rowspan="1">204 (35.5%)</td>
                <td align="left" colspan="1" rowspan="1">74 (16.8%)</td>
                <td align="left" colspan="1" rowspan="1">89 (35.5%)</td>
                <td align="left" colspan="1" rowspan="1">286 (46.6%)</td>
                <td align="left" colspan="1" rowspan="1">227 (35.8%)</td>
                <td align="left" colspan="1" rowspan="1">31 (13.2%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with consistent starts after GMV<xref ref-type="table-fn" rid="nt102">b</xref></td>
                <td align="left" colspan="1" rowspan="1">3494 (96.2%)</td>
                <td align="left" colspan="1" rowspan="1">2188 (89.5%)</td>
                <td align="left" colspan="1" rowspan="1">1044 (73.8%)</td>
                <td align="left" colspan="1" rowspan="1">515 (52.1%)</td>
                <td align="left" colspan="1" rowspan="1">3109 (95.0%)</td>
                <td align="left" colspan="1" rowspan="1">1805 (84.6%)</td>
                <td align="left" colspan="1" rowspan="1">910 (69.1%)</td>
                <td align="left" colspan="1" rowspan="1">176 (46.3%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with at least one consistent start<xref ref-type="table-fn" rid="nt102">b</xref></td>
                <td align="left" colspan="1" rowspan="1">3626 (99.8%)</td>
                <td align="left" colspan="1" rowspan="1">2428 (99.3%)</td>
                <td align="left" colspan="1" rowspan="1">1326 (93.8%)</td>
                <td align="left" colspan="1" rowspan="1">863 (87.3%)</td>
                <td align="left" colspan="1" rowspan="1">3269 (99.9%)</td>
                <td align="left" colspan="1" rowspan="1">2098 (98.4%)</td>
                <td align="left" colspan="1" rowspan="1">1215 (92.3%)</td>
                <td align="left" colspan="1" rowspan="1">310 (81.6%)</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt101">
              <label>a</label>
              <p>The genomes in each set are listed in Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s010">Table S1</xref>.</p>
            </fn>
            <fn id="nt102">
              <label>b</label>
              <p>Percentage is with respect to total # of ortholog sets generated in the pipeline.</p>
            </fn>
            <fn id="nt103">
              <label>c</label>
              <p>Percentage is with respect to # of ortholog sets for which Prodigal starts were initially inconsistent.</p>
            </fn>
          </table-wrap-foot></table-wrap>
        <p>The maximum level of consistency that could theoretically be imposed ranged from 81.6% to 99.9% (<xref ref-type="table" rid="pcbi-1002284-t001">Table 1</xref>, last row). The difference between the theoretical maximum and the levels achieved by GMV ranged from 3.6% to 35.2%; these differences increased monotonically with the diversity of the genome test sets. The range of differences encompasses a value of 18% calculated from the results for five <italic>Burkholderia</italic> genomes <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>, demonstrating the general consistency of the previous results with the results presented here. The ortholog sets not revised by GMV involve choosing alternative start sites for the majority of genes. These ortholog sets might represent real biological variation in the location of gene start sites, as we discussed previously <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>. In contrast, GenePRIMP <xref ref-type="bibr" rid="pcbi.1002284-Pati1">[9]</xref> also uses homologs to identify potential start site prediction errors, but does not appear to have a mechanism that could distinguish errors from true biological variation.</p>
      </sec>
      <sec id="s2e">
        <title>GMV changes typically preserve start codons</title>
        <p>Among the ortholog sets revised by GMV, <named-content content-type="gene" xlink:type="simple">ATG</named-content> was the most common start codon, as expected (<xref ref-type="table" rid="pcbi-1002284-t002">Table 2</xref>). We calculated statistics for start codon changes for the medium and high diversity genome test sets, which accounted for the largest number of revised ortholog sets. The start codon identity was preserved in 69%–75% of GMV revisions. The start codon distribution for ortholog sets before revision by GMV, calculated as the mean among four test sets, was approximately 87% <named-content content-type="gene" xlink:type="simple">ATG</named-content>, 9% <named-content content-type="gene" xlink:type="simple">GTG</named-content>, and 4% <named-content content-type="gene" xlink:type="simple">TTG</named-content>. After revision, the distribution was 79.5% <named-content content-type="gene" xlink:type="simple">ATG</named-content>, 14.5% <named-content content-type="gene" xlink:type="simple">GTG</named-content>, and 6% <named-content content-type="gene" xlink:type="simple">TTG</named-content>.</p>
        <table-wrap id="pcbi-1002284-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t002</object-id><label>Table 2</label><caption>
            <title>Codon change statistics for GMV start site changes in medium and high diversity genome test sets.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t002" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="2" rowspan="1">5 genomes</td>
                <td align="left" colspan="2" rowspan="1">10 genomes</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Codon before change</td>
                <td align="left" colspan="1" rowspan="1">Codon after change</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">243</td>
                <td align="left" colspan="1" rowspan="1">184</td>
                <td align="left" colspan="1" rowspan="1">354</td>
                <td align="left" colspan="1" rowspan="1">249</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">47</td>
                <td align="left" colspan="1" rowspan="1">26</td>
                <td align="left" colspan="1" rowspan="1">66</td>
                <td align="left" colspan="1" rowspan="1">41</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">16</td>
                <td align="left" colspan="1" rowspan="1">10</td>
                <td align="left" colspan="1" rowspan="1">33</td>
                <td align="left" colspan="1" rowspan="1">26</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">31</td>
                <td align="left" colspan="1" rowspan="1">15</td>
                <td align="left" colspan="1" rowspan="1">42</td>
                <td align="left" colspan="1" rowspan="1">22</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">8</td>
                <td align="left" colspan="1" rowspan="1">5</td>
                <td align="left" colspan="1" rowspan="1">5</td>
                <td align="left" colspan="1" rowspan="1">7</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">1</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">ATG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">9</td>
                <td align="left" colspan="1" rowspan="1">10</td>
                <td align="left" colspan="1" rowspan="1">14</td>
                <td align="left" colspan="1" rowspan="1">11</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">GTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">3</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">7</td>
                <td align="left" colspan="1" rowspan="1">5</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <named-content content-type="gene" xlink:type="simple">TTG</named-content>
                </td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">1</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total Changes</td>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">357</td>
                <td align="left" colspan="1" rowspan="1">252</td>
                <td align="left" colspan="1" rowspan="1">522</td>
                <td align="left" colspan="1" rowspan="1">363</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Same codon</td>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">251</td>
                <td align="left" colspan="1" rowspan="1">189</td>
                <td align="left" colspan="1" rowspan="1">360</td>
                <td align="left" colspan="1" rowspan="1">257</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Different codon</td>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">106</td>
                <td align="left" colspan="1" rowspan="1">63</td>
                <td align="left" colspan="1" rowspan="1">162</td>
                <td align="left" colspan="1" rowspan="1">106</td>
              </tr>
            </tbody>
          </table></alternatives></table-wrap>
      </sec>
      <sec id="s2f">
        <title>GMV increases gene prediction accuracy</title>
        <p>Before applying GMV, we first note evidence of an association between consistency and accuracy of gene start sites among orthologs. Among ortholog sets with consistent start sites, the <italic>E. coli</italic> start site accuracy ( = 100% – [error rate]) ranged from 96% to 100% (<xref ref-type="table" rid="pcbi-1002284-t003">Table 3</xref>, row 5) in low to high diversity genome test sets. The start site accuracy was lower for orthologs with inconsistent start sites, ranging from 69.2% to 91.8% (<xref ref-type="table" rid="pcbi-1002284-t003">Table 3</xref>, row 6). Overall, the error rate for consistent start sites was about 15% lower than the error rate for inconsistent start sites (<xref ref-type="table" rid="pcbi-1002284-t003">Table 3</xref>, subtract row 6 from row 5 and calculate the mean). This observation supports the notion that the pursuit of consistency can improve accuracy.</p>
        <table-wrap id="pcbi-1002284-t003" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t003</object-id><label>Table 3</label><caption>
            <title>Validation statistics for ortholog sets.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t003-3" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t003" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="4" rowspan="1">5 genomes</td>
                <td align="left" colspan="4" rowspan="1">10 genomes</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which <italic>E. coli</italic> validation was available</td>
                <td align="left" colspan="1" rowspan="1">833</td>
                <td align="left" colspan="1" rowspan="1">683</td>
                <td align="left" colspan="1" rowspan="1">457</td>
                <td align="left" colspan="1" rowspan="1">274</td>
                <td align="left" colspan="1" rowspan="1">800</td>
                <td align="left" colspan="1" rowspan="1">618</td>
                <td align="left" colspan="1" rowspan="1">414</td>
                <td align="left" colspan="1" rowspan="1">129</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which <italic>E. coli</italic> validation was available and for which Prodigal predictions were already consistent<xref ref-type="table-fn" rid="nt104">a</xref></td>
                <td align="left" colspan="1" rowspan="1">825 (99.0%)</td>
                <td align="left" colspan="1" rowspan="1">613 (89.8%)</td>
                <td align="left" colspan="1" rowspan="1">382 (83.6%)</td>
                <td align="left" colspan="1" rowspan="1">245 (89.4%)</td>
                <td align="left" colspan="1" rowspan="1">787 (98.4%)</td>
                <td align="left" colspan="1" rowspan="1">546 (88.3%)</td>
                <td align="left" colspan="1" rowspan="1">329 (79.5%)</td>
                <td align="left" colspan="1" rowspan="1">107 (82.9%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which <italic>E. coli</italic> validation was available and for which Prodigal predictions were inconsistent<xref ref-type="table-fn" rid="nt104">a</xref></td>
                <td align="left" colspan="1" rowspan="1">8 (0.96%)</td>
                <td align="left" colspan="1" rowspan="1">70 (10.2%)</td>
                <td align="left" colspan="1" rowspan="1">75 (16.4%)</td>
                <td align="left" colspan="1" rowspan="1">29 (10.6%)</td>
                <td align="left" colspan="1" rowspan="1">13 (1.63%)</td>
                <td align="left" colspan="1" rowspan="1">72 (11.7%)</td>
                <td align="left" colspan="1" rowspan="1">85 (20.5%)</td>
                <td align="left" colspan="1" rowspan="1">22 (17.1%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with start sites matching a validated <italic>E. coli</italic> start<xref ref-type="table-fn" rid="nt104">a</xref></td>
                <td align="left" colspan="1" rowspan="1">799 (95.9%)</td>
                <td align="left" colspan="1" rowspan="1">664 (97.2%)</td>
                <td align="left" colspan="1" rowspan="1">444 (97.2%)</td>
                <td align="left" colspan="1" rowspan="1">271 (98.9%)</td>
                <td align="left" colspan="1" rowspan="1">769 (96.1%)</td>
                <td align="left" colspan="1" rowspan="1">602 (97.4%)</td>
                <td align="left" colspan="1" rowspan="1">406 (98.1%)</td>
                <td align="left" colspan="1" rowspan="1">126 (97.7%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with start sites matching a validated <italic>E. coli</italic> start and for which Prodigal predictions were already consistent<xref ref-type="table-fn" rid="nt105">b</xref></td>
                <td align="left" colspan="1" rowspan="1">792 (96.0%)</td>
                <td align="left" colspan="1" rowspan="1">609 (99.3%)</td>
                <td align="left" colspan="1" rowspan="1">381 (99.7%)</td>
                <td align="left" colspan="1" rowspan="1">245 (100%)</td>
                <td align="left" colspan="1" rowspan="1">760 (96.6%)</td>
                <td align="left" colspan="1" rowspan="1">544 (99.6%)</td>
                <td align="left" colspan="1" rowspan="1">328 (99.7%)</td>
                <td align="left" colspan="1" rowspan="1">107 (100%)</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with start sites matching a validated <italic>E. coli</italic> start and for which Prodigal predictions were inconsistent<xref ref-type="table-fn" rid="nt106">c</xref></td>
                <td align="left" colspan="1" rowspan="1">7 (87.5%)</td>
                <td align="left" colspan="1" rowspan="1">55 (78.6%)</td>
                <td align="left" colspan="1" rowspan="1">63 (84.0%)</td>
                <td align="left" colspan="1" rowspan="1">26 (89.7%)</td>
                <td align="left" colspan="1" rowspan="1">9 (69.2%)</td>
                <td align="left" colspan="1" rowspan="1">58 (80.6%)</td>
                <td align="left" colspan="1" rowspan="1">78 (91.8%)</td>
                <td align="left" colspan="1" rowspan="1">19 (86.3%)</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt104">
              <label>a</label>
              <p>Percentage is with respect to total # of ortholog sets.</p>
            </fn>
            <fn id="nt105">
              <label>b</label>
              <p>Percentage is with respect to # of ortholog sets for which <italic>E. coli</italic> validation was available and for which all Prodigal predictions were already consistent. This represents accuracy of the consistent subset.</p>
            </fn>
            <fn id="nt106">
              <label>c</label>
              <p>Percentage is with respect to # of ortholog sets for which <italic>E. coli</italic> validation was available and for which Prodigal predictions were inconsistent. This represents accuracy of the inconsistent subset.</p>
            </fn>
          </table-wrap-foot></table-wrap>
        <p>The GMV pipeline corrected the most errors when applied to the high and medium diversity test sets (<xref ref-type="table" rid="pcbi-1002284-t004">Table 4</xref>). Error rates (<italic>i.e.</italic> rates of inappropriate corrections) were lower for the high diversity sets, and more corrections were produced for the medium diversity sets. In the high diversity 5-genome test set, GMV yielded 41 modifications in <italic>E. coli</italic>, which included 13 genes with validated start sites. For the 13 ground truth positives (GP), GMV corrected 11 errors but also incorrectly shifted 2 previously correct start sites. In other words, GMV yielded 11 true positives (<italic>TP</italic>) and 2 false positives (<italic>FP</italic>) for this data set. The sensitivity was <italic>S</italic> = <italic>TP</italic>/<italic>GP</italic> = 0.846, and the error rate was <italic>E</italic> = <italic>FP</italic>/(<italic>TP</italic>+<italic>FP</italic>) = 0.154. Applying this error rate to all 41 modified starts in <italic>E. coli</italic> yields an estimated 35 correct changes and 6 incorrect changes (<xref ref-type="fig" rid="pcbi-1002284-g003">Fig. 3</xref>). For the other genomes in the test set, GMV changed a total of 252 start sites, 88 of which were in ortholog sets for which <italic>E. coli</italic> validation information was available. The positions of 82 of the 88 changes coincided with a validated <italic>E. coli</italic> start site, while the other 6 were erroneous. These numbers yield an error rate of 6/82 = 0.07. This is about half the error rate calculated for <italic>E. coli</italic> alone and may be more representative because it was derived from a larger sample size (n = 82, compared to n = 13); if the <italic>E. coli</italic> predictions had yielded 1 false positive instead of 2, the <italic>E. coli</italic> rate would also have been 0.07, illustrating the sensitivity of statistics to small changes when the sample size is small. Applying the 0.07 error rate to all 252 changes that GMV made, we expect 235 of these to be correct and 17 to be incorrect changes. With the 10-genome high-diversity test set, there were more modifications, and the error rate was lower: GMV changed a total of 363 start sites, and only 3 of the changes were predicted to be incorrect. For the medium-diversity test sets, there were more modifications at the cost of a higher error rate: with 5 genomes, GMV changed a total of 357 start sites, 30 of which are predicted to be incorrect; with 10 genomes, 522 start sites were changed, 54 of which are predicted to be incorrect.</p>
        <fig id="pcbi-1002284-g003" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.g003</object-id>
          <label>Figure 3</label>
          <caption>
            <title>Impact of gene prediction changes in high diversity genome sets.</title>
            <p>Number of correct and incorrect changes are estimated using validated starts in <italic>E. coli</italic>, as described in the text. A) <italic>E. coli</italic>; B) All genomes.</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.g003" xlink:type="simple"/>
        </fig>
        <table-wrap id="pcbi-1002284-t004" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t004</object-id><label>Table 4</label><caption>
            <title>Validation statistics for GMV algorithm corrections to Prodigal gene maps.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t004-4" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t004" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="4" rowspan="1">5 genomes</td>
                <td align="left" colspan="4" rowspan="1">10 genomes</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
                <td align="left" colspan="1" rowspan="1">Low</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Very High</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets with an incorrect <italic>E. coli</italic> start (<italic>GP</italic>)</td>
                <td align="left" colspan="1" rowspan="1">34</td>
                <td align="left" colspan="1" rowspan="1">19</td>
                <td align="left" colspan="1" rowspan="1">13</td>
                <td align="left" colspan="1" rowspan="1">3</td>
                <td align="left" colspan="1" rowspan="1">31</td>
                <td align="left" colspan="1" rowspan="1">16</td>
                <td align="left" colspan="1" rowspan="1">8</td>
                <td align="left" colspan="1" rowspan="1">3</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of corrected validated starts in <italic>E. coli</italic> (<italic>TP</italic>)</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">9</td>
                <td align="left" colspan="1" rowspan="1">11</td>
                <td align="left" colspan="1" rowspan="1">3</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">8</td>
                <td align="left" colspan="1" rowspan="1">7</td>
                <td align="left" colspan="1" rowspan="1">3</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"># of <italic>E. coli</italic> errors introduced (<italic>FP</italic>)</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">2</td>
                <td align="left" colspan="1" rowspan="1">2</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Error Rate (<italic>E</italic><xref ref-type="table-fn" rid="nt107">a</xref>)</td>
                <td align="left" colspan="1" rowspan="1">1.00</td>
                <td align="left" colspan="1" rowspan="1">0.182</td>
                <td align="left" colspan="1" rowspan="1">0.154</td>
                <td align="left" colspan="1" rowspan="1">0.25</td>
                <td align="left" colspan="1" rowspan="1">0.5<xref ref-type="table-fn" rid="nt108">b</xref></td>
                <td align="left" colspan="1" rowspan="1">0.111<xref ref-type="table-fn" rid="nt108">b</xref></td>
                <td align="left" colspan="1" rowspan="1">0.125<xref ref-type="table-fn" rid="nt108">b</xref></td>
                <td align="left" colspan="1" rowspan="1">0.25<xref ref-type="table-fn" rid="nt108">b</xref></td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Sensitivity (<italic>S</italic><xref ref-type="table-fn" rid="nt109">c</xref>)</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0.474</td>
                <td align="left" colspan="1" rowspan="1">0.846</td>
                <td align="left" colspan="1" rowspan="1">1.0</td>
                <td align="left" colspan="1" rowspan="1">0.032</td>
                <td align="left" colspan="1" rowspan="1">0.5</td>
                <td align="left" colspan="1" rowspan="1">0.875</td>
                <td align="left" colspan="1" rowspan="1">1.00</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total # of changes in <italic>E. coli</italic></td>
                <td align="left" colspan="1" rowspan="1">13</td>
                <td align="left" colspan="1" rowspan="1">51</td>
                <td align="left" colspan="1" rowspan="1">41</td>
                <td align="left" colspan="1" rowspan="1">12</td>
                <td align="left" colspan="1" rowspan="1">9</td>
                <td align="left" colspan="1" rowspan="1">38</td>
                <td align="left" colspan="1" rowspan="1">21</td>
                <td align="left" colspan="1" rowspan="1">4</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total # of changes in all genomes</td>
                <td align="left" colspan="1" rowspan="1">92</td>
                <td align="left" colspan="1" rowspan="1">357</td>
                <td align="left" colspan="1" rowspan="1">252</td>
                <td align="left" colspan="1" rowspan="1">88</td>
                <td align="left" colspan="1" rowspan="1">169</td>
                <td align="left" colspan="1" rowspan="1">522</td>
                <td align="left" colspan="1" rowspan="1">363</td>
                <td align="left" colspan="1" rowspan="1">40</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total # of changes that agree with a validated start</td>
                <td align="left" colspan="1" rowspan="1">9</td>
                <td align="left" colspan="1" rowspan="1">76</td>
                <td align="left" colspan="1" rowspan="1">82</td>
                <td align="left" colspan="1" rowspan="1">31</td>
                <td align="left" colspan="1" rowspan="1">20</td>
                <td align="left" colspan="1" rowspan="1">114</td>
                <td align="left" colspan="1" rowspan="1">126</td>
                <td align="left" colspan="1" rowspan="1">28</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Total # of changes that disagree with a validated start</td>
                <td align="left" colspan="1" rowspan="1">4</td>
                <td align="left" colspan="1" rowspan="1">7</td>
                <td align="left" colspan="1" rowspan="1">6</td>
                <td align="left" colspan="1" rowspan="1">1</td>
                <td align="left" colspan="1" rowspan="1">4</td>
                <td align="left" colspan="1" rowspan="1">15</td>
                <td align="left" colspan="1" rowspan="1">0</td>
                <td align="left" colspan="1" rowspan="1">0</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt107">
              <label>a</label>
              <p><italic>E = FP</italic>/(<italic>TP</italic>+<italic>FP</italic>), where <italic>TP</italic> = number of true positives (second row), and <italic>FP</italic> = number of false positives (third row).</p>
            </fn>
            <fn id="nt108">
              <label>b</label>
              <p>Estimated by adding one additional false positive to obtain a nonzero value.</p>
            </fn>
            <fn id="nt109">
              <label>c</label>
              <p><italic>S = TP/GP</italic>, where <italic>TP = </italic>number of true positives (second row), and <italic>GP = </italic>number of ground truth positives (first row).</p>
            </fn>
          </table-wrap-foot></table-wrap>
      </sec>
      <sec id="s2g">
        <title>Projected impact on consistency of microbial ortholog sets</title>
        <p>To estimate the broader impact of GMV, we calculated the increase in consistency for the medium and high diversity genome test sets with either 5 or 10 genomes and then applied these rates to 39 genera. For each test set we obtained the number of genes <italic>m<sub>k</sub></italic> predicted by Prodigal for each genome <italic>k</italic> and selected the smallest number, <italic>M</italic> = min(<italic>m</italic><sub>1</sub>, <italic>m</italic><sub>2</sub>, …, <italic>m<sub>k</sub></italic>). The value of <italic>M</italic> corresponds to the maximum possible number of ortholog sets for a genome set. Values of <italic>M</italic> are given in <xref ref-type="table" rid="pcbi-1002284-t005">Table 5</xref>. Next, we calculated the ortholog set yield <italic>Y</italic> = <italic>O</italic>/<italic>M</italic>, where <italic>O</italic> is the actual number of ortholog sets obtained for each genome set (see first row of <xref ref-type="table" rid="pcbi-1002284-t001">Table 1</xref>). The yield for medium diversity was <italic>Y</italic>≈1/2, and for high diversity <italic>Y</italic>≈1/3, roughly independent of the number of genomes in the set (<xref ref-type="table" rid="pcbi-1002284-t005">Table 5</xref>). Finally we calculated the increase in consistency after running GMV, which ranged from <italic>I</italic> = 11.4% to <italic>I</italic> = 17.2%, calculated as a percentage of the number of actual ortholog sets (<xref ref-type="table" rid="pcbi-1002284-t005">Table 5</xref>). To estimate the number of ortholog sets with increased consistency, <italic>n<sub>I</sub></italic>, after running GMV, we used the equation<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e008" xlink:type="simple"/><label>(4)</label></disp-formula></p>
        <table-wrap id="pcbi-1002284-t005" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t005</object-id><label>Table 5</label><caption>
            <title>Ortholog set yield calculated for medium and high diversity genome test sets.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t005-5" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t005" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="2" rowspan="1">5 genomes</td>
                <td align="left" colspan="2" rowspan="1">10 genomes</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
                <td align="left" colspan="1" rowspan="1">Medium</td>
                <td align="left" colspan="1" rowspan="1">High</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">Maximum possible # ortholog sets, <italic>M</italic></td>
                <td align="left" colspan="1" rowspan="1">4282</td>
                <td align="left" colspan="1" rowspan="1">4332</td>
                <td align="left" colspan="1" rowspan="1">4151</td>
                <td align="left" colspan="1" rowspan="1">3710</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Ortholog set yield, <italic>Y</italic><xref ref-type="table-fn" rid="nt110">a</xref></td>
                <td align="left" colspan="1" rowspan="1">57.1%</td>
                <td align="left" colspan="1" rowspan="1">32.6%</td>
                <td align="left" colspan="1" rowspan="1">51.3%</td>
                <td align="left" colspan="1" rowspan="1">35.4%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Increase in consistency after applying GMV, <italic>I</italic><xref ref-type="table-fn" rid="nt111">b</xref></td>
                <td align="left" colspan="1" rowspan="1">11.4%</td>
                <td align="left" colspan="1" rowspan="1">14.4%</td>
                <td align="left" colspan="1" rowspan="1">13.4%</td>
                <td align="left" colspan="1" rowspan="1">17.2%</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt110">
              <label>a</label>
              <p>Calculated as percentage of <italic>M</italic> using values from the first row of <xref ref-type="table" rid="pcbi-1002284-t001">Table 1</xref>.</p>
            </fn>
            <fn id="nt111">
              <label>b</label>
              <p>Calculated as percentage of the number actual ortholog sets by subtracting the third from the fifth row of <xref ref-type="table" rid="pcbi-1002284-t001">Table 1</xref>.</p>
            </fn>
          </table-wrap-foot></table-wrap>
        <p>We identified suitable target genomes from a list of finished microbial genomes from the Integrated Microbial Genomes resource (<ext-link ext-link-type="uri" xlink:href="http://img.jgi.doe.gov/cgi-bin/pub/main.cgi" xlink:type="simple">http://img.jgi.doe.gov/cgi-bin/pub/main.cgi</ext-link>). We identified 39 genera, each containing a minimum number of 5 finished genomes, as likely targets for our method (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s011">Table S2</xref>). We calculated <italic>M</italic> for each genus and obtained a conservative estimate of the total number of ortholog sets made consistent for each genus using Equation (4) with the lowest values <italic>Y</italic> = 0.326 and <italic>I</italic> = 0.114 from <xref ref-type="table" rid="pcbi-1002284-t005">Table 5</xref>. The total estimated number of ortholog sets made consistent for 467 genomes was about 4,000.</p>
        <p>Although the precise values of <italic>Y</italic> and <italic>I</italic> will differ depending on the genus, using the above values is reasonable for our conservative rough estimate. In <xref ref-type="table" rid="pcbi-1002284-t005">Table 5</xref> the value of <italic>Y</italic> only weakly varies with the number of genomes (e.g. <italic>Y</italic> declined only about 10% when the number of genomes doubled from 5 to 10), but is sensitive to sequence diversity. Sequence diversity does change depending on the genus; however, we visually inspected a bacterial phylogenetic tree (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s009">Fig. S9</xref>) and found that the evolutionary divergence among the genus-level genome sets is similar to the medium and high diversity genome sets analyzed above, providing evidence that the value of <italic>Y</italic> is at least in the right ballpark. In addition, because Prodigal performs comparatively very well on <italic>E. coli</italic> <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>, the value of <italic>I</italic> is likely a lower bound on the value that would apply to the 39 genera. For many of these genera, higher initial error rates in gene start site prediction would likely yield a larger number of ortholog sets that could be corrected, as we reported previously with <italic>Burkholderia</italic> genomes <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>.</p>
      </sec>
      <sec id="s2h">
        <title>Consistency of GenBank versus Prodigal maps</title>
        <p>The impact of GMV estimated with Prodigal gene maps is likely to be conservative because Prodigal gene predictions (for orthologs) tend to be more consistent than extant Genbank data. By “Genbank data”, we mean the owner-approved or “curated” maps that are accessed by default in Genbank. As described previously <xref ref-type="bibr" rid="pcbi.1002284-Pallej1">[6]</xref>, the percentage of ortholog sets with inconsistent start sites among 5 <italic>Burkholderia</italic> genomes (representing a medium diversity set) was 53% and 35%, respectively, based on Genbank maps and Prodigal maps. Similar results were obtained in the current study with ortholog sets from the comparable 5-genome, medium diversity <italic>E. coli</italic> test set (<xref ref-type="table" rid="pcbi-1002284-t006">Table 6</xref>; Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s012">Table S3</xref>). In this test set, 2,289 ortholog sets were common to GenBank maps and Prodigal maps, enabling a direct comparison of consistency rates. Twice as many ortholog sets had inconsistent GenBank start sites (925, or 40.4%) as had inconsistent Prodigal start sites (455, or 19.9%). Prodigal made 60% (552) of the 925 GenBank ortholog sets consistent, and GMV made an additional 21% consistent. Together, Prodigal and GMV made 81% (746) of the GenBank ortholog sets consistent. This corresponds to an increase in consistency of <italic>I</italic> = 746/2289 = 0.326.</p>
        <table-wrap id="pcbi-1002284-t006" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t006</object-id><label>Table 6</label><caption>
            <title>Comparison of inconsistencies for Prodigal vs. GenBank or Glimmer3 start sites.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t006-6" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t006" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1">5 genomes Medium Diversity</td>
                <td align="left" colspan="1" rowspan="1">5 genomes High Diversity</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">Prodigal vs. GenBank</td>
                <td align="left" colspan="1" rowspan="1"># of shared ortholog sets</td>
                <td align="left" colspan="1" rowspan="1">2289</td>
                <td align="left" colspan="1" rowspan="1">1234</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which Prodigal starts were initially inconsistent</td>
                <td align="left" colspan="1" rowspan="1">455</td>
                <td align="left" colspan="1" rowspan="1">413</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which GenBank starts were initially inconsistent</td>
                <td align="left" colspan="1" rowspan="1">925</td>
                <td align="left" colspan="1" rowspan="1">311</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># made consistent by Prodigal</td>
                <td align="left" colspan="1" rowspan="1">552</td>
                <td align="left" colspan="1" rowspan="1">50</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># made consistent by GMV</td>
                <td align="left" colspan="1" rowspan="1">194</td>
                <td align="left" colspan="1" rowspan="1">47</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Prodigal vs. Glimmer3</td>
                <td align="left" colspan="1" rowspan="1"># of shared ortholog sets</td>
                <td align="left" colspan="1" rowspan="1">2427</td>
                <td align="left" colspan="1" rowspan="1">1398</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which Prodigal starts were initially inconsistent</td>
                <td align="left" colspan="1" rowspan="1">532</td>
                <td align="left" colspan="1" rowspan="1">566</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># of ortholog sets for which GenBank starts were initially inconsistent</td>
                <td align="left" colspan="1" rowspan="1">869</td>
                <td align="left" colspan="1" rowspan="1">767</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># made consistent by Prodigal</td>
                <td align="left" colspan="1" rowspan="1">432</td>
                <td align="left" colspan="1" rowspan="1">248</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1"/>
                <td align="left" colspan="1" rowspan="1"># made consistent by GMV</td>
                <td align="left" colspan="1" rowspan="1">193</td>
                <td align="left" colspan="1" rowspan="1">155</td>
              </tr>
            </tbody>
          </table></alternatives></table-wrap>
        <p>The above conservative estimate suggests applying our pipeline could significantly increase consistency of GenBank gene maps, with Prodigal accounting for ¾ and GMV accounting for ¼ of the total impact. We also obtained an alternative, less conservative estimate of the impact by modifying the GMV algorithm to preferentially use the gene calls already in GenBank as opposed to new Prodigal gene calls. In the modified algorithm, if the GenBank start sites already coincide in a multiple sequence alignment, or if a majority of these start sites do not align, nothing is done. Otherwise, if a majority of the GenBank start sites coincide, an alternative, consistent Prodigal start site is sought in the minority genomes. If one is found, then in the minority genomes the GenBank start sites are replaced with the consistent Prodigal start sites. Applying this algorithm to the same 2,289 ortholog sets that were common to GenBank maps and Prodigal maps in the 5-genome, medium diversity <italic>E. coli</italic> test set made 78% (717) of the 925 inconsistent GenBank ortholog sets consistent. This number is comparable to the 81% made consistent by first substituting all of the GenBank start sites with Prodigal start sites, and then applying GMV; however, in this alternative mode GMV was responsible for all of the changes as opposed to ¼ of the changes in the original mode. Using the method described for the Prodigal projection above, we project that running GMV in this alternative mode on 467 currently sequenced microbial genomes (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s011">Table S2</xref>) would make more than 10,000 ortholog sets consistent. Unfortunately, although GMV used in this alternative mode does improve consistency, because the GenBank gene maps for <italic>E. coli</italic> have already been modified to account for the 871 experimentally validated start sites, we cannot say whether such an application of GMV increases gene map accuracy. Therefore we have no good basis on which to recommend that GMV be used in this alternative mode, and we adhere to the more conservative projection that GMV would resolve about 4,000 inconsistencies in Prodigal gene maps, as estimated in the previous section.</p>
      </sec>
      <sec id="s2i">
        <title>Projected impact on accuracy of microbial gene maps</title>
        <p>To project the broader impact of the GMV method on the accuracy of Prodigal gene maps, we first calculated a correction rate for the medium and high diversity genome test sets with either 5 or 10 genomes and then applied this rate to 39 suitable genera. The correction rate <italic>R</italic> per genome was calculated as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.e009" xlink:type="simple"/><label>(5)</label></disp-formula>where <italic>M</italic> is the maximum number of possible ortholog sets as defined in the previous section, <italic>C</italic> is the total number of changes in the set from <xref ref-type="table" rid="pcbi-1002284-t004">Table 4</xref>, and <italic>N</italic> is the number of genomes in the set. Correction rates, <italic>R</italic>, for 5-genome test sets with medium and high diversity were 1.7% and 1.2% respectively, and were 1.2% and 1% for corresponding 10-genome test sets. The entire range was therefore 1% to 1.7%. To obtain the projected impact on accuracy, we used the same set of 39 genera (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s011">Table S2</xref>) that were used to estimate the impact on consistency. With correction rates of 1% or 1.7%, the total estimated number of corrections for 467 genomes was 13,700 and 23,300, respectively. The estimated rate of erroneous corrections varied widely. The error rates calculated from <xref ref-type="table" rid="pcbi-1002284-t004">Table 4</xref> were 8.4% and 6.8% for 5-genome test sets of medium and high diversity, respectively, and 12% and 0.8% for 10-genome test sets of medium and high diversity. With the worst-case scenario (12% error rate), we project GMV to yield more than 10,000 valid gene start corrections in 467 microbial genomes.</p>
        <p>The projection above applies to Prodigal gene maps; an assessment for existing Genbank gene maps is also desired. Accuracy can be directly measured for organisms, like <italic>E. coli</italic>, that have a gene map and a set of experimentally validated gene start sites. However, the Genbank gene map for <italic>E. coli</italic> has already been revised to include the 871 experimentally validated gene start sites, and therefore the accuracy of the map for these genes cannot be improved further. We must therefore estimate the impact based on the following logic: 1) the vast majority of Genbank maps are inferior in quality compared to the <italic>E. coli</italic> map, which has benefited from a rigorous community annotation effort <xref ref-type="bibr" rid="pcbi.1002284-Riley1">[11]</xref>; 2) Genbank gene maps have as much as twice as many inconsistencies as Prodigal gene maps; and 3) we have directly measured the impact of GMV on the accuracy of Prodigal gene maps. Using the 12% error rate from the high diversity, 10-genome test set as a worst-case scenario for erroneous corrections, more than 20,000 valid corrections are projected for Genbank gene maps for 467 microbial genomes.</p>
        <p>It is conceivable that the impact will be lower for new gene maps obtained from recent improvements in annotation pipelines. Newer maps may include information from servers such as MaGe <xref ref-type="bibr" rid="pcbi.1002284-Vallenet1">[12]</xref>, RAST <xref ref-type="bibr" rid="pcbi.1002284-Aziz1">[8]</xref>, or GenePRIMP <xref ref-type="bibr" rid="pcbi.1002284-Pati1">[9]</xref> that use comparative genomics methods for genome annotation, including leveraging information from experimentally validated gene starts. Given the evolving quality of newer gene maps, the true value of the GMV method in correcting errors in GenBank genomes will depend on accumulation of data from a broader set of users.</p>
      </sec>
      <sec id="s2j">
        <title>Significance of the GMV algorithm in light of other methods</title>
        <p>Ours is one of several approaches to leveraging multiple genomes for improving gene predictions. Numerous methods have used conservation patterns in pairwise sequence alignments to distinguish coding from non-coding regions in eukaryotic (SLAM <xref ref-type="bibr" rid="pcbi.1002284-Alexandersson1">[13]</xref>, SGP2 <xref ref-type="bibr" rid="pcbi.1002284-Parra1">[14]</xref>, TWINSCAN <xref ref-type="bibr" rid="pcbi.1002284-Flicek1">[15]</xref>, <xref ref-type="bibr" rid="pcbi.1002284-Korf1">[16]</xref>, <xref ref-type="bibr" rid="pcbi.1002284-Tenney1">[17]</xref>, Guigó et al. <xref ref-type="bibr" rid="pcbi.1002284-Guig1">[18]</xref>) or prokaryotic genomes (Walker et al. <xref ref-type="bibr" rid="pcbi.1002284-Walker1">[19]</xref>,). The use of more than two sequences to improve prediction of gene boundaries is a more recent addition.</p>
        <p>RAST <xref ref-type="bibr" rid="pcbi.1002284-Aziz1">[8]</xref> and GenePRIMP <xref ref-type="bibr" rid="pcbi.1002284-Pati1">[9]</xref> both use homologs identified via BLAST <xref ref-type="bibr" rid="pcbi.1002284-Altschul1">[20]</xref> to refine assignment of gene starts to new genomes. However, the decision rules and implementation details for revising gene start sites using these methods were not documented in detail. For example, the total number of orthologs that are used for comparison, selection of diversity among orthologs (when a spectrum of diversity is available), and the definition of consistency for these methods are unclear. These methods might make an effort to minimize 5′ length differences among orthologs, which does enforce a kind of consistency, in the spirit of GMV. However, our <italic>Burkholderia</italic> study <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref> appears to be the first example of a strict rule to enforce consistency of start sites in a multiple sequence alignment. Another important distinction between GMV and these methods is their use of “old” gene predictions as reference material (obtained when acquiring homologs from databases with archived, error-prone information) versus contemporary gene predictions. RAST and GenePRIMP both exploit archived material, in which the extent of errors is unknown. As shown above for the 5-genome, medium diversity genome set, our GMV algorithm can, in principle, exploit older gene maps; however, we can only recommend exploiting new gene predictions. As discussed above, new predictions are expected to be more accurate and, compared to gene maps already deposited in the databases, the lack of manual adjustment of these predictions enabled them to be used to rigorously assess the performance of GMV using experimentally validated genes. The automation of this pipeline provides a facile means to periodically upgrade gene maps for collections of older genomes, as well as improving gene start site predictions for new genomes.</p>
        <p>N-SCAN <xref ref-type="bibr" rid="pcbi.1002284-Gross1">[21]</xref> and CONTRAST <xref ref-type="bibr" rid="pcbi.1002284-Gross2">[22]</xref> can produce gene calls using information from more than two genomes. N-SCAN leverages a phylogenetic model to improve gene prediction. CONTRAST improved on N-SCAN by doing away with the phylogenetic model in favor of a machine learning approach. It is notable that, with the exception of CONTRAST, prior comparative genomics approaches were unable to demonstrate improved gene start site predictions beyond adding a second genome, much less more <xref ref-type="bibr" rid="pcbi.1002284-Brent1">[23]</xref>. CONTRAST demonstrated small improvements as the number of genomes was increased to five <xref ref-type="bibr" rid="pcbi.1002284-Gross2">[22]</xref>. The fact that the performance of GMV improved when the number of genomes was increased from five to ten makes it unique among comparative genomics methods.</p>
        <p>A shared feature of prior approaches is that the multiple genomes are input at the front end and are used to develop a tightly integrated gene prediction model. By contrast, the GMV algorithm is run as a post-processing step. The main disadvantage of this is the additional compute time required to refine gene calls: running GMV on a 5-genome set takes about ½ a day on a single processor machine. The compute time is limited by the BLAST step, which scales like the number of genomes squared; however, the speed of the BLAST step (and all other steps of the pipeline) can be substantially improved by parallel processing. A major advantage in implementing GMV is that it can be coupled to any gene prediction software so long as a list of alternative start sites is provided. Aside from the great flexibility it provides in applications, the modular nature of GMV allowed us to treat it as an error correction method, enabling a well-controlled means of evaluating its performance.</p>
      </sec>
      <sec id="s2k">
        <title>Conclusions</title>
        <p>The GMV algorithm dramatically decreases inconsistencies in the location of predicted gene start sites, and is projected to eliminate thousands of inconsistencies in currently sequenced microbial genomes, facilitating comparative genomics studies. At the same time, it is capable of correcting hundreds of errors in sets of 5–10 genomes and is potentially capable of correcting more than 10,000 errors in microbial gene maps. Moreover, GMV provides a straightforward solution to the challenging problem of improving gene start site predictions using more than two genomes. Overall, GMV is a simple and logical solution that resolves inconsistencies and increases the accuracy of gene maps.</p>
      </sec>
    </sec>
    <sec id="s3" sec-type="methods">
      <title>Methods</title>
      <sec id="s3a">
        <title>Genome sets</title>
        <p>Genome sets were selected with the aid of a bacterial phylogenetic tree (Benjamin McMahon, personal communication). The tree was derived by aligning the concatenated amino acid sequences of the β and β′ subunits of RNA polymerase from over 400 bacterial genomes. The 400 bacterial genomes were downloaded from NCBI (completed) and JGI (draft) in June of 2009. The amino acid sequences of the beta and beta-prime subunits of the RNA polymerase were extracted from each genome and concatenated. An initial multiple sequence alignment was calculated using MUSCLE <xref ref-type="bibr" rid="pcbi.1002284-Edgar1">[24]</xref>, followed by iterative manual curation of the alignment with BioEdit (<ext-link ext-link-type="uri" xlink:href="http://www.mbio.ncsu.edu/bioedit/bioedit.html" xlink:type="simple">http://www.mbio.ncsu.edu/bioedit/bioedit.html</ext-link>) based on the known three-dimensional structure, and tree building with a maximum likelihood method employing a minimal model of protein functional pressure (RIND <xref ref-type="bibr" rid="pcbi.1002284-Bruno1">[25]</xref> and WEIGHBOR <xref ref-type="bibr" rid="pcbi.1002284-Bruno2">[26]</xref>). The phylogenetic tree was calculated from the aligned sequences. The root of the tree was placed at the long branch connecting gram-positive and gram-negative bacteria, in accord with current understanding of bacterial evolution <xref ref-type="bibr" rid="pcbi.1002284-Skophammer1">[27]</xref>. The resulting tree (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s009">9</xref>) compares well to those in the literature <xref ref-type="bibr" rid="pcbi.1002284-Herlemann1">[28]</xref> and with 16S rRNA-based trees; it disagrees with the less-detailed NCBI taxonomy (where available) in only a handful of cases.</p>
        <p>The genome sets used for testing GMV are listed in Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s011">Table S2</xref>. Sets of 10-genomes were selected to represent low, medium, and high, and very high levels of diversity. The low diversity sets consist of randomly selected substrains of <italic>E. coli</italic>. The medium diversity sets were selected with the aid of the phylogenetic tree (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s009">Fig. S9</xref>) to approximately span a maximum evolutionary distance similar to that spanned by the <italic>Burkholderia</italic> genus, which was the subject of our previous analysis of gene start consistency <xref ref-type="bibr" rid="pcbi.1002284-Dunbar1">[7]</xref>. The genomes selected for the medium diversity sets cover a portion of the <italic>Enterobacteriaceae</italic> family. The high diversity datasets were selected to achieve approximately a twofold increase in the maximum evolutionary distance over the medium diversity datasets and cover a larger portion of the <italic>Enterobacteriaceae</italic> family. The very high diversity datasets were selected to increase the evolutionary distance by another factor of two. The very high diversity datasets include genomes from two families of Gamma Proteobacteria: <italic>Enterobacteriaceae</italic> and <italic>Pasteurellaceae</italic>. After selecting the 10-genome sets, subsets of 5 genomes were down-selected for each diversity level.</p>
        <p><xref ref-type="table" rid="pcbi-1002284-t007">Table 7</xref> summarizes the diversity in each of the 8 test sets using the median of the minimum sequence identity in each set. <xref ref-type="fig" rid="pcbi-1002284-g004">Fig. 4</xref> illustrates more detailed statistics on the sequence identity in the high diversity genome sets, which yielded the best performance for the GMV algorithm. Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s001">Figs. S<sub>1</sub>–S<sub>8</sub></xref> provide this information for all genome sets. Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s010">Table S1</xref> lists the genomes in all genome sets. (Note that the low diversity genome test sets include several strains of <italic>E. coli</italic> genomes; in this paper, we refer to <italic>E. Coli</italic> K-12 MG1655, the reference genome, as <italic>E. coli</italic>.)</p>
        <fig id="pcbi-1002284-g004" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.g004</object-id>
          <label>Figure 4</label>
          <caption>
            <title>Sequence identity statistics for the high diversity ortholog sets.</title>
            <p>A) 5-genome set; B) 10-genome test set.</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.g004" xlink:type="simple"/>
        </fig>
        <table-wrap id="pcbi-1002284-t007" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1002284.t007</object-id><label>Table 7</label><caption>
            <title>Median sequence identities among orthologs from all genome test sets.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pcbi-1002284-t007-7" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.t007" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1"># Genomes, Diversity</td>
                <td align="left" colspan="1" rowspan="1">Median Sequence Identity<xref ref-type="table-fn" rid="nt112">a</xref></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">5, Low</td>
                <td align="left" colspan="1" rowspan="1">99.3%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">5, Medium</td>
                <td align="left" colspan="1" rowspan="1">85.2%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">5, High</td>
                <td align="left" colspan="1" rowspan="1">71.6%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">5, Very High</td>
                <td align="left" colspan="1" rowspan="1">64.4%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">10, Low</td>
                <td align="left" colspan="1" rowspan="1">98.8%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">10, Medium</td>
                <td align="left" colspan="1" rowspan="1">82.5%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">10, High</td>
                <td align="left" colspan="1" rowspan="1">69.5%</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">10, Very High</td>
                <td align="left" colspan="1" rowspan="1">51.3%</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt112">
              <label>a</label>
              <p>The sequence identity used is the minimum value among all gene pairs in each ortholog set. The percentage value is normalized using sequence length information (<xref ref-type="sec" rid="s3">Methods</xref>).</p>
            </fn>
          </table-wrap-foot></table-wrap>
      </sec>
      <sec id="s3b">
        <title>GMV algorithm</title>
        <p>The GMV algorithm was implemented in an automated pipeline to predict consistent start sites, illustrated in <xref ref-type="fig" rid="pcbi-1002284-g001">Fig. 1</xref>. The software is distributed freely under a New BSD license and is available at <ext-link ext-link-type="uri" xlink:href="http://code.google.com/p/gmv/" xlink:type="simple">http://code.google.com/p/gmv/</ext-link>. The input is a set of similar genomes whose start sites are to be predicted. These genomes are provided in FASTA format to the GMV algorithm. The different steps in the pipeline involve four different software programs, each of which is automatically activated:</p>
        <sec id="s3b1">
          <title>Step A</title>
          <p>In this step, gene predictions are made for each genome in the set using Prodigal (we used version 1.10 here, which is no longer available; the version we distribute uses versions 2.00–2.50) <xref ref-type="bibr" rid="pcbi.1002284-Hyatt1">[3]</xref>. Like most gene finding programs, Prodigal selects a single best start site but also evaluates other potential start sites for each gene, computing a quality score for each start site that it proposes. The GMV algorithm uses the alternative start sites in the subsequent steps of the pipeline.</p>
        </sec>
        <sec id="s3b2">
          <title>Step B</title>
          <p>In this step, alternative start sites for each gene in each genome are obtained from the Prodigal output files.</p>
        </sec>
        <sec id="s3b3">
          <title>Step C</title>
          <p>In this step, gene predictions from Step B are used to derive ortholog sets by a pan-reciprocal best hit approach using BLASTP (version 2.2.20) with default settings <xref ref-type="bibr" rid="pcbi.1002284-Altschul1">[20]</xref>. First, BLASTP is used to obtain sequence identity for all pairs of proteins corresponding to the genes predicted by Prodigal. The sequence identity score computed by BLASTP is normalized by multiplying it by the number of aligned bases and dividing it by the number of bases in the longer of the two compared sequences. Sets of orthologous genes that include a single pan-reciprocal best BLASTP match for each genome are identified; matches are ranked by the normalized sequence identity computed above. Each ortholog set contains exactly one representative from each genome in the set.</p>
        </sec>
        <sec id="s3b4">
          <title>Step D</title>
          <p>Multiple sequence alignment is performed for each ortholog set using MUSCLE (version 3.7) with default settings <xref ref-type="bibr" rid="pcbi.1002284-Edgar1">[24]</xref>. The sequence of each gene in the alignment includes the 250 bp DNA sequence upstream of the earliest of the possible starts. Nucleotide sequences are used to construct multiple sequence alignments.</p>
        </sec>
        <sec id="s3b5">
          <title>Step E</title>
          <p>This is the final step in the GMV pipeline and it involves prediction of consistent start sites. If the positions of all of the original start sites coincide in the multiple sequence alignment, the predictions are accepted as is. Otherwise, look for a position where the original start sites coincide for a majority of genomes, and where an alternative start site coincides in each of the remaining genomes. Use the alternative sites as modified predictions for the remaining genomes. If there is no consistent start site that obeys the majority rule, flag the prediction as inconsistent.</p>
          <p>It is important to note that the GMV pipeline is not restricted to using Prodigal for gene prediction and MUSCLE for multiple sequence alignment. GMV can be made to work with any gene prediction software that can output alternative start sites in Step A. The current requirements for input to GMV is described in the manual included in the package, distributed at <ext-link ext-link-type="uri" xlink:href="http://code.google.com/p/gmv/" xlink:type="simple">http://code.google.com/p/gmv/</ext-link>. Similarly, it is possible to use any multiple sequence alignment software that can handle nucleotide data in Step D of the pipeline. The entire pipeline can be run automatically, without any manual intervention. It is also possible to run each step of the pipeline separately, if necessary.</p>
        </sec>
      </sec>
      <sec id="s3c">
        <title>Software</title>
        <p>The GMV algorithm pipeline was developed using Java (JDK 1.6) and Perl 5.8. The software has been tested on both Linux and MacOS X operating systems. Software is available under the New BSD open source license and is freely available at <ext-link ext-link-type="uri" xlink:href="http://code.google.com/p/gmv" xlink:type="simple">http://code.google.com/p/gmv</ext-link>.</p>
      </sec>
    </sec>
    <sec id="s4">
      <title>Supporting Information</title>
      <supplementary-material id="pcbi.1002284.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s001" xlink:type="simple">
        <label>Figure S1</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the low diversity, 5 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s002" xlink:type="simple">
        <label>Figure S2</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the medium diversity, 5 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s003" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s003" xlink:type="simple">
        <label>Figure S3</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the high diversity, 5 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s004" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s004" xlink:type="simple">
        <label>Figure S4</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the very high diversity, 5 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s005" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s005" xlink:type="simple">
        <label>Figure S5</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the low diversity, 10 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s006" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s006" xlink:type="simple">
        <label>Figure S6</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the medium diversity, 10 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s007" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s007" xlink:type="simple">
        <label>Figure S7</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the high diversity, 10 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s008" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s008" xlink:type="simple">
        <label>Figure S8</label>
        <caption>
          <p>Histogram of mean (top) and minimum (bottom) identity score between genes in ortholog sets derived from the very high diversity, 10 genome set.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s009" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s009" xlink:type="simple">
        <label>Figure S9</label>
        <caption>
          <p>Bacterial phylogenetic tree. The tree is based on aligning the beta and beta-prime subunits of the RNA polymerase and was generated using a maximum likelihood method <xref ref-type="bibr" rid="pcbi.1002284-Bruno1">[25]</xref>, <xref ref-type="bibr" rid="pcbi.1002284-Bruno2">[26]</xref>. The root of the tree is at the left, on the long branch connecting gram-positive and gram-negative bacteria. The lengths of horizontal lines correspond to a measure of evolutionary distance.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s010" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s010" xlink:type="simple">
        <label>Table S1</label>
        <caption>
          <p>List of genomes in each genome set. The FASTA files were downloaded June–July 2010.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s011" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s011" xlink:type="simple">
        <label>Table S2</label>
        <caption>
          <p>List of genomes used to estimate projected impact of GMV on consistency and error rates in gene predictions. The 467 genomes were organized into 39 genera for the estimate. The genome list was obtained from the Integrated Microbial Genomes resource at the DOE Joint Genome Institute (<ext-link ext-link-type="uri" xlink:href="http://img.jgi.doe.gov/cgi-bin/pub/main.cgi" xlink:type="simple">http://img.jgi.doe.gov/cgi-bin/pub/main.cgi</ext-link>) in September 2010.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pcbi.1002284.s012" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1002284.s012" xlink:type="simple">
        <label>Table S3</label>
        <caption>
          <p>Source files for GenBank default and Glimmer3 gene start sites for 5 genome sets of medium and high diversity.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>We are grateful to Benjamin McMahon for providing the bacterial phylogenetic tree (Supplementary <xref ref-type="supplementary-material" rid="pcbi.1002284.s009">Fig. S9</xref>).</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pcbi.1002284-Delcher1">
        <label>1</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Delcher</surname><given-names>AL</given-names></name><name name-style="western"><surname>Bratke</surname><given-names>KA</given-names></name><name name-style="western"><surname>Powers</surname><given-names>EC</given-names></name><name name-style="western"><surname>Salzberg</surname><given-names>SL</given-names></name></person-group>             <year>2007</year>             <article-title>Identifying bacterial genes and endosymbiont DNA with Glimmer.</article-title>             <source>Bioinformatics</source>             <volume>23</volume>             <fpage>673</fpage>             <lpage>679</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Besemer1">
        <label>2</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Besemer</surname><given-names>J</given-names></name><name name-style="western"><surname>Borodovsky</surname><given-names>M</given-names></name></person-group>             <year>2005</year>             <article-title>GeneMark: web software for gene finding in prokaryotes, eukaryotes and viruses.</article-title>             <source>Nucleic Acids Res</source>             <volume>33</volume>             <fpage>W451</fpage>             <lpage>454</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Hyatt1">
        <label>3</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Hyatt</surname><given-names>D</given-names></name><name name-style="western"><surname>Chen</surname><given-names>GL</given-names></name><name name-style="western"><surname>Locascio</surname><given-names>PF</given-names></name><name name-style="western"><surname>Land</surname><given-names>ML</given-names></name><name name-style="western"><surname>Larimer</surname><given-names>FW</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>Prodigal: prokaryotic gene recognition and translation initiation site identification.</article-title>             <source>BMC Bioinformatics</source>             <volume>11</volume>             <fpage>119</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Dai1">
        <label>4</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>M</given-names></name><name name-style="western"><surname>Wang</surname><given-names>P</given-names></name><name name-style="western"><surname>Boyd</surname><given-names>AD</given-names></name><name name-style="western"><surname>Kostov</surname><given-names>G</given-names></name><name name-style="western"><surname>Athey</surname><given-names>B</given-names></name><etal/></person-group>             <year>2005</year>             <article-title>Evolving gene/transcript definitions significantly alter the interpretation of GeneChip data.</article-title>             <source>Nucleic Acids Res</source>             <volume>33</volume>             <fpage>e175</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Poptsova1">
        <label>5</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Poptsova</surname><given-names>MS</given-names></name><name name-style="western"><surname>Gogarten</surname><given-names>JP</given-names></name></person-group>             <year>2010</year>             <article-title>Using comparative genome analysis to identify problems in annotated microbial genomes.</article-title>             <source>Microbiology</source>             <volume>156</volume>             <fpage>1909</fpage>             <lpage>1917</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Pallej1">
        <label>6</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Pallejà</surname><given-names>A</given-names></name><name name-style="western"><surname>Harrington</surname><given-names>ED</given-names></name><name name-style="western"><surname>Bork</surname><given-names>P</given-names></name></person-group>             <year>2008</year>             <article-title>Large gene overlaps in prokaryotic genomes: result of functional constraints or mispredictions?</article-title>             <source>BMC Genomics</source>             <volume>9</volume>             <fpage>335</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Dunbar1">
        <label>7</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Dunbar</surname><given-names>J</given-names></name><name name-style="western"><surname>Cohn</surname><given-names>JD</given-names></name><name name-style="western"><surname>Wall</surname><given-names>ME</given-names></name></person-group>             <year>2011</year>             <article-title>Consistency of gene starts among <italic>Burkholderia</italic> genomes.</article-title>             <source>BMC Bioinformatics</source>             <volume>12</volume>             <fpage>125</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Aziz1">
        <label>8</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Aziz</surname><given-names>RK</given-names></name><name name-style="western"><surname>Bartels</surname><given-names>D</given-names></name><name name-style="western"><surname>Best</surname><given-names>AA</given-names></name><name name-style="western"><surname>DeJongh</surname><given-names>M</given-names></name><name name-style="western"><surname>Disz</surname><given-names>T</given-names></name><etal/></person-group>             <year>2008</year>             <article-title>The RAST Server: rapid annotations using subsystems technology.</article-title>             <source>BMC Genomics</source>             <volume>9</volume>             <fpage>75</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Pati1">
        <label>9</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Pati</surname><given-names>A</given-names></name><name name-style="western"><surname>Ivanova</surname><given-names>NN</given-names></name><name name-style="western"><surname>Mikhailova</surname><given-names>N</given-names></name><name name-style="western"><surname>Ovchinnikova</surname><given-names>G</given-names></name><name name-style="western"><surname>Hooper</surname><given-names>SD</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>GenePRIMP: a gene prediction improvement pipeline for prokaryotic genomes.</article-title>             <source>Nat Methods</source>             <volume>7</volume>             <fpage>455</fpage>             <lpage>457</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Rudd1">
        <label>10</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Rudd</surname><given-names>KE</given-names></name></person-group>             <year>2000</year>             <article-title>EcoGene: a genome sequence database for <italic>Escherichia coli</italic> K-12.</article-title>             <source>Nucleic Acids Res</source>             <volume>28</volume>             <fpage>60</fpage>             <lpage>64</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Riley1">
        <label>11</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Riley</surname><given-names>M</given-names></name><name name-style="western"><surname>Abe</surname><given-names>T</given-names></name><name name-style="western"><surname>Arnaud</surname><given-names>MB</given-names></name><name name-style="western"><surname>Berlyn</surname><given-names>MK</given-names></name><name name-style="western"><surname>Blattner</surname><given-names>FR</given-names></name><etal/></person-group>             <year>2006</year>             <article-title><italic>Escherichia coli</italic> K-12: a cooperatively developed annotation snapshot–2005.</article-title>             <source>Nucleic Acids Res</source>             <volume>34</volume>             <fpage>1</fpage>             <lpage>9</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Vallenet1">
        <label>12</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Vallenet</surname><given-names>D</given-names></name><name name-style="western"><surname>Labarre</surname><given-names>L</given-names></name><name name-style="western"><surname>Rouy</surname><given-names>Z</given-names></name><name name-style="western"><surname>Barbe</surname><given-names>V</given-names></name><name name-style="western"><surname>Bocs</surname><given-names>S</given-names></name><etal/></person-group>             <year>2006</year>             <article-title>MaGe: a microbial genome annotation system supported by synteny results.</article-title>             <source>Nucleic Acids Res</source>             <volume>34</volume>             <fpage>53</fpage>             <lpage>65</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Alexandersson1">
        <label>13</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Alexandersson</surname><given-names>M</given-names></name><name name-style="western"><surname>Cawley</surname><given-names>S</given-names></name><name name-style="western"><surname>Pachter</surname><given-names>L</given-names></name></person-group>             <year>2003</year>             <article-title>SLAM: cross-species gene finding and alignment with a generalized pair hidden Markov model.</article-title>             <source>Genome Res</source>             <volume>13</volume>             <fpage>496</fpage>             <lpage>502</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Parra1">
        <label>14</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Parra</surname><given-names>G</given-names></name><name name-style="western"><surname>Agarwal</surname><given-names>P</given-names></name><name name-style="western"><surname>Abril</surname><given-names>JF</given-names></name><name name-style="western"><surname>Wiehe</surname><given-names>T</given-names></name><name name-style="western"><surname>Fickett</surname><given-names>JW</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>Comparative gene prediction in human and mouse.</article-title>             <source>Genome Res</source>             <volume>13</volume>             <fpage>108</fpage>             <lpage>117</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Flicek1">
        <label>15</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Flicek</surname><given-names>P</given-names></name><name name-style="western"><surname>Keibler</surname><given-names>E</given-names></name><name name-style="western"><surname>Hu</surname><given-names>P</given-names></name><name name-style="western"><surname>Korf</surname><given-names>I</given-names></name><name name-style="western"><surname>Brent</surname><given-names>MR</given-names></name></person-group>             <year>2003</year>             <article-title>Leveraging the mouse genome for gene prediction in human: from whole-genome shotgun reads to a global synteny map.</article-title>             <source>Genome Res</source>             <volume>13</volume>             <fpage>46</fpage>             <lpage>54</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Korf1">
        <label>16</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Korf</surname><given-names>I</given-names></name><name name-style="western"><surname>Flicek</surname><given-names>P</given-names></name><name name-style="western"><surname>Duan</surname><given-names>D</given-names></name><name name-style="western"><surname>Brent</surname><given-names>MR</given-names></name></person-group>             <year>2001</year>             <article-title>Integrating genomic homology into gene structure prediction.</article-title>             <source>Bioinformatics</source>             <volume>17</volume>             <supplement>Suppl 1</supplement>             <fpage>S140</fpage>             <lpage>148</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Tenney1">
        <label>17</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Tenney</surname><given-names>AE</given-names></name><name name-style="western"><surname>Brown</surname><given-names>RH</given-names></name><name name-style="western"><surname>Vaske</surname><given-names>C</given-names></name><name name-style="western"><surname>Lodge</surname><given-names>JK</given-names></name><name name-style="western"><surname>Doering</surname><given-names>TL</given-names></name><etal/></person-group>             <year>2004</year>             <article-title>Gene prediction and verification in a compact genome with numerous small introns.</article-title>             <source>Genome Res</source>             <volume>14</volume>             <fpage>2330</fpage>             <lpage>2335</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Guig1">
        <label>18</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Guigó</surname><given-names>R</given-names></name><name name-style="western"><surname>Dermitzakis</surname><given-names>ET</given-names></name><name name-style="western"><surname>Agarwal</surname><given-names>P</given-names></name><name name-style="western"><surname>Ponting</surname><given-names>CP</given-names></name><name name-style="western"><surname>Parra</surname><given-names>G</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>Comparison of mouse and human genomes followed by experimental verification yields an estimated 1,019 additional genes.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>100</volume>             <fpage>1140</fpage>             <lpage>1145</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Walker1">
        <label>19</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Walker</surname><given-names>M</given-names></name><name name-style="western"><surname>Pavlovic</surname><given-names>V</given-names></name><name name-style="western"><surname>Kasif</surname><given-names>S</given-names></name></person-group>             <year>2002</year>             <article-title>A comparative genomic method for computational identification of prokaryotic translation initiation sites.</article-title>             <source>Nucleic Acids Res</source>             <volume>30</volume>             <fpage>3181</fpage>             <lpage>3191</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Altschul1">
        <label>20</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Altschul</surname><given-names>SF</given-names></name><name name-style="western"><surname>Gish</surname><given-names>W</given-names></name><name name-style="western"><surname>Miller</surname><given-names>W</given-names></name><name name-style="western"><surname>Myers</surname><given-names>EW</given-names></name><name name-style="western"><surname>Lipman</surname><given-names>DJ</given-names></name></person-group>             <year>1990</year>             <article-title>Basic local alignment search tool.</article-title>             <source>J Mol Biol</source>             <volume>215</volume>             <fpage>403</fpage>             <lpage>410</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Gross1">
        <label>21</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gross</surname><given-names>SS</given-names></name><name name-style="western"><surname>Brent</surname><given-names>MR</given-names></name></person-group>             <year>2006</year>             <article-title>Using multiple alignments to improve gene prediction.</article-title>             <source>J Comput Biol</source>             <volume>13</volume>             <fpage>379</fpage>             <lpage>393</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Gross2">
        <label>22</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gross</surname><given-names>SS</given-names></name><name name-style="western"><surname>Do</surname><given-names>CB</given-names></name><name name-style="western"><surname>Sirota</surname><given-names>M</given-names></name><name name-style="western"><surname>Batzoglou</surname><given-names>S</given-names></name></person-group>             <year>2007</year>             <article-title>CONTRAST: a discriminative, phylogeny-free approach to multiple informant de novo gene prediction.</article-title>             <source>Genome Biol</source>             <volume>8</volume>             <fpage>R269</fpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Brent1">
        <label>23</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Brent</surname><given-names>MR</given-names></name></person-group>             <year>2008</year>             <article-title>Steady progress and recent breakthroughs in the accuracy of automated genome annotation.</article-title>             <source>Nat Rev Genet</source>             <volume>9</volume>             <fpage>62</fpage>             <lpage>73</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Edgar1">
        <label>24</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Edgar</surname><given-names>RC</given-names></name></person-group>             <year>2004</year>             <article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput.</article-title>             <source>Nucleic Acids Res</source>             <volume>32</volume>             <fpage>1792</fpage>             <lpage>1797</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Bruno1">
        <label>25</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Bruno</surname><given-names>WJ</given-names></name></person-group>             <year>1996</year>             <article-title>Modeling residue usage in aligned protein sequences via maximum likelihood.</article-title>             <source>Mol Biol Evol</source>             <volume>13</volume>             <fpage>1368</fpage>             <lpage>1374</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Bruno2">
        <label>26</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Bruno</surname><given-names>WJ</given-names></name><name name-style="western"><surname>Socci</surname><given-names>ND</given-names></name><name name-style="western"><surname>Halpern</surname><given-names>AL</given-names></name></person-group>             <year>2000</year>             <article-title>Weighted neighbor joining: a likelihood-based approach to distance-based phylogeny reconstruction.</article-title>             <source>Mol Biol Evol</source>             <volume>17</volume>             <fpage>189</fpage>             <lpage>197</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Skophammer1">
        <label>27</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Skophammer</surname><given-names>RG</given-names></name><name name-style="western"><surname>Servin</surname><given-names>JA</given-names></name><name name-style="western"><surname>Herbold</surname><given-names>CW</given-names></name><name name-style="western"><surname>Lake</surname><given-names>JA</given-names></name></person-group>             <year>2007</year>             <article-title>Evidence for a gram-positive, eubacterial root of the tree of life.</article-title>             <source>Mol Biol Evol</source>             <volume>24</volume>             <fpage>1761</fpage>             <lpage>1768</lpage>          </element-citation>
      </ref>
      <ref id="pcbi.1002284-Herlemann1">
        <label>28</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Herlemann</surname><given-names>DP</given-names></name><name name-style="western"><surname>Geissinger</surname><given-names>O</given-names></name><name name-style="western"><surname>Ikeda-Ohtsubo</surname><given-names>W</given-names></name><name name-style="western"><surname>Kunin</surname><given-names>V</given-names></name><name name-style="western"><surname>Sun</surname><given-names>H</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Genomic analysis of “<italic>Elusimicrobium minutum</italic>,” the first cultivated representative of the phylum “<italic>Elusimicrobia</italic>” (formerly termite group 1).</article-title>             <source>Appl Environ Microbiol</source>             <volume>75</volume>             <fpage>2841</fpage>             <lpage>2849</lpage>          </element-citation>
      </ref>
    </ref-list>
    
  </back>
</article>