<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
  <front>
    <journal-meta><journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="pmc">plosone</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS ONE</journal-title></journal-title-group><issn pub-type="epub">1932-6203</issn><publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher></journal-meta>
    <article-meta><article-id pub-id-type="publisher-id">PONE-D-10-02652</article-id><article-id pub-id-type="doi">10.1371/journal.pone.0017293</article-id><article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Computational biology</subject>
            <subj-group>
              <subject>Genomics</subject>
              <subj-group>
                <subject>Comparative genomics</subject>
                <subject>Genome analysis tools</subject>
                <subject>Genome sequencing</subject>
              </subj-group>
            </subj-group>
            <subj-group>
              <subject>Evolutionary modeling</subject>
              <subject>Molecular genetics</subject>
              <subject>Sequence analysis</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Evolutionary biology</subject>
            <subj-group>
              <subject>Evolutionary systematics</subject>
              <subj-group>
                <subject>Phylogenetics</subject>
              </subj-group>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Genetics</subject>
            <subj-group>
              <subject>Molecular genetics</subject>
              <subj-group>
                <subject>Gene identification and analysis</subject>
              </subj-group>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Microbiology</subject>
            <subj-group>
              <subject>Virology</subject>
              <subj-group>
                <subject>Viral classification</subject>
              </subj-group>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Genetics and Genomics</subject>
          <subject>Virology</subject>
          <subject>Computational Biology</subject>
          <subject>Evolutionary Biology</subject>
        </subj-group>
      </article-categories><title-group><article-title>A Novel Method of Characterizing Genetic Sequences: Genome Space with Biological Distance and Applications</article-title><alt-title alt-title-type="running-head">A Novel Method of Characterizing Genetic Sequences</alt-title></title-group><contrib-group>
        <contrib contrib-type="author" equal-contrib="yes" xlink:type="simple">
          <name name-style="western">
            <surname>Deng</surname>
            <given-names>Mo</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" equal-contrib="yes" xlink:type="simple">
          <name name-style="western">
            <surname>Yu</surname>
            <given-names>Chenglong</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Liang</surname>
            <given-names>Qian</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>He</surname>
            <given-names>Rong L.</given-names>
          </name>
          <xref ref-type="aff" rid="aff3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Yau</surname>
            <given-names>Stephen S.-T.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
      </contrib-group><aff id="aff1"><label>1</label><addr-line>Department of Mathematics, Statistics and Computer Science, University of Illinois at Chicago, Chicago, Illinois, United States of America</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>The Institute of Mathematical Sciences, The Chinese University of Hong Kong, Shatin, Hong Kong, People's Republic of China</addr-line>       </aff><aff id="aff3"><label>3</label><addr-line>Department of Biological Sciences, Chicago State University, Chicago, Illinois, United States of America</addr-line>       </aff><contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Gadagkar</surname>
            <given-names>Sudhindra</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group><aff id="edit1">Midwestern University, United States of America</aff><author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">yau@uic.edu</email></corresp>
        <fn fn-type="con">
          <p>Performed the experiments: MD CY RLH. Analyzed the data: MD CY RLH. Wrote the paper: MD CY. Conceived and designed method: SSTY. Normalizations of method: MD. Programming: MD QL. Helped write the paper: RLH SSTY.</p>
        </fn>
      <fn fn-type="conflict">
        <p>The authors have declared that no competing interests exist.</p>
      </fn></author-notes><pub-date pub-type="collection">
        <year>2011</year>
      </pub-date><pub-date pub-type="epub">
        <day>2</day>
        <month>3</month>
        <year>2011</year>
      </pub-date><volume>6</volume><issue>3</issue><elocation-id>e17293</elocation-id><history>
        <date date-type="received">
          <day>27</day>
          <month>9</month>
          <year>2010</year>
        </date>
        <date date-type="accepted">
          <day>28</day>
          <month>1</month>
          <year>2011</year>
        </date>
      </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2011</copyright-year><copyright-holder>Deng et al</copyright-holder><license><license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions><abstract>
        <sec>
          <title>Background</title>
          <p>Most existing methods for phylogenetic analysis involve developing an evolutionary model and then using some type of computational algorithm to perform multiple sequence alignment. There are two problems with this approach: (1) different evolutionary models can lead to different results, and (2) the computation time required for multiple alignments makes it impossible to analyse the phylogeny of a whole genome. This motivates us to create a new approach to characterize genetic sequences.</p>
        </sec>
        <sec>
          <title>Methodology</title>
          <p>To each DNA sequence, we associate a natural vector based on the distributions of nucleotides. This produces a one-to-one correspondence between the DNA sequence and its natural vector. We define the distance between two DNA sequences to be the distance between their associated natural vectors. This creates a genome space with a biological distance which makes global comparison of genomes with same topology possible. We use our proposed method to analyze the genomes of the new influenza A (H1N1) virus, human rhinoviruses (HRV) and mammalian mitochondrial. The result shows that a triple-reassortant swine virus circulating in North America and the Eurasian swine virus belong to the lineage of the influenza A (H1N1) virus. For the HRV and mammalian mitochondrial genomes, the results coincide with biologists' analyses.</p>
        </sec>
        <sec>
          <title>Conclusions</title>
          <p>Our approach provides a powerful new tool for analyzing and annotating genomes and their phylogenetic relationships. Whole or partial genomes can be handled more easily and more quickly than using multiple alignment methods. Once a genome space has been constructed, it can be stored in a database. There is no need to reconstruct the genome space for subsequent applications, whereas in multiple alignment methods, realignment is needed to add new sequences. Furthermore, one can make a global comparison of all genomes simultaneously, which no other existing method can achieve.</p>
        </sec>
      </abstract><funding-group><funding-statement>The authors have no support or funding to report.</funding-statement></funding-group><counts>
        <page-count count="9"/>
      </counts></article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>Computational and statistical methods to cluster the DNA or protein sequences have been successfully applied in clustering DNA, protein sequences and microarray data <xref ref-type="bibr" rid="pone.0017293-Amano1">[1]</xref>–<xref ref-type="bibr" rid="pone.0017293-Nakashima1">[8]</xref>. Yau and his group showed that the genomic space method was an efficient way to cluster the DNA or protein sequences <xref ref-type="bibr" rid="pone.0017293-Yau1">[9]</xref>–<xref ref-type="bibr" rid="pone.0017293-Yu1">[13]</xref>. In <xref ref-type="bibr" rid="pone.0017293-Yau1">[9]</xref>, <xref ref-type="bibr" rid="pone.0017293-Yau2">[11]</xref> each nucleic base or amino acid was assigned a specific value. For example, nucleic base adenine A was assigned to the pair <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e001" xlink:type="simple"/></inline-formula> <xref ref-type="bibr" rid="pone.0017293-Yau1">[9]</xref>. This method can be used successfully to represent a DNA sequence in the Cartesian coordinate plane, however the nucleotides are artificially assigned to specific values which are not inherently related to DNA or protein sequences. In contrast, the parameters used in this work are natural because they are based on the numbers and distributions of nucleotides in the sequence.</p>
      <p>In this paper we propose a method of characterizing DNA sequences, which uses a specific mathematical description of distributions of nucleotides in a DNA sequence that represents the biological information in the sequence. To each DNA sequence we associate a natural sequence of parameters, called a natural vector, describing the numbers and distributions of nucleotides in the sequence. We show that the correspondence between a natural vector and a DNA sequence is one-to-one. A natural distance between two genes is the distance between their corresponding natural vectors. This creates a genome space with biological distance, which allows us to do phylogenetic analysis in the most natural and easy manner. This alignment-free method is much faster than conventional multiple sequence alignment methods. Multiple sequence alignment (MSA) can be seen as a generalization of pair-wise sequence alignment, in which, instead of aligning two sequences, k sequences are aligned simultaneously. MSA is the most powerful method to analyze the genetic sequences and most of state-of-art algorithms are constructed based on it <xref ref-type="bibr" rid="pone.0017293-Larkin1">[14]</xref>–<xref ref-type="bibr" rid="pone.0017293-Katoh1">[16]</xref>. It is however, an NP-hard computational optimization problem which is implausible for a huge amount of sequences <xref ref-type="bibr" rid="pone.0017293-Wang1">[17]</xref>.</p>
      <p>For our first application, we analysed the new influenza A (H1N1) virus based on the whole genome (<xref ref-type="fig" rid="pone-0017293-g001">figure 1</xref>). The previous research of A (H1N1) only focuses on individual segmented genes <xref ref-type="bibr" rid="pone.0017293-Garten1">[18]</xref>. We analyze the whole genome and individual genes as well. The influenza A virus (H1N1) genome contains 8 genes: polymerase PB2, PB1, PA, hemagglutinin HA, neuraminidase NA, nucleocapsid NP, matrix protein MP and nonstructural gene NS. The results of our whole genome, natural vector method analysis show that the lineage of the influenza A (H1N1) virus includes a triple-reassortant swine virus circulating in North America and the Eurasian swine virus, but does not include either human seasonal influenza and avian viruses. In addition, our results of analysis on each individual gene coincide with Garten et al <xref ref-type="bibr" rid="pone.0017293-Garten1">[18]</xref>. We also analyzed the whole genomes of human rhinoviruses (HRV), which cause serious upper and lower respiratory tract disease worldwide. The result, shown in the phylogenetic tree constructed from known HRV whole genomes, demonstrates that the five clusters HRV-A, HRV-B, HRV-C, HEV-B and HEV-C are clearly separated from each other (<xref ref-type="fig" rid="pone-0017293-g002">figure 2</xref>). This result coincides with Palmenberg et al. 's result <xref ref-type="bibr" rid="pone.0017293-Palmenberg1">[19]</xref>. As another biological application, a dataset of 31 mammalian mitochondrial genomes was analyzed by our method (<xref ref-type="fig" rid="pone-0017293-g003">figure 3</xref>). Our approach also considers circular genomes. The result shows that these 31 genomes are well clustered. The natural vector method gives us the natural distance between two genes or genomes while the distance obtained from other methods depends on the choice of evolutionary models. As a result, there are big variations among the phylogenetic trees obtained from various models. Here we performed the maximum likelihood (ML) method and neighbor-joining (NJ) method on a dataset of 21 flu virus genomes. We found that the swine flu viruses are not clustered correctly by using the ML method with the J-C model (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(a)</xref>) and the origin of A H1N1 virus is not clear by NJ method with Kimura model (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(b)</xref>) since A H1N1 genomes are all very far away from other genomes. The result obtained by the NJ method with the Jukes-Cantor model (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(c)</xref>) fails to cluster swine flu virus correctly and the result is totally different from that by the Kimura model (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(b)</xref>).</p>
      <fig id="pone-0017293-g001" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0017293.g001</object-id>
        <label>Figure 1</label>
        <caption>
          <title>Genome analysis.</title>
          <p>We apply our method to analyze 59 influenza viruses based on their whole genomes. The natural vector and the hierarchical clustering methods are used to reconstruct the phylogenetic tree for nucleotide sequences of the whole genome sequences of selected influenza viruses. The selected viruses are chosen to be representative from among all available relevant sequences in GenBank. Sequences have both high and low divergence to avoid biasing the distribution of branch lengths. Strains are representative of the major gene lineages from different hosts. The robustness of individual nodes of the tree is assessed using a bootstrap resampling analysis with 1000 replicates shown in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>. From this figure, we can clearly see that new influenza A (H1N1) viruses originate from North American triple-reassortant swine virus and Eurasian classical swine virus lineage. We note that (A/swine/Nakhon pathom/NIAH586-1/2005(H3N2)), (A/duck/Nanchang/4-165/2000(H4N6)) and American avian (A/blue-winged teal/Ohio/1864/2006(H3N8)) are not clustered with A (H1N1) genomes from the same geographical regions respectively. This result is caused by the different structures of these genomes and the traditional A (H1N1) subtypes. In addition, we check the distance matrix of these genomes obtained by natural vectors and the result shows that (A/duck/Nanchang/4-165/2000(H4N6)) is the closest to A/duck/NY/185502/2002(H5N2). Meanwhile, A/blue-winged teal/Ohio/1864/2006(H3N8) is the closest to A/chicken/Korea/ES/03(H5N1) and A/egret/Hong Kong/757.2/2003(H5N1) respectively, which means that A/blue-winged teal/Ohio/1864/2006(H3N8) is evolutionary related with H5N1 avian virus outbreak in Asian countries from 2003 to 2006. As for A/swine/Nakhon pathom/NIAH586-1/2005(H3N2), it is the closest to A/swine/Tianjin/01/2004(H1N1) and then to A/swine/Ontario/55383/04(H1N2) with and A/swine/OH/511445/2007(H1N1). This H3N2 is the closest related to Eurasian swine even if it is clustered within American swine clade (The large distance matrix data is not shown and available upon request).</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.g001" xlink:type="simple"/>
      </fig>
      <fig id="pone-0017293-g002" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0017293.g002</object-id>
        <label>Figure 2</label>
        <caption>
          <title>The Natural vector method is used for clustering the HRV genome virus at the whole genome level.</title>
          <p>All HRV data are provided in <xref ref-type="bibr" rid="pone.0017293-Palmenberg1">[19]</xref> and the corresponding details are described in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>. This figure shows relationships between all known HRV serotypes created on the basis of full genome sequences. The HEV-B, C sequences are used as outgroups. The five clusters listed around the circular tree, HRV-C, HRV-B, HRV-A, HEV-B and HEV-C are separated clearly (HEV-B, C are outgroups) by using MEGA software <xref ref-type="bibr" rid="pone.0017293-Kumar1">[27]</xref>. This clustering result is the same as Palmenberg et al's result shown in figure S6a in their paper <xref ref-type="bibr" rid="pone.0017293-Palmenberg1">[19]</xref>. This method only needs 18 seconds to obtain this clustering result while it takes more than 19 hours for the multiple alignment method on the same dataset.</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.g002" xlink:type="simple"/>
      </fig>
      <fig id="pone-0017293-g003" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0017293.g003</object-id>
        <label>Figure 3</label>
        <caption>
          <title>Genome analysis on 31 mammalian mitochondrial genomes.</title>
          <p>We applied our method to analyze 31 mammalian mitochondrial genomes. From our clustering analysis, we can see that all 31 genomes are correctly clustered into 7 known clusters: Erinaceomorpha (cluster 1), Primates (cluster 2), Carnivore (cluster 3), Perissodactyla (cluster 4), Cetacea and Artiodactyla (cluster 5), Lagomorpha (cluster 6), Rodentia (cluster 7). Data are provided in <xref ref-type="table" rid="pone-0017293-t001">Table 1</xref>. For the primates and carnivores subgroups, the clades are a little different from those obtained by using mitochondrial DNA coding sequences. In this experiment, we use the whole genome sequences containing all tRNA, sRNA, polypeptide-encoding genes and D-loop rather than mtDNA coding sequences, which may lead slightly different results. In fact, the distance matrix obtained by natural vectors shows that human is the closest to c.chimpanzee and p.chimpanzee with the distance of 994123.7 and 1346597.8 respectively, while giant panda is the closest to black bear with the distance of 2468063, although they are not clustered together.</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.g003" xlink:type="simple"/>
      </fig>
      <fig id="pone-0017293-g004" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0017293.g004</object-id>
        <label>Figure 4</label>
        <caption>
          <title>Phylogenetic trees are reconstructed by using the maximum likelihood (ML) alignment method with Jukes-Cantor model, the neighbor-joining (NJ) method with the Kimura 2 parameter model and with the Jukes-Cantor model.</title>
          <p>It is clear that swine flu viruses are not clustered correctly using the ML method (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(a)</xref>). The NJ method with the Kimura and Jukes-Cantor models yields totally different phylogenetic trees. The Kimura model fails to distinguish the origin of A H1N1 virus since A H1N1 genomes are all very far away from other genomes (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(b)</xref>), while the Jukes-Cantor model fails to cluster swine flu viruses correctly(<xref ref-type="fig" rid="pone-0017293-g004">figure 4(c)</xref>). The data is described in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>.</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.g004" xlink:type="simple"/>
      </fig>
      <table-wrap id="pone-0017293-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0017293.t001</object-id><label>Table 1</label><caption>
          <title>Description of 31 mammalian mitochondrial genome data in <xref ref-type="fig" rid="pone-0017293-g003">figure 3</xref>.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0017293-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.t001" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Number</td>
              <td align="left" colspan="1" rowspan="1">Genome name on the tree</td>
              <td align="left" colspan="1" rowspan="1">GenBank ID</td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="1" rowspan="1">1</td>
              <td align="left" colspan="1" rowspan="1">Human</td>
              <td align="left" colspan="1" rowspan="1">V00662</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">2</td>
              <td align="left" colspan="1" rowspan="1">pigmy chimpanzee</td>
              <td align="left" colspan="1" rowspan="1">D38116</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">3</td>
              <td align="left" colspan="1" rowspan="1">common chimpanzee</td>
              <td align="left" colspan="1" rowspan="1">D38113</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">4</td>
              <td align="left" colspan="1" rowspan="1">Gibbon</td>
              <td align="left" colspan="1" rowspan="1">X99256</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">5</td>
              <td align="left" colspan="1" rowspan="1">Baboon</td>
              <td align="left" colspan="1" rowspan="1">Y18001</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">6</td>
              <td align="left" colspan="1" rowspan="1">vervet monkey</td>
              <td align="left" colspan="1" rowspan="1">AY863426</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">7</td>
              <td align="left" colspan="1" rowspan="1">Macaca thibetana</td>
              <td align="left" colspan="1" rowspan="1">NC 002764</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">8</td>
              <td align="left" colspan="1" rowspan="1">bornean orang-utan</td>
              <td align="left" colspan="1" rowspan="1">D38115</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">9</td>
              <td align="left" colspan="1" rowspan="1">sumatran orang-utan</td>
              <td align="left" colspan="1" rowspan="1">NC 002083</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">10</td>
              <td align="left" colspan="1" rowspan="1">Gorilla</td>
              <td align="left" colspan="1" rowspan="1">D38114</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">11</td>
              <td align="left" colspan="1" rowspan="1">Cat</td>
              <td align="left" colspan="1" rowspan="1">U20753</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">12</td>
              <td align="left" colspan="1" rowspan="1">Dog</td>
              <td align="left" colspan="1" rowspan="1">U96639</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">13</td>
              <td align="left" colspan="1" rowspan="1">Pig</td>
              <td align="left" colspan="1" rowspan="1">AJ002189</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">14</td>
              <td align="left" colspan="1" rowspan="1">Sheep</td>
              <td align="left" colspan="1" rowspan="1">AF010406</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">15</td>
              <td align="left" colspan="1" rowspan="1">Goat</td>
              <td align="left" colspan="1" rowspan="1">AF533441</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">16</td>
              <td align="left" colspan="1" rowspan="1">Cow</td>
              <td align="left" colspan="1" rowspan="1">V00654</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">17</td>
              <td align="left" colspan="1" rowspan="1">Buffalo</td>
              <td align="left" colspan="1" rowspan="1">AY488491</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">18</td>
              <td align="left" colspan="1" rowspan="1">Wolf</td>
              <td align="left" colspan="1" rowspan="1">EU442884</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">19</td>
              <td align="left" colspan="1" rowspan="1">Tiger</td>
              <td align="left" colspan="1" rowspan="1">EF551003</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">20</td>
              <td align="left" colspan="1" rowspan="1">Leopard</td>
              <td align="left" colspan="1" rowspan="1">EF551002</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">21</td>
              <td align="left" colspan="1" rowspan="1">indian rhinoceros</td>
              <td align="left" colspan="1" rowspan="1">X97336</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">22</td>
              <td align="left" colspan="1" rowspan="1">white rhinoceros</td>
              <td align="left" colspan="1" rowspan="1">Y07726</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">23</td>
              <td align="left" colspan="1" rowspan="1">black bear</td>
              <td align="left" colspan="1" rowspan="1">DQ402478</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">24</td>
              <td align="left" colspan="1" rowspan="1">brown bear</td>
              <td align="left" colspan="1" rowspan="1">AF303110</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">25</td>
              <td align="left" colspan="1" rowspan="1">polar bear</td>
              <td align="left" colspan="1" rowspan="1">AF303111</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">26</td>
              <td align="left" colspan="1" rowspan="1">giant panda</td>
              <td align="left" colspan="1" rowspan="1">EF212882</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">27</td>
              <td align="left" colspan="1" rowspan="1">Rabbit</td>
              <td align="left" colspan="1" rowspan="1">AJ001588</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">28</td>
              <td align="left" colspan="1" rowspan="1">Hedgehog</td>
              <td align="left" colspan="1" rowspan="1">X88898</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">29</td>
              <td align="left" colspan="1" rowspan="1">Dormouse</td>
              <td align="left" colspan="1" rowspan="1">AJ001562</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">30</td>
              <td align="left" colspan="1" rowspan="1">Squirrel</td>
              <td align="left" colspan="1" rowspan="1">AJ238588</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">31</td>
              <td align="left" colspan="1" rowspan="1">blue whale</td>
              <td align="left" colspan="1" rowspan="1">X72204</td>
            </tr>
          </tbody>
        </table></alternatives></table-wrap>
    </sec>
    <sec id="s2">
      <title>Results</title>
      <p>As an application, we first use our method to analyze the new influenza A (H1N1) virus. Recent reports of widespread transmission of swine-origin influenza A (H1N1) viruses in humans in Mexico, the United States, and elsewhere, highlighted this ever-present threat to global public health <xref ref-type="bibr" rid="pone.0017293-Shinde1">[20]</xref>. Much effort has been made by using the experimental method and many important results have been obtained in the past <xref ref-type="bibr" rid="pone.0017293-Garten1">[18]</xref>, <xref ref-type="bibr" rid="pone.0017293-Novel1">[21]</xref>. Pigs have been hypothesized to act as a mixing vessel for the reassortment of avian, swine, and human influenza viruses and might play an important role in the emergence of novel influenza viruses capable of causing a human pandemic <xref ref-type="bibr" rid="pone.0017293-Scholtissek1">[22]</xref>–<xref ref-type="bibr" rid="pone.0017293-Ma1">[24]</xref>. There were many reports of recent transmissions of swine influenza viruses in humans <xref ref-type="bibr" rid="pone.0017293-Belshe1">[25]</xref>. The new strain was initially described as triple reassortants of viruses from pigs, humans, and birds, called triple-reassortant swine influenza A (H1) viruses, which have circulated in pigs for more than a decade <xref ref-type="bibr" rid="pone.0017293-Shinde1">[20]</xref>. Subsequent analysis suggested it was a reassortment of just two strains, both found in swine <xref ref-type="bibr" rid="pone.0017293-Novel1">[21]</xref>. Although initial reports identified the new strain as swine influenza (i.e., a zoonosis originating in swine), its origin is unknown from the point of view of whole genomes. Here we used our proposed method to verify the origin of A (H1N1) genomes. To demonstrate that our natural vector can be truly useful for answering biological questions, we performed hierarchical clustering analysis on the natural vectors of the genes of the swine influenza A (H1N1) virus. The Euclidean distance was used to measure the distance between natural vectors. Genomes of the outbreak of swine influenza A (H1N1), North American and Eurasian swine influenza virus genomes, avian and human seasonal influenza virus genomes were analyzed. Each complete genome contains 8 complete gene-coding segments. So we used a 96-dimensional natural vector to represent a whole genome since each segment can be characterized very well by using a 12-dimensional natural vector. Based on our novel mathematical method and result, we can predict that the genome of new swine influenza A (H1N1) is similar to swine viruses rather than human seasonal influenza and avian viruses. Using the natural vector method and clustering method, we have reconstructed the complex reassortment history of the outbreak of swine influenza A (H1N1), summarized in <xref ref-type="fig" rid="pone-0017293-g001">figure 1</xref>. Our analysis showed that the swine influenza A (H1N1) genome was nested within a well-established triple-reassortant swine influenza A and Eurasian swine influenza A lineage (that is, a lineage circulating primarily in swine before the current outbreak). In addition, we also analyzed 8 segments: polymerase PB2, PB1, PA, hemagglutinin HA, neuraminidase NA, nucleocapsid NP, matrix protein MP and nonstructural gene NS respectively in A H1N1 genome. Our results showed that HA, NP, NS genes resemble those of classical swine influenza A viruses and PB2, PB1, PA genes resemble those of triple-reassortant swine influenza A viruses circulating in pigs in North America while the genes NA and MP are most closely related to those in influenza A viruses circulating in swine populations in Eurasia. The clustering results of these 8 gene segments obtained by our method coincides with the phylogenetic analysis results from Garten et al. <xref ref-type="bibr" rid="pone.0017293-Garten1">[18]</xref> and Novel Swine-Origin Influenza A (H1N1) Virus Investigation Team <xref ref-type="bibr" rid="pone.0017293-Novel1">[21]</xref>. These conclusions have been widely accepted by other scientists <xref ref-type="bibr" rid="pone.0017293-Kingsford1">[35]</xref> in the scientific community. Therefore, this result shows that Kou et al.'s conclusion <xref ref-type="bibr" rid="pone.0017293-Kou1">[26]</xref> was not fully convincing since they concluded that PB2 and PA genes came from avian influenza virus and PB1 from human seasonal influenza virus. As an illustration, the phylogenetic analysis result of PB2 is shown in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>. The rest results of seven individual segments are available from the author upon request. In this biological experiment, 12 dimensional natural vectors, &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e002" xlink:type="simple"/></inline-formula>&gt; were used for clustering the swine influenza A (H1N1) based on the gene sequences, since the higher moments in the natural vector were too small to play a role when <italic>n</italic> is large. The gene and genome data are provided in the section of Supporting Information. They can be downloaded from Flu Database of GenBank (<ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov/genomes/FLU/FLU.html" xlink:type="simple">http://www.ncbi.nlm.nih.gov/genomes/FLU/FLU.html</ext-link>).</p>
      <p>In addition, we applied our approach to study another group of viruses, human rhinovirus (HRV). Infection by HRV is a major cause of upper and lower respiratory disease worldwide and displays considerable phenotypic variation. Recently, Palmenberg et al. <xref ref-type="bibr" rid="pone.0017293-Palmenberg1">[19]</xref> reported a comprehensive sequencing and analyzed result for all known HRV genomes based on the whole genome. In their article, the authors used the multiple alignment method to reconstruct the evolutionary tree. In that tree, five groups HRV-A, HRV-B, HRV-C, HEV-B and HEV-C were clearly identified. We used our natural vector method to perform clustering analysis for the same dataset (<xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>). We associated each whole genome sequence with a natural vector, and then by computing the Euclidean distances among these natural vectors we obtained the evolutionary tree (<xref ref-type="fig" rid="pone-0017293-g002">figure 2</xref>) for all HRVs. MEGA software was used to draw the tree <xref ref-type="bibr" rid="pone.0017293-Kumar1">[27]</xref>. According to our result, the five clusters HRV-A, HRV-B, HRV-C, HEV-B, and HEV-C are clearly separated from each other. Our method takes only 18 seconds to complete the clustering analysis result while it takes more than 19 hours for the multiple alignment method. Both methods yield the same clustering result.</p>
      <p>As another biological application, we consider the phylogeny of mitochondrial genomes. Mitochondrial DNA is not highly conserved and has a rapid mutation rate, thus it is very useful for studying the evolutionary relationships of organisms <xref ref-type="bibr" rid="pone.0017293-Brown1">[28]</xref>. We extracted 31 representative cases of complete mammalian mitochondrial genome sequences from the GenBank, each of which has length of more than 16000 nucleotides. Moreover, they have double-strand and circular structures. As mentioned in the section of <xref ref-type="sec" rid="s4">Materials and Methods</xref>, we just treat them as the single-strand (by using the heavy strand) circular genomes, because the gene contest of both strands of these genomes is already known. For this case, we treat every point as the starting point in this circular sequence of length n, and then we get n linear single-strand genomes. For every linear single-strand genome sequence, we can compute its (<italic>n+4</italic>)-dimensional natural vector. Then we take the average to get a normalized vector &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e003" xlink:type="simple"/></inline-formula>&gt;. Here we use the first 16 moments of the natural vector, i.e., &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e004" xlink:type="simple"/></inline-formula>&gt; to characterize these 31 genomes. By computing the Euclidean distances between these points, we obtain the distance matrix for these 31 organisms. The phylogenetic tree is shown in <xref ref-type="fig" rid="pone-0017293-g003">figure 3</xref>. The result shows these 31 genomes are well clustered into 7 clusters: Erinaceomorpha, Primates, Carnivore, Perissodactyla, Cetacea and Artiodactyla, Lagomorpha and Rodentia, where Cetacea and Artiodactyla form a sister-group since they are grouped together. This result coincides with the conclusion found by Liu et al <xref ref-type="bibr" rid="pone.0017293-Liu2">[29]</xref>, Raina et al <xref ref-type="bibr" rid="pone.0017293-Raina1">[30]</xref>, and Kullberg et al <xref ref-type="bibr" rid="pone.0017293-Kullberg1">[31]</xref>.</p>
      <p>In the <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>, we discuss the distribution of distance of the corresponding natural vectors under controlled simulation. 1000 simulated sequences are generated by shuffling a real genome sequence and the distribution of pair-wise distance of the natural vectors Ls is plotted. We also consider the following simulated experiment on gene rearrangement (please refer to <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref> for more details). We choose a human mitochondrial genome denoted by human (GenBank ID: V00662), and then invert its two genes ATPase 6 and Cytochrome oxidase to get a simulated genome, which is denoted by human-inv. We can treat this new simulated genome as the result of the inversion of genes from the original human mitochondrial genome. Next we randomly generate a genome sequence which has the same length and nucleotide content as the original human genome, which we denote by human-ran. Thus, we have 3 genomes of the same length and nucleotide content: human, human-inv and human-ran. In addition, we also choose another mitochondrial genome, chimpanzee as comparison since chimpanzee and human are so close evolutionarily. By using the 16-dimmensional natural vector, we calculate the distance among these 4 genomes and get a distance matrix in the <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>. We find that the distance between human and human-inv is as small as 1.7 units. This means that the gene-inverted genome still has a very short distance to the original genome even if some gene rearrangement happens in this genome. As a result, the original genome and its gene rearrangement genome cannot be treated separately by using our method since the evolutionarily very close genome to human is chimpanzee which has the distance of 16.88 units to human. More importantly, this simulation demonstrates that our method can be applied to do clustering or phylogenetic analysis. If two genetic sequences are close in the distance, they should be close in the evolutionary tree. For example, the distance between two real genomes, human and chimpanzee is 16.88 units. Since the distance between the original human genome and all shuffled genomes ranges from 114.49 to 1009.00 with the mean value of 190.60 units (<xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>), all those randomly shuffled sequences cannot be clustered together with the human. Here we use standard multiple alignments with substitutions and indels on 21 flu virus genomes (<xref ref-type="fig" rid="pone-0017293-g004">figure 4</xref>). We performed the multiple alignment method using ClustalX and Phylip to draw the phylogenetic tree for the dataset consisting of 6 influenza A (H1N1) genomes, 6 swine flu virus genomes, 6 avian virus genomes and 3 human seasonal flu virus genomes (Please see <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref> for the description of these genomes). These genomes are selected from the dataset shown in <xref ref-type="fig" rid="pone-0017293-g001">figure 1</xref>. We reconstructed the phylogenetic trees by maximum likelihood method and neighbor-joining method with different models shown in <xref ref-type="fig" rid="pone-0017293-g004">figure 4(a, b, c)</xref>. The phylogenetic result obtained by using maximum likelihood method (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(a)</xref>) showed that the swine flu virus genomes were not clustered correctly. The result by using neighbor-joining method with Kimura model did not obviously show the origin of A (H1N1) genomes (<xref ref-type="fig" rid="pone-0017293-g004">figure 4(b)</xref>), because all A (H1N1) genomes were very far away from other genomes. Besides, the result obtained from neighbour-joining method with Jukes-Cantor model also fails to cluster swine flu viruses correctly. Our method showed that influenza A (H1N1) genomes are close to swine flu virus genomes (<xref ref-type="fig" rid="pone-0017293-g001">figure 1</xref>). Therefore, we predict that A (H1N1) genomes are originally from swine flu virus genome lineage (triple-reassortant North American and Eurasian swine virus genomes). The advantage of our method is the quick speed. It took us 60 minutes 31 seconds to finish the complete alignment for these 21 sequences by using multiple alignment method <xref ref-type="bibr" rid="pone.0017293-Larkin1">[14]</xref> whereas we just need 6 seconds to get the result.</p>
      <p>In order to compare the computation time of the natural vector method and state-of-the-art methods ClustalW2, MUSCLE and MAFFT <xref ref-type="bibr" rid="pone.0017293-Larkin1">[14]</xref>, <xref ref-type="bibr" rid="pone.0017293-Edgar1">[15]</xref>, <xref ref-type="bibr" rid="pone.0017293-Katoh1">[16]</xref>, we performed the test on two sets of sequences. The first set included 8 datasets. The datasets contain 10, 20, 30, 40, 50, 60, 70 and 80 sequences respectively, where the lengths of all the sequences are around 4000. Another set was created with 8 datasets. Each of these dataset had 40 sequences. The lengths of all sequences in these 8 datasets were 1000, 2000, 3000, 4000, 5000, 6000, 7000 and 8000 respectively. We built the trees on each of the datasets by using the four methods and recorded the time that each method took. The results shown in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref> demonstrate that natural vector method is much faster than the other three methods. The time of our method increases linearly as the number of sequences or the length of sequences increase, whereas the time for other three methods increases much faster. The actual time differences are much larger than the visual differences in the figure since we are using the logarithm of time as the label of y-axis.</p>
    </sec>
    <sec id="s3">
      <title>Discussion</title>
      <p>In this paper, we report a new mathematical method to characterize a genetic sequence as a natural vector so we can perform clustering analysis and create a phylogenetic tree based on it. A natural vector system to represent a DNA sequence is introduced, and the correspondence between a DNA sequence and its natural vector is mathematically proved to be one-to-one. With this natural vector system, each genome sequence can be represented as a multidimensional vector. Genomes with a close evolutionary relationship and similar properties are plotted close to each other when we construct the phylogenetic tree. Thus, it will provide a new powerful tool for analyzing and annotating genomes and their phylogenetic relationships. Our method is easier and quicker in handling whole or partial genomes than multiple alignment methods. There are four major advantages to our method: (1) once a genome space has been constructed, it can be stored in a database. There is no need to reconstruct the genome space for any subsequent application, whereas in multiple alignment methods, realignment is needed for adding new sequences. (2) One can perform global comparison of all genomes simultaneously, which no other existing method can achieve. (3) Our method is quicker than alignment methods and easier to manipulate, because not all dimensions of natural vectors are needed for computing. Instead, the first several dimensions of natural vectors are good enough to cluster DNA sequences or genomes. Generally, we select the first <italic>N</italic> dimensions such that the clustering result remains stable even if we choose higher moments. N = 12 in our experiments is good enough to characterise all sequences. We can compare all genes, DNA and genome sequences with different lengths by truncating all different (n+4) natural vectors into the same number of dimensions. The one-to-one correspondence between the truncated natural vectors (with 12 or more dimensions) and sequences is still valid. (4) The current standard methods involve the evolutionary models. The different choices of these evolutionary models can lead to inconsistent results (<xref ref-type="fig" rid="pone-0017293-g004">figure 4 (a, b, c)</xref>). There is no evidence to show which model can best fit all biological datasets without human intervention (likelihood ratio test used as a priority). This motivates us to create a new mathematical method without any model. Our method does not involve these models and it totally depends on the natural vectors constructed from the whole sequences. Therefore, this method is stable, natural and produces a unique clustering or phylogenetic result.</p>
      <p>Although the natural vector method can be used to reconstruct the phylogenetic trees of DNA sequences, genes and whole genomes, this method may not be a suitab le substitute for local multiple sequence alignment when one wants to identify the similarity of genomic subsequences and does not know a priori which subsequences to identify.</p>
    </sec>
    <sec id="s4" sec-type="materials|methods">
      <title>Materials and Methods</title>
      <sec id="s4a">
        <title>Natural vector of a DNA sequence</title>
        <p>Let us first introduce the definition of normalized central moments which is the most important part of natural vector method. Normalized central moments are defined as follows:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e005" xlink:type="simple"/></disp-formula>where <italic>k</italic> = A, C, G, T. Here, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e006" xlink:type="simple"/></inline-formula> denotes the number of nucleotide k in the DNA sequence and n is the length of the DNA sequence. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e007" xlink:type="simple"/></inline-formula> is the distance from the first nucleotide (regarded as origin) to the ith nucleotide k in the DNA sequence. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e008" xlink:type="simple"/></inline-formula> denotes the total distance of each set of A, C, G, T from the origin, k = A, C, G, T. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e009" xlink:type="simple"/></inline-formula>, which is the mean value of the distances of the nucleic bases from the origin. Therefore, we have the sequence of central moments: &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e010" xlink:type="simple"/></inline-formula>&gt;, Observe that these are natural parameters associated to a DNA sequence.</p>
        <p>Our method described below is to give a complete understanding of the distribution of four nucleotides A, C, G and T.</p>
        <p>The quantities of the four nucleotides: A, C, G and T of a DNA sequence are chosen as the first four parameters of the natural vector. Four integers <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e011" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e012" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e013" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e014" xlink:type="simple"/></inline-formula> denote the numbers of nucleic bases A, C, G and T in the DNA sequence.</p>
        <p>The second group of numerical parameters which are a part of the natural vector are the mean values of total distance, one for each of the four nucleotide bases: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e015" xlink:type="simple"/></inline-formula>, <italic>k = A, C, G and T</italic>.</p>
        <p>As a simple illustration for the DNA sequence GTTCAATACT: The total distance of A is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e016" xlink:type="simple"/></inline-formula>, since the distance of origin to the three nucleotide As is 4, 5 and 7 respectively. Then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e017" xlink:type="simple"/></inline-formula>. The arithmetic mean value of total distance for other nucleotide base G, C and T can be obtained in the same way.</p>
        <p>The final group of parameters that we include in the natural vector are composed of normalized central moments. The first normalized central moment is:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e018" xlink:type="simple"/></disp-formula>Because the first central moment is zero, we start with the second normalized central moment. The second normalized central moment is the variance of the distance distribution for each base: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e019" xlink:type="simple"/></inline-formula>, where <italic>k</italic> = A, C, G, T. If the distribution of each nucleotide base is different, DNA sequences cannot be same even though they may have the same nucleotide contents and the same total distance measurement. Therefore, the information about distribution has also been included in the natural vector. As described above, each subset of numerical parameters is not sufficient to annotate DNA sequences. However, the combined numerical parameters are sufficient to characterize each DNA sequence. So the natural vector is given as follows:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e020" xlink:type="simple"/></disp-formula>In order to express the vector elegantly and prove the theorem easily, we rewrite it as follows:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e021" xlink:type="simple"/><label>(1)</label></disp-formula>Alternatively, the natural vector can be written as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e022" xlink:type="simple"/><label>(2)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e023" xlink:type="simple"/></inline-formula>. By the definition, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e024" xlink:type="simple"/></inline-formula>, if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e025" xlink:type="simple"/></inline-formula>. For instance, this case happens when we compute &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e026" xlink:type="simple"/></inline-formula>&gt; if there are 19 A, 15 C, 21 G and 22 T in the sequence. Therefore, the 20<sup>th</sup> moment will be &lt;0, 0, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e027" xlink:type="simple"/></inline-formula>&gt;. The natural vector is obtained by concatenating the first group of parameters (the number of each base) and the second group of parameters (the mean value of total distance of each base) to the normalized central moments.</p>
        <p>Obviously, higher moments converge to 0 for a random generated sequence since for any given <italic>k</italic>,<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e028" xlink:type="simple"/></disp-formula>It is clear that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e029" xlink:type="simple"/></inline-formula>, otherwise, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e030" xlink:type="simple"/></inline-formula>. From the viewpoint of probability, suppose that the expectation value of any nucleic base is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e031" xlink:type="simple"/></inline-formula> (uniform distribution) for a sequence with given length <italic>n</italic>, therefore<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.e032" xlink:type="simple"/></disp-formula>Clearly, this limit goes to 0 as <italic>j</italic> approaches <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e033" xlink:type="simple"/></inline-formula>. In molecular biology, we can simply discuss the number <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e034" xlink:type="simple"/></inline-formula> from GC-content. GC-content (or guanine-cytosine content), in molecular biology, is the percentage of nitrogenous bases on a DNA molecule which are either guanine or cytosine. GC content is found to be variable with different organisms. Because of the nature of the genetic code, it is however virtually impossible for an organism to have a genome with a GC-content approaching either 0% or 100%. A species with an extremely low GC-content is Plasmodium falciparum (GC% = ∼20%) <xref ref-type="bibr" rid="pone.0017293-Musto1">[32]</xref>. Therefore, for any simulated dataset or biological dataset, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e035" xlink:type="simple"/></inline-formula> always converges to 0 when <italic>j</italic> approaches <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e036" xlink:type="simple"/></inline-formula>. For a simulation, we generate a random sequence with length of 10000 nucleotides by using Hidden Markov model (Matlab, bioinformatic toolbox). The simulated sequence contains 2345 A, 2761 C, 2544 G and 2350 T. For adenines, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e037" xlink:type="simple"/></inline-formula>, i.e., higher normalized central moments starting from 4<sup>th</sup> moment will converge to 0.</p>
        <p>We have used natural vector to obtain a good numerical characterization of DNA sequence. We now discuss the construction of natural vectors of genomes. Generally, for a linear single-strand genome, we treat it as a linear DNA sequence while we treat every point as the starting point and then take average for circular single-strand genomes. For general double-strand genomes, we treat them as two single-strand genomes and then take average. More details are discussed in <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>.</p>
      </sec>
      <sec id="s4b">
        <title>Theorem</title>
        <p>One of the most important things in this paper is that we can prove that the correspondence between a DNA sequence and its natural vector is one-to-one (see <xref ref-type="supplementary-material" rid="pone.0017293.s001">Supporting Information S1</xref>).</p>
        <p><bold>Theorem:</bold> Suppose a DNA sequence has <italic>n</italic> nucleotides. Then the correspondence between a DNA sequence and its natural vector &lt;<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e038" xlink:type="simple"/></inline-formula>&gt; is one-to-one, where n = <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e039" xlink:type="simple"/></inline-formula>+<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e040" xlink:type="simple"/></inline-formula>+<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e041" xlink:type="simple"/></inline-formula>+<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e042" xlink:type="simple"/></inline-formula>.</p>
        <p>The Euclidean distance between two sequences <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e043" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e044" xlink:type="simple"/></inline-formula> is defined as the distance of their corresponding natural vectors: <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e045" xlink:type="simple"/></inline-formula>, where <italic>i</italic> = A, C, G, T; <italic>j</italic> = <italic>n</italic>,<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e046" xlink:type="simple"/></inline-formula>,<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e047" xlink:type="simple"/></inline-formula>,<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e048" xlink:type="simple"/></inline-formula>,…,<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e049" xlink:type="simple"/></inline-formula>. <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0017293.e050" xlink:type="simple"/></inline-formula> is the number of each A, C, G, T.</p>
        <p>We used Matlab to calculate the natural vectors of genes and genomes. The package HCLUST of R language (for algorithmic details, please refer to <xref ref-type="bibr" rid="pone.0017293-Murtagh1">[33]</xref>) was used to perform the hierarchical cluster analysis of genomes and genes. R package PVCLUST <xref ref-type="bibr" rid="pone.0017293-Kamimura1">[34]</xref> was used to calculate approximately unbiased p-value and bootstrap probability value by multiscale bootstrap resampling and draw the standard error plot. The codes are available from the author upon request.</p>
      </sec>
    </sec>
    <sec id="s5">
      <title>Supporting Information</title>
      <supplementary-material id="pone.0017293.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0017293.s001" xlink:type="simple">
        <label>Supporting Information S1</label>
        <caption>
          <p>Supporting information S1 contains the complete proof of the correspondence theorem, the bootstrapping analysis on A H1N1 genomes, distribution of the distance between each pair of random shuffled genomes under simulation, clustering of the segmented gene PB2, computational time chart of natural vector method, ClustalW2, MUSCLE and MAFFT, and the Genbank ID of the data used in this paper.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>We thank Dr. Max Benson for critically reading and editing the manuscript. We also thank the editor and anonymous reviewers for thorough review and constructive comments.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pone.0017293-Amano1">
        <label>1</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Amano</surname><given-names>K</given-names></name><name name-style="western"><surname>Nakamura</surname><given-names>H</given-names></name></person-group>             <year>2003</year>             <article-title>Self-organizing clustering: a novel non-hierarchical method for clustering large amount of DNA sequences.</article-title>             <source>Genome Inform</source>             <volume>14</volume>             <fpage>575</fpage>             <lpage>576</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Emrich1">
        <label>2</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Emrich</surname><given-names>SJ</given-names></name><name name-style="western"><surname>Kalyanaraman</surname><given-names>A</given-names></name><name name-style="western"><surname>Aluru</surname><given-names>S</given-names></name></person-group>             <year>2006</year>             <article-title>Algorithms for large-scale clustering and assembly of biological sequence data.</article-title>             <person-group person-group-type="editor"><name name-style="western"><surname>Aluru</surname><given-names>S</given-names></name></person-group>             <source>Handbook of Computational Molecular Biology</source>             <fpage>13.1</fpage>             <lpage>13.30</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-FitzGerald1">
        <label>3</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>FitzGerald</surname><given-names>PC</given-names></name><name name-style="western"><surname>Shlyakhtenko</surname><given-names>A</given-names></name><name name-style="western"><surname>Mir</surname><given-names>A</given-names></name><name name-style="western"><surname>Vinson</surname><given-names>C</given-names></name></person-group>             <year>2004</year>             <article-title>Clustering of DNA sequences in human promoters.</article-title>             <source>Genome Res</source>             <volume>14</volume>             <fpage>1562</fpage>             <lpage>1574</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Waterman1">
        <label>4</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Waterman</surname><given-names>SM</given-names></name></person-group>             <year>1995</year>             <source>Introduction to computational biology: maps, sequences and genomes</source>             <publisher-loc>Boca Raton</publisher-loc>             <publisher-name>Chapman &amp; Hall/CRC Press</publisher-name> <!--===== Restructure page-count as size[@units="page"] =====--><size units="page">431</size>           </element-citation>
      </ref>
      <ref id="pone.0017293-Abe1">
        <label>5</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Abe</surname><given-names>T</given-names></name><name name-style="western"><surname>Kanaya</surname><given-names>S</given-names></name><name name-style="western"><surname>Kinouchi</surname><given-names>M</given-names></name><name name-style="western"><surname>Ichiba</surname><given-names>Y</given-names></name><name name-style="western"><surname>Kozuki</surname><given-names>T</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>Informatics for unveiling hidden genome signatures.</article-title>             <source>Genome Research</source>             <volume>13</volume>             <fpage>693</fpage>             <lpage>702</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Chuzhanova1">
        <label>6</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chuzhanova</surname><given-names>NA</given-names></name><name name-style="western"><surname>Jones</surname><given-names>AJ</given-names></name><name name-style="western"><surname>Margetts</surname><given-names>S</given-names></name></person-group>             <year>1998</year>             <article-title>Feature selection for genetic sequence classification.</article-title>             <source>Bioinformatics</source>             <volume>14</volume>             <fpage>139</fpage>             <lpage>143</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Karlin1">
        <label>7</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Karlin</surname><given-names>S</given-names></name><name name-style="western"><surname>Ladunga</surname><given-names>I</given-names></name></person-group>             <year>1994</year>             <article-title>Comparisons of eukaryotic genomic sequences.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>91</volume>             <fpage>12832</fpage>             <lpage>12836</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Nakashima1">
        <label>8</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nakashima</surname><given-names>H</given-names></name><name name-style="western"><surname>Ota</surname><given-names>M</given-names></name><name name-style="western"><surname>Nishikawa</surname><given-names>K</given-names></name><name name-style="western"><surname>Ooi</surname><given-names>T</given-names></name></person-group>             <year>1998</year>             <article-title>Genes from nine genomes are separated into their organisms in the dinucleotide composition space.</article-title>             <source>DNA Res</source>             <volume>5</volume>             <fpage>251</fpage>             <lpage>259</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Yau1">
        <label>9</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Yau</surname><given-names>S</given-names></name><name name-style="western"><surname>Wang</surname><given-names>J</given-names></name><name name-style="western"><surname>Niknejad</surname><given-names>A</given-names></name><name name-style="western"><surname>Lu</surname><given-names>C</given-names></name><name name-style="western"><surname>Jin</surname><given-names>N</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>DNA sequence representation without degeneracy.</article-title>             <source>Nucl Acids Res</source>             <volume>31</volume>             <fpage>3078</fpage>             <lpage>3080</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Liu1">
        <label>10</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names></name><name name-style="western"><surname>Ho</surname><given-names>Y-K</given-names></name><name name-style="western"><surname>Yau</surname><given-names>S</given-names></name></person-group>             <year>2006</year>             <article-title>Clustering DNA sequences by feature vectors.</article-title>             <source>Mol Phyl Evol</source>             <volume>41</volume>             <fpage>64</fpage>             <lpage>69</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Yau2">
        <label>11</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Yau</surname><given-names>SS-T</given-names></name><name name-style="western"><surname>Yu</surname><given-names>C</given-names></name><name name-style="western"><surname>He</surname><given-names>R</given-names></name></person-group>             <year>2008</year>             <article-title>A protein map and its application.</article-title>             <source>DNA and Cell Biol</source>             <volume>27</volume>             <fpage>241</fpage>             <lpage>250</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Carr1">
        <label>12</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Carr</surname><given-names>K</given-names></name><name name-style="western"><surname>Murray</surname><given-names>E</given-names></name><name name-style="western"><surname>Armah</surname><given-names>E</given-names></name><name name-style="western"><surname>He</surname><given-names>RL</given-names></name><name name-style="western"><surname>Yau</surname><given-names>SS-T</given-names></name></person-group>             <year>2010</year>             <article-title>A rapid method for characterization of protein relatedness using feature vectors.</article-title>             <source>PLoS One</source>             <volume>5</volume>             <issue>3</issue>             <fpage>e9550</fpage>             <comment>doi:<ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1371/journal.pone.0009550" xlink:type="simple">10.1371/journal.pone.0009550</ext-link></comment>          </element-citation>
      </ref>
      <ref id="pone.0017293-Yu1">
        <label>13</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>C</given-names></name><name name-style="western"><surname>Liang</surname><given-names>Q</given-names></name><name name-style="western"><surname>Yin</surname><given-names>C</given-names></name><name name-style="western"><surname>He</surname><given-names>RL</given-names></name><name name-style="western"><surname>Yau</surname><given-names>SS-T</given-names></name></person-group>             <year>2010</year>             <article-title>A novel construction of genome space with biological geometry.</article-title>             <source>DNA Res</source>             <volume>17</volume>             <issue>3</issue>             <fpage>155</fpage>             <lpage>168</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Larkin1">
        <label>14</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Larkin</surname><given-names>MA</given-names></name><name name-style="western"><surname>Blackshields</surname><given-names>G</given-names></name><name name-style="western"><surname>Brown</surname><given-names>NP</given-names></name><name name-style="western"><surname>Chenna</surname><given-names>R</given-names></name><name name-style="western"><surname>McGettigan</surname><given-names>PA</given-names></name><etal/></person-group>             <year>2007</year>             <article-title>Clustal W and Clustal X version 2.0.</article-title>             <source>Bioinformatics</source>             <volume>23</volume>             <fpage>2947</fpage>             <lpage>2948</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Edgar1">
        <label>15</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Edgar</surname><given-names>RC</given-names></name></person-group>             <year>2004</year>             <article-title>MUSCLE: a multiple sequence alignment method with reduced time and space complexity.</article-title>             <source>BMC Bioinformatics</source>             <volume>5</volume>             <fpage>113</fpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Katoh1">
        <label>16</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Katoh</surname><given-names>K</given-names></name><name name-style="western"><surname>Misawa</surname><given-names>K</given-names></name><name name-style="western"><surname>Kuma</surname><given-names>K</given-names></name><name name-style="western"><surname>Miyata</surname><given-names>T</given-names></name></person-group>             <year>2002</year>             <article-title>MAFFT: a novel method for rapid multiple sequence alignment based on fast Fourier transform.</article-title>             <source>Nucl Acids Res</source>             <volume>30</volume>             <issue>14</issue>             <fpage>3059</fpage>             <lpage>3066</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Wang1">
        <label>17</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names></name><name name-style="western"><surname>Jiang</surname><given-names>T</given-names></name></person-group>             <year>1994</year>             <article-title>On the complexity of multiple sequence alignment.</article-title>             <source>J Comput Biol</source>             <volume>13</volume>             <issue>7</issue>             <fpage>1323</fpage>             <lpage>1339</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Garten1">
        <label>18</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Garten</surname><given-names>RJ</given-names></name><name name-style="western"><surname>Davis</surname><given-names>CT</given-names></name><name name-style="western"><surname>Russell</surname><given-names>CA</given-names></name><name name-style="western"><surname>Shu</surname><given-names>B</given-names></name><name name-style="western"><surname>Lindstrom</surname><given-names>S</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Antigenic and Genetic Characteristics of Swine-Origin 2009 A (H1N1) Influenza Viruses Circulating in Humans.</article-title>             <source>Science</source>             <volume>325</volume>             <issue>5937</issue>             <fpage>197</fpage>             <lpage>201</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Palmenberg1">
        <label>19</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Palmenberg</surname><given-names>A</given-names></name><name name-style="western"><surname>Spiro</surname><given-names>D</given-names></name><name name-style="western"><surname>Kuzmickas</surname><given-names>R</given-names></name><name name-style="western"><surname>Wang</surname><given-names>S</given-names></name><name name-style="western"><surname>Djikeng</surname><given-names>A</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Sequencing and analyses of all known human rhinovirus genomes reveal structure and evolution.</article-title>             <source>Science</source>             <volume>324</volume>             <fpage>55</fpage>             <lpage>59</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Shinde1">
        <label>20</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Shinde</surname><given-names>V</given-names></name><name name-style="western"><surname>Bridges</surname><given-names>CB</given-names></name><name name-style="western"><surname>Uyeki</surname><given-names>TM</given-names></name><name name-style="western"><surname>Shu</surname><given-names>B</given-names></name><name name-style="western"><surname>Balish</surname><given-names>A</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Triple-reassortant swine influenza A (H1) in humans in the United States, 2005–2009.</article-title>             <source>New Engl J Med</source>             <volume>360</volume>             <fpage>2616</fpage>             <lpage>2625</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Novel1">
        <label>21</label>
        <element-citation publication-type="journal" xlink:type="simple">             <collab xlink:type="simple">Novel Swine-Origin Influenza A (H1N1) Virus Investigation Team</collab>             <year>2009</year>             <article-title>Emergence of a novel swine-origin influenza A (H1N1) virus in humans.</article-title>             <source>New Engl J Med</source>             <volume>360</volume>             <issue>25</issue>             <fpage>2605</fpage>             <lpage>2615</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Scholtissek1">
        <label>22</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Scholtissek</surname><given-names>C</given-names></name></person-group>             <year>1990</year>             <article-title>Pigs as ‘mixing vessels’ for the creation of new pandemic influenza A viruses.</article-title>             <source>Med Princ Pract</source>             <volume>2</volume>             <fpage>65</fpage>             <lpage>71</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Ito1">
        <label>23</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ito</surname><given-names>T</given-names></name><name name-style="western"><surname>Couceiro</surname><given-names>JN</given-names></name><name name-style="western"><surname>Kelm</surname><given-names>S</given-names></name><name name-style="western"><surname>Baum</surname><given-names>LG</given-names></name><name name-style="western"><surname>Krauss</surname><given-names>S</given-names></name><etal/></person-group>             <year>1998</year>             <article-title>Molecular basis for the generation in pigs of influenza A viruses with pandemic potential.</article-title>             <source>J Virol</source>             <volume>72</volume>             <fpage>7367</fpage>             <lpage>7373</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Ma1">
        <label>24</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>W</given-names></name><name name-style="western"><surname>Kahn</surname><given-names>RE</given-names></name><name name-style="western"><surname>Richt</surname><given-names>JA</given-names></name></person-group>             <year>2009</year>             <article-title>The pig as a mixing vessel for influenza viruses: human and veterinary implications.</article-title>             <source>J Mol Genet Med</source>             <volume>3</volume>             <fpage>158</fpage>             <lpage>166</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Belshe1">
        <label>25</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Belshe</surname><given-names>RB</given-names></name></person-group>             <year>2009</year>             <article-title>Implications of the emergence of a novel H1 Influenza virus.</article-title>             <source>New Engl J Med</source>             <volume>360</volume>             <issue>25</issue>             <fpage>2667</fpage>             <lpage>2668</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Kou1">
        <label>26</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kou</surname><given-names>Z</given-names></name><name name-style="western"><surname>Hu</surname><given-names>S</given-names></name><name name-style="western"><surname>Li</surname><given-names>T</given-names></name></person-group>             <year>2009</year>             <article-title>Genome evolution of novel influenza A (H1N1) viruses in humans.</article-title>             <source>Chin Sci Bul</source>             <volume>54</volume>             <fpage>2159</fpage>             <lpage>2163</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Kumar1">
        <label>27</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kumar</surname><given-names>S</given-names></name><name name-style="western"><surname>Dudley</surname><given-names>J</given-names></name><name name-style="western"><surname>Nei</surname><given-names>M</given-names></name><name name-style="western"><surname>Tamura</surname><given-names>K</given-names></name></person-group>             <year>2008</year>             <article-title>MEGA: A biologist-centric software for evolutionary analysis of DNA and protein sequences.</article-title>             <source>Brief Bioinform</source>             <volume>9</volume>             <fpage>299</fpage>             <lpage>306</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Brown1">
        <label>28</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>WM</given-names></name><name name-style="western"><surname>Prager</surname><given-names>EM</given-names></name><name name-style="western"><surname>Wang</surname><given-names>A</given-names></name><name name-style="western"><surname>Wilson</surname><given-names>AC</given-names></name></person-group>             <year>1982</year>             <article-title>Mitochondrial DNA sequences of primates: Tempo and mode of evolution.</article-title>             <source>J Mol Evol</source>             <volume>18</volume>             <fpage>225</fpage>             <lpage>239</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Liu2">
        <label>29</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>F</given-names></name><name name-style="western"><surname>Miyamoto</surname><given-names>M</given-names></name><name name-style="western"><surname>Freire</surname><given-names>N</given-names></name><name name-style="western"><surname>Ong</surname><given-names>P</given-names></name><name name-style="western"><surname>Tennant</surname><given-names>M</given-names></name><etal/></person-group>             <year>2001</year>             <article-title>Molecular and morphological supertrees for eutherian (placental) mammals.</article-title>             <source>Science</source>             <volume>291</volume>             <fpage>1786</fpage>             <lpage>1789</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Raina1">
        <label>30</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Raina</surname><given-names>SZ</given-names></name><name name-style="western"><surname>Faith</surname><given-names>JJ</given-names></name><name name-style="western"><surname>Disotell</surname><given-names>TR</given-names></name><name name-style="western"><surname>Seligmann</surname><given-names>H</given-names></name><name name-style="western"><surname>Stewart</surname><given-names>CB</given-names></name><etal/></person-group>             <year>2005</year>             <article-title>Evolution of base-substitution gradients in primate mitochondrial genomes.</article-title>             <source>Genome Res</source>             <volume>15</volume>             <fpage>665</fpage>             <lpage>673</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Kullberg1">
        <label>31</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kullberg</surname><given-names>M</given-names></name><name name-style="western"><surname>Nilsson</surname><given-names>M</given-names></name><name name-style="western"><surname>Arnason</surname><given-names>U</given-names></name><name name-style="western"><surname>Harley</surname><given-names>E</given-names></name><name name-style="western"><surname>Janke</surname><given-names>A</given-names></name></person-group>             <year>2006</year>             <article-title>Housekeeping genes for phylogenetic analysis of Eutherian relationships.</article-title>             <source>Mol Biol Evol</source>             <volume>23</volume>             <issue>8</issue>             <fpage>1493</fpage>             <lpage>1503</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Musto1">
        <label>32</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Musto</surname><given-names>H</given-names></name><name name-style="western"><surname>Cacciò</surname><given-names>S</given-names></name><name name-style="western"><surname>Rodríguez-Maseda</surname><given-names>H</given-names></name><name name-style="western"><surname>Bernardi</surname><given-names>G</given-names></name></person-group>             <year>1997</year>             <article-title>Compositional constraints in the extremely GC-poor genome of <italic>Plasmodium falciparum</italic>.</article-title>             <source><italic>Mem</italic> Inst Oswaldo Cruz</source>             <volume>92</volume>             <issue>6</issue>             <fpage>835</fpage>             <lpage>841</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Murtagh1">
        <label>33</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Murtagh</surname><given-names>F</given-names></name></person-group>             <year>1985</year>             <article-title>Multidimensional clustering algorithms.</article-title>             <person-group person-group-type="editor"><name name-style="western"><surname>Chambers</surname><given-names>JM</given-names></name><name name-style="western"><surname>Gordesch</surname><given-names>J</given-names></name><name name-style="western"><surname>Klas</surname><given-names>A</given-names></name><name name-style="western"><surname>Lebart</surname><given-names>L</given-names></name><name name-style="western"><surname>Sint</surname><given-names>PP</given-names></name></person-group>             <source>COMPSTAT lectures vol 4</source>             <publisher-loc>Vienna</publisher-loc>             <publisher-name>Physica-Verlag/Springer Press</publisher-name> <!--===== Restructure page-count as size[@units="page"] =====--><size units="page">131</size>           </element-citation>
      </ref>
      <ref id="pone.0017293-Kamimura1">
        <label>34</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kamimura</surname><given-names>T</given-names></name><name name-style="western"><surname>Shimodaira</surname><given-names>H</given-names></name><name name-style="western"><surname>Imoto</surname><given-names>S</given-names></name><name name-style="western"><surname>Kim</surname><given-names>S</given-names></name><name name-style="western"><surname>Tashiro</surname><given-names>K</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>Multiscale bootstrap analysis of gene networks based on Bayesian networks and nonparametric regression.</article-title>             <source>Genome Inform</source>             <volume>14</volume>             <fpage>350</fpage>             <lpage>351</lpage>          </element-citation>
      </ref>
      <ref id="pone.0017293-Kingsford1">
        <label>35</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kingsford</surname><given-names>C</given-names></name><name name-style="western"><surname>Nagarajan</surname></name><name name-style="western"><surname>Salzberg</surname><given-names>S</given-names></name></person-group>             <year>2009</year>             <article-title>2009 swine-origin influenza A (H1N1) resembles previous influenza isolates.</article-title>             <source>PLoS ONE</source>             <volume>4</volume>             <issue>7</issue>             <fpage>e6402</fpage>             <comment>doi:<ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.1371/journal.pone.0006402" xlink:type="simple">10.1371/journal.pone.0006402</ext-link></comment>          </element-citation>
      </ref>
    </ref-list>
    
  </back>
</article>