<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
  <front>
    <journal-meta><journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="pmc">plosone</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS ONE</journal-title></journal-title-group><issn pub-type="epub">1932-6203</issn><publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher></journal-meta>
    <article-meta><article-id pub-id-type="publisher-id">PONE-D-11-03920</article-id><article-id pub-id-type="doi">10.1371/journal.pone.0020592</article-id><article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Biochemistry</subject>
            <subj-group>
              <subject>Proteins</subject>
              <subj-group>
                <subject>DNA-binding proteins</subject>
                <subject>Protein classes</subject>
                <subject>Protein structure</subject>
                <subject>Proteome</subject>
                <subject>Structural proteins</subject>
              </subj-group>
            </subj-group>
            <subj-group>
              <subject>Drug discovery</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Biophysics</subject>
            <subj-group>
              <subject>Protein folding</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Biotechnology</subject>
            <subj-group>
              <subject>Drug discovery</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Computational biology</subject>
            <subj-group>
              <subject>Sequence analysis</subject>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Chemistry</subject>
          <subj-group>
            <subject>Chemical biology</subject>
          </subj-group>
          <subj-group>
            <subject>Computational chemistry</subject>
            <subj-group>
              <subject>Molecular mechanics</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Medicinal chemistry</subject>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Computer science</subject>
          <subj-group>
            <subject>Computer modeling</subject>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Chemistry</subject>
          <subject>Biotechnology</subject>
          <subject>Computational Biology</subject>
          <subject>Biophysics</subject>
          <subject>Computer Science</subject>
          <subject>Biochemistry</subject>
        </subj-group>
      </article-categories><title-group><article-title>A Multi-Label Classifier for Predicting the Subcellular Localization of Gram-Negative Bacterial Proteins with Both Single and Multiple Sites</article-title><alt-title alt-title-type="running-head">Predicting Protein Subcellular Localization</alt-title></title-group><contrib-group>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Xiao</surname>
            <given-names>Xuan</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Wu</surname>
            <given-names>Zhi-Cheng</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Chou</surname>
            <given-names>Kuo-Chen</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
      </contrib-group><aff id="aff1"><label>1</label><addr-line>Computer Department, Jing-De-Zhen Ceramic Institute, Jing-De-Zhen, China</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>Gordon Life Science Institute, San Diego, California, United States of America</addr-line>       </aff><contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Fraternali</surname>
            <given-names>Franca</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group><aff id="edit1">King's College London, United Kingdom</aff><author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">xiaoxuan0326@yahoo.com.cn</email></corresp>
        <fn fn-type="con">
          <p>Conceived and designed the experiments: ZCW XX KCC. Performed the experiments: ZCW XX. Analyzed the data: ZCW XX KCC. Contributed reagents/materials/analysis tools: ZCW XX. Wrote the paper: XX KCC.</p>
        </fn>
      <fn fn-type="conflict">
        <p>The authors have declared that no competing interests exist.</p>
      </fn></author-notes><pub-date pub-type="collection">
        <year>2011</year>
      </pub-date><pub-date pub-type="epub">
        <day>17</day>
        <month>6</month>
        <year>2011</year>
      </pub-date><volume>6</volume><issue>6</issue><elocation-id>e20592</elocation-id><history>
        <date date-type="received">
          <day>26</day>
          <month>2</month>
          <year>2011</year>
        </date>
        <date date-type="accepted">
          <day>4</day>
          <month>5</month>
          <year>2011</year>
        </date>
      </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2011</copyright-year><copyright-holder>Xiao et al</copyright-holder><license><license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions><abstract>
        <p>Prediction of protein subcellular localization is a challenging problem, particularly when the system concerned contains both singleplex and multiplex proteins. In this paper, by introducing the “multi-label scale” and hybridizing the information of gene ontology with the sequential evolution information, a novel predictor called <bold>iLoc-Gneg</bold> is developed for predicting the subcellular localization of Gram-positive bacterial proteins with both single-location and multiple-location sites. For facilitating comparison, the same stringent benchmark dataset used to estimate the accuracy of <bold>Gneg-mPLoc</bold> was adopted to demonstrate the power of <bold>iLoc-Gneg</bold>. The dataset contains 1,392 Gram-negative bacterial proteins classified into the following eight locations: (1) cytoplasm, (2) extracellular, (3) fimbrium, (4) flagellum, (5) inner membrane, (6) nucleoid, (7) outer membrane, and (8) periplasm. Of the 1,392 proteins, 1,328 are each with only one subcellular location and the other 64 are each with two subcellular locations, but none of the proteins included has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e001" xlink:type="simple"/></inline-formula> pairwise sequence identity to any other in a same subset (subcellular location). It was observed that the overall success rate by jackknife test on such a stringent benchmark dataset by <bold>iLoc-Gneg</bold> was over 91%, which is about 6% higher than that by <bold>Gneg-mPLoc</bold>. As a user-friendly web-server, <bold>iLoc-Gneg</bold> is freely accessible to the public at <ext-link ext-link-type="uri" xlink:href="http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg" xlink:type="simple">http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg</ext-link>. Meanwhile, a step-by-step guide is provided on how to use the web-server to get the desired results. Furthermore, for the user's convenience, the <bold>iLoc-Gneg</bold> web-server also has the function to accept the batch job submission, which is not available in the existing version of <bold>Gneg-mPLoc</bold> web-server. It is anticipated that <bold>iLoc-Gneg</bold> may become a useful high throughput tool for Molecular Cell Biology, Proteomics, System Biology, and Drug Development.</p>
      </abstract><funding-group><funding-statement>This work was supported by grants from the National Natural Science Foundation of China (No. 60961003), the Key Project of Chinese Ministry of Education (No. 210116), the Province National Natural Science Foundation of JiangXi (2009GZS0064 and 2010GZS0122), the Department of Education of Jiang-Xi Province (No. GJJ09271), and the plan for training youth scientists (stars of Jing-Gang) of Jiangxi Province. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts>
        <page-count count="10"/>
      </counts></article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>Bacteria can be divided into two groups: Gram-positive and Gram-negative. Gram-positive bacteria are those that are stained dark blue or violet by Gram staining; while Gram-negative bacteria cannot retain the stain, instead taking up the counter-stain and appearing red or pink.</p>
      <p>It has special meaning for both basic research and drug design to study bacteria because (1) they are the workhorses for the fields of molecular biology, biochemistry, and genetics due to their ability to quickly grow and being relatively easier to be manipulated, and (2) they are both harmful and useful. With the explosion of protein sequences generated in the post-genomic era, we are challenged to develop computational methods for timely and accurately identifying the subcellular locations of newly discovered bacterial proteins based on their sequence information alone because this kind of knowledge will be very useful for selecting proper bacterial proteins for a special target, or screening and prioritizing candidates in drug design.</p>
      <p>Actually, numerous predictors were developed for identifying subcellular localization of proteins in various organisms (see <xref ref-type="bibr" rid="pone.0020592-Nakai1">[1]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref> as well as the long list of references cited in the two review papers). However, those that are specialized for dealing with Gram-negative proteins are only a few. They are called “<bold>PSORT</bold>” <xref ref-type="bibr" rid="pone.0020592-Nakai1">[1]</xref>, <xref ref-type="bibr" rid="pone.0020592-Nakai2">[3]</xref>, <xref ref-type="bibr" rid="pone.0020592-Nakai3">[4]</xref>, “<bold>PSORT-B</bold>” <xref ref-type="bibr" rid="pone.0020592-Gardy1">[5]</xref>, and <bold>PSORTb v.2.0</bold> <xref ref-type="bibr" rid="pone.0020592-Gardy2">[6]</xref>. All these methods have played important roles in stimulating the development of this area. To improve the prediction coverage scope and the quality of benchmark datasets, the predictor called <bold>Gneg-PLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Chou2">[7]</xref> was developed. Compared with the previous methods, <bold>Gneg-PLoc</bold> extended the coverage scope from five to eight subcellular location sites. Also, the benchmark datasets used to train and test the predictor have been significantly refined. For instance, the benchmark datasets used in <bold>PSORT-B</bold> <xref ref-type="bibr" rid="pone.0020592-Gardy1">[5]</xref> contain many proteins with pairwise sequence identity higher than 90%, while in the benchmark datasets of <bold>Gneg-PLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Chou2">[7]</xref> none of the proteins included has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e002" xlink:type="simple"/></inline-formula> pairwise sequence identity to any other in a same subcellular location; i.e., the latter is much more stringent and rigorous than the former in excluding the homology bias and redundancy. Also, <bold>Gneg-PLoc</bold> was able to yield higher success rates.</p>
      <p>However, all the aforementioned predictors cannot be used to deal with multiplex proteins that may simultaneously exist at, or move between, two or more different subcellular locations. Proteins with multiple locations or dynamic feature of this kind are particularly interesting because they may have some very special biological functions intriguing to investigators in both basic research and drug discovery <xref ref-type="bibr" rid="pone.0020592-Smith1">[8]</xref>, <xref ref-type="bibr" rid="pone.0020592-Glory1">[9]</xref>. Particularly, as pointed out by Millar et al. <xref ref-type="bibr" rid="pone.0020592-Millar1">[10]</xref>, recent evidences have indicated that an increasing number of proteins have multiple locations in the cell.</p>
      <p>To make <bold>Gneg-PLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Chou2">[7]</xref> be able to deal with multiplex Gram-negative proteins as well, a predictor called <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> was developed recently, where the character “m” in front of “PLoc” stands for “multiple”, meaning that it can be also used to deal with Gram-negative bacterial proteins with multiple locations.</p>
      <p>However, <bold>Gneg-mPLoc</bold> has the following shortcomings. <bold>(1)</bold> In predicting the number of subcellular location sites for a query Gram-negative protein, an optimal threshold factor <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e003" xlink:type="simple"/></inline-formula> (see Eq.48 of <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>) was adopted without providing its statistical implication and detailed learning process. It would be more instructive if we could find a more intuitive approach to determine this with a more natural manner. <bold>(2)</bold> In formulating the protein samples, only the integer numbers 0 and 1 were used to reflect the GO (gene ontology) information <xref ref-type="bibr" rid="pone.0020592-Ashburner1">[12]</xref>, <xref ref-type="bibr" rid="pone.0020592-Camon1">[13]</xref>. Such an over-simplified formulation might cause some useful information lost so as to limit the prediction quality. <bold>(3)</bold> Although a web-server for <bold>Gneg-mPLoc</bold> has been established at <ext-link ext-link-type="uri" xlink:href="http://www.csbio.sjtu.edu.cn/bioinf/Gneg-multi/" xlink:type="simple">http://www.csbio.sjtu.edu.cn/bioinf/Gneg-multi/</ext-link>, only one query protein sequence at a time is allowed when using the web-server to conduct prediction. For the convenience of users in handling many query Gram-negative protein sequences, such a rigid limit should be improved.</p>
      <p>The present study was dedicated to develop a new and more powerful predictor, called <bold>iLoc-Gneg</bold>, for predicting Gram-negative bacterial protein subcellular localization by addressing the above three problems.</p>
      <p>To establish a really useful statistical predictor for protein system, we usually need to consider the following procedures <xref ref-type="bibr" rid="pone.0020592-Chou3">[14]</xref>: (1) select or construct a valid benchmark dataset to train and test the predictor; (2) formulate the protein samples with an effective mathematical expression that can truly reflect their intrinsic correlation with the attribute to be predicted; (3) introduce or develop a powerful algorithm (or engine) to operate the prediction; (4) properly perform cross-validation tests to objectively evaluate the anticipated accuracy of the predictor; (5) establish a user-friendly web-server <xref ref-type="bibr" rid="pone.0020592-Chou4">[15]</xref> for the predictor that is accessible to the public. Below, let us describe how to realize these steps one by one.</p>
    </sec>
    <sec id="s2" sec-type="materials|methods">
      <title>Materials and Methods</title>
      <p>Here, we choose to use the same dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e004" xlink:type="simple"/></inline-formula> in establishing <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> as the benchmark dataset for the current study. The reasons doing so are as follows. <bold>(1)</bold> The dataset was constructed specialized for Gram-negative bacterial proteins and it can cover 8 subcellular location sites; compared with the other datasets such as the one in <bold>PSORTb v.2.0</bold> <xref ref-type="bibr" rid="pone.0020592-Gardy2">[6]</xref> that only covered 5 subcellular locations, the coverage scope of the dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e005" xlink:type="simple"/></inline-formula> from <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> is much wider. <bold>(2)</bold> None of proteins included in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e006" xlink:type="simple"/></inline-formula> has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e007" xlink:type="simple"/></inline-formula> pairwise sequence identity to any other in a same subcellular location; compared with most of the other benchmark datasets in this area, the dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e008" xlink:type="simple"/></inline-formula> is much more rigorous in excluding homology bias and redundancy. <bold>(3)</bold> It contains both singleplex and multiplex proteins and hence can be used to train and test a predictor developed aimed at being able to deal with proteins with both single and multiple location sites. <bold>(4)</bold> Using the dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e009" xlink:type="simple"/></inline-formula> will also make it easier to compare the new predictor with the existing one because the tested results by <bold>Gneg-mPLoc</bold> on <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e010" xlink:type="simple"/></inline-formula> have been well documented and reported <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</p>
      <p>The dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e011" xlink:type="simple"/></inline-formula> contains 1,392 Gram-negative bacterial protein sequences, of which 1,328 belong to one subcellular location, 64 to two locations, and none to three or more locations. The dataset covers 8 subcellular locations (<xref ref-type="fig" rid="pone-0020592-g001"><bold>Fig. 1</bold></xref>), as can be formulated by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e012" xlink:type="simple"/><label>(1)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e013" xlink:type="simple"/></inline-formula> represents the subset for the subcellular location of cell inner membrane, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e014" xlink:type="simple"/></inline-formula> for cell outer membrane, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e015" xlink:type="simple"/></inline-formula> for cytoplasm, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e016" xlink:type="simple"/></inline-formula> for extracellular, and so forth (<xref ref-type="table" rid="pone-0020592-t001"><bold>Table 1</bold></xref>); while <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e017" xlink:type="simple"/></inline-formula> represents the symbol for “union” in the set theory. To avoid homology bias and redundancy, none of the proteins in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e018" xlink:type="simple"/></inline-formula> has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e019" xlink:type="simple"/></inline-formula> pairwise sequence identity to any other in a same subset. For convenience, hereafter let us just use the subscripts of <bold>Eq.1</bold> as the codes of the 8 location sites; i.e., “1” for “cell membrane”, “2” for “cell wall”, “3” for “chloroplast”, and so forth (<xref ref-type="table" rid="pone-0020592-t002"><bold>Table 2</bold></xref>).</p>
      <fig id="pone-0020592-g001" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0020592.g001</object-id>
        <label>Figure 1</label>
        <caption>
          <title>Illustration to show the 8 subcellular locations of Gram-negative bacterial proteins.</title>
          <p>The 8 locations are: (1) cytoplasm, (2) extracellular, (3) fimbrium, (4) flagellum, (5) inner membrane, (6) nucleoid, (7) outer membrane, and (8) periplasm. Note that in prokaryotic life forms, the nucleoid region is the part of the cell that contains the DNA molecule; unlike the true nucleus of eukaryotes, it is not delimited by a membrane.</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.g001" xlink:type="simple"/>
      </fig>
      <table-wrap id="pone-0020592-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0020592.t001</object-id><label>Table 1</label><caption>
          <title>Breakdown of the Gram-negative bacterial protein benchmark dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e020" xlink:type="simple"/></inline-formula> taken from <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0020592-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.t001" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Subset</td>
              <td align="left" colspan="1" rowspan="1">Subcellular location</td>
              <td align="left" colspan="1" rowspan="1">Number of proteins</td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e021" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Cell inner membrane</td>
              <td align="left" colspan="1" rowspan="1">557</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e022" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Cell outer membrane</td>
              <td align="left" colspan="1" rowspan="1">124</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e023" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Cytoplasm</td>
              <td align="left" colspan="1" rowspan="1">410</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e024" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Extracellular</td>
              <td align="left" colspan="1" rowspan="1">133</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e025" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Fimbrium</td>
              <td align="left" colspan="1" rowspan="1">32</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e026" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Flagellum</td>
              <td align="left" colspan="1" rowspan="1">12</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e027" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Nucleoid</td>
              <td align="left" colspan="1" rowspan="1">8</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">
                <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e028" xlink:type="simple"/></inline-formula>
              </td>
              <td align="left" colspan="1" rowspan="1">Periplasm</td>
              <td align="left" colspan="1" rowspan="1">180</td>
            </tr>
            <tr>
              <td align="left" colspan="2" rowspan="1">Total number of locative proteins <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e029" xlink:type="simple"/></inline-formula></td>
              <td align="left" colspan="1" rowspan="1">1,456<xref ref-type="table-fn" rid="nt102">a</xref></td>
            </tr>
            <tr>
              <td align="left" colspan="2" rowspan="1">Total number of different proteins <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e030" xlink:type="simple"/></inline-formula></td>
              <td align="left" colspan="1" rowspan="1">1,392<xref ref-type="table-fn" rid="nt103">b</xref></td>
            </tr>
          </tbody>
        </table></alternatives><table-wrap-foot>
          <fn id="nt101">
            <label/>
            <p>None of proteins included here has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e031" xlink:type="simple"/></inline-formula> sequence identity to any other in a same subcellular location.</p>
          </fn>
          <fn id="nt102">
            <label>a</label>
            <p>See Eqs.36–38 of <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref> for the definition about the number of locative proteins, and its relation with the number of different proteins.</p>
          </fn>
          <fn id="nt103">
            <label>b</label>
            <p>Of the 1,392 different proteins, 1,328 have one subcellular location, 64 have two locations, and none have three or more locations.</p>
          </fn>
        </table-wrap-foot></table-wrap>
      <table-wrap id="pone-0020592-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0020592.t002</object-id><label>Table 2</label><caption>
          <title>A comparison of the jackknife success rates by <bold>Gnec-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> and the current <bold>iLoc-Gneg</bold> on the benchmark dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e032" xlink:type="simple"/></inline-formula> (cf. <underline><xref ref-type="supplementary-material" rid="pone.0020592.s001">Supporting Information S1</xref></underline>) that covers 8 location sites of Gram-negative bacterial proteins in which none of the proteins included has <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e033" xlink:type="simple"/></inline-formula>25% pairwise sequence identity to any other in a same location.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0020592-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.t002" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Code</td>
              <td align="left" colspan="1" rowspan="1">Subcellular location</td>
              <td align="left" colspan="2" rowspan="1">Success rate by jackknife test</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">Gneg-mPLoc<xref ref-type="table-fn" rid="nt104">a</xref></td>
              <td align="left" colspan="1" rowspan="1">iLoc-Gneg<xref ref-type="table-fn" rid="nt105">b</xref></td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="1" rowspan="1">1</td>
              <td align="left" colspan="1" rowspan="1">Cell inner membrane</td>
              <td align="left" colspan="1" rowspan="1">525/557 = 94.3%</td>
              <td align="left" colspan="1" rowspan="1">539/557 = 96.8%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">2</td>
              <td align="left" colspan="1" rowspan="1">Cell outer membrane</td>
              <td align="left" colspan="1" rowspan="1">105/124 = 84.7%</td>
              <td align="left" colspan="1" rowspan="1">103/124 = 83.1%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">3</td>
              <td align="left" colspan="1" rowspan="1">Cytoplasm</td>
              <td align="left" colspan="1" rowspan="1">357/410 = 87.1%</td>
              <td align="left" colspan="1" rowspan="1">367/410 = 89.5%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">4</td>
              <td align="left" colspan="1" rowspan="1">Extracellular</td>
              <td align="left" colspan="1" rowspan="1">79/133 = 59.4%</td>
              <td align="left" colspan="1" rowspan="1">115/133 = 86.5%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">5</td>
              <td align="left" colspan="1" rowspan="1">Fimbrium</td>
              <td align="left" colspan="1" rowspan="1">28/32 = 87.5%</td>
              <td align="left" colspan="1" rowspan="1">30/32 = 93.8%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">6</td>
              <td align="left" colspan="1" rowspan="1">Flagellum</td>
              <td align="left" colspan="1" rowspan="1">0/12 = 0.0%</td>
              <td align="left" colspan="1" rowspan="1">12/12 = 100%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">7</td>
              <td align="left" colspan="1" rowspan="1">Nucleoid</td>
              <td align="left" colspan="1" rowspan="1">0/8 = 0.0%</td>
              <td align="left" colspan="1" rowspan="1">4/8 = 50%</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">8</td>
              <td align="left" colspan="1" rowspan="1">Periplasm</td>
              <td align="left" colspan="1" rowspan="1">154/180 = 85.6%</td>
              <td align="left" colspan="1" rowspan="1">161/180 = 89.4%</td>
            </tr>
            <tr>
              <td align="left" colspan="2" rowspan="1">Overall<xref ref-type="table-fn" rid="nt106">c</xref></td>
              <td align="left" colspan="1" rowspan="1">1248/1456 = <bold>85.7%</bold></td>
              <td align="left" colspan="1" rowspan="1">1331/1456 = <bold>91.4%</bold></td>
            </tr>
          </tbody>
        </table></alternatives><table-wrap-foot>
          <fn id="nt104">
            <label>a</label>
            <p>The predictor from <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</p>
          </fn>
          <fn id="nt105">
            <label>b</label>
            <p>The predictor proposed in this paper.</p>
          </fn>
          <fn id="nt106">
            <label>c</label>
            <p>Note that instead of 1,392 (the number of total different Gram-positive bacterial proteins), here we use 1,456 (the number of total different locative proteins) for the denominator. This is because some of the Gram-negative bacterial proteins in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e034" xlink:type="simple"/></inline-formula> may have more than one location site. See footnotes a and b of <xref ref-type="table" rid="pone-0020592-t001">Table 1</xref> for further explanation.</p>
          </fn>
        </table-wrap-foot></table-wrap>
      <p>For readers' convenience, the corresponding accession numbers and protein sequences in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e035" xlink:type="simple"/></inline-formula> are given in <underline><xref ref-type="supplementary-material" rid="pone.0020592.s001">Supporting Information S1</xref></underline>.</p>
      <p>Note that because some proteins may occur in two or more locations, the 1,392 Gram-negative proteins actually correspond to 1,456 locative proteins. The concept of “locative proteins” was introduced for studying proteins with multiple subcellular location sites, as elaborated in <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>.</p>
      <p>To develop a powerful method for statistically predicting protein subcellular localization according to the sequence information, one of the most important things is to formulate the protein sequences with an effective mathematical expression that can truly reflect the intrinsic correlation with their subcellular localization <xref ref-type="bibr" rid="pone.0020592-Chou3">[14]</xref>. However, it is by no means an easy job to realize this because this kind of correlation is usually deeply “buried” or hidden in piles of complicated sequences.</p>
      <p>The most straightforward method to formulate the sample of a query protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e036" xlink:type="simple"/></inline-formula> was just using its entire amino acid sequence, as can be generally written by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e037" xlink:type="simple"/><label>(2)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e038" xlink:type="simple"/></inline-formula> represents the 1<sup>st</sup> residue of the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e039" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e040" xlink:type="simple"/></inline-formula> the 2<sup>nd</sup> residue, …, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e041" xlink:type="simple"/></inline-formula> the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e042" xlink:type="simple"/></inline-formula> residue, and they each belong to one of the 20 native amino acids. In order to identify its subcellular location(s), the sequence-similarity-search-based tools, such as BLAST <xref ref-type="bibr" rid="pone.0020592-Altschul1">[16]</xref>, <xref ref-type="bibr" rid="pone.0020592-Wootton1">[17]</xref>, was utilized to search protein database for those proteins that have high sequence similarity to the query protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e043" xlink:type="simple"/></inline-formula>. Subsequently, the subcellular location annotations of the proteins thus found were used to deduce the subcellular location(s) for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e044" xlink:type="simple"/></inline-formula>. Unfortunately, although it was quite intuitive and able to contain the entire information of a protein sequence, this kind of straightforward sequential model failed to work when the query protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e045" xlink:type="simple"/></inline-formula> did not have significant sequence similarity to any location-known proteins.</p>
      <p>Thus, various non-sequential or discrete models to formulate protein samples were proposed in hopes to establish some sort of correlation or cluster manner by which the prediction quality could be improved.</p>
      <p>Among the discrete models for a protein sample, the simplest one is its amino acid (AA) composition or AAC <xref ref-type="bibr" rid="pone.0020592-Chou5">[18]</xref>. According to the AAC-discrete model, the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e046" xlink:type="simple"/></inline-formula> of <bold>Eq.2</bold> can be formulated by <xref ref-type="bibr" rid="pone.0020592-Nakashima1">[19]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou6">[20]</xref><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e047" xlink:type="simple"/><label>(3)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e048" xlink:type="simple"/></inline-formula> are the normalized occurrence frequencies of the 20 native amino acids in protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e049" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e050" xlink:type="simple"/></inline-formula> the transposing operator. Many methods for predicting protein subcellular localization were based on the AAC-discrete model (see, e.g., <xref ref-type="bibr" rid="pone.0020592-Nakashima1">[19]</xref>, <xref ref-type="bibr" rid="pone.0020592-Cedano1">[21]</xref>, <xref ref-type="bibr" rid="pone.0020592-Reinhardt1">[22]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou7">[23]</xref>, <xref ref-type="bibr" rid="pone.0020592-Zhou1">[24]</xref>). However, as we can see from <bold>Eq.3</bold>, if using the ACC model to represent the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e051" xlink:type="simple"/></inline-formula>, all its sequence-order effects would be lost, and hence the prediction quality might be limited.</p>
      <p>To avoid completely lose the sequence-order information, the pseudo amino acid composition (PseAAC) was proposed to represent the sample of a protein, as formulated by <xref ref-type="bibr" rid="pone.0020592-Chou8">[25]</xref><disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e052" xlink:type="simple"/><label>(4)</label></disp-formula>where the first 20 elements are associated with the 20 elements in <bold>Eq.3</bold> or the 20 amino acid components of the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e053" xlink:type="simple"/></inline-formula>, while the additional <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e054" xlink:type="simple"/></inline-formula> factors are used to incorporate some sequence-order information via a series of rank-different correlation factors along a protein chain. For a brief introduction about PseAAC, please see a Wikipedia article at <ext-link ext-link-type="uri" xlink:href="http://en.wikipedia.org/wiki/Pseudo_amino_acid_composition" xlink:type="simple">http://en.wikipedia.org/wiki/Pseudo_amino_acid_composition</ext-link>.</p>
      <p>According to <xref ref-type="bibr" rid="pone.0020592-Chou3">[14]</xref>, the PseAAC for a protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e055" xlink:type="simple"/></inline-formula> can be generally formulated as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e056" xlink:type="simple"/><label>(5)</label></disp-formula>where the subscript <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e057" xlink:type="simple"/></inline-formula> is an integer, and its value as well as the components <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e058" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e059" xlink:type="simple"/></inline-formula>, … will depend on how to extract the desired information from the amino acid sequence of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e060" xlink:type="simple"/></inline-formula> (cf. <bold>Eq.2</bold>). As a general form, <bold>Eq.5</bold> can cover various different modes of PseAAC. For example, when its elements are given by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e061" xlink:type="simple"/><label>(6)</label></disp-formula>we immediately obtain the formulation of PseAAC as originally introduced in <xref ref-type="bibr" rid="pone.0020592-Chou8">[25]</xref>, where the meanings for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e062" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e063" xlink:type="simple"/></inline-formula>, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e064" xlink:type="simple"/></inline-formula> were clearly elaborated and hence there is no need to repeat here.</p>
      <p>Below, let us use the general form of PseAAC (<bold>Eq.5</bold>) to find the formulations to reflect the core and essential features of protein samples that are closely correlated with their subcellular localization.</p>
      <sec id="s2a">
        <title>1. GO (Gene Ontology) Formulation</title>
        <p>GO database <xref ref-type="bibr" rid="pone.0020592-Ashburner1">[12]</xref> was established according to the molecular function, biological process, and cellular component. Accordingly, protein samples defined in a GO database space would be clustered in a way better reflecting their subcellular locations <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou9">[26]</xref>. However, in order to incorporate more information, instead of only using 0 and 1 elements as done in <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, here let us use a different approach as described below.</p>
        <sec id="s2a1">
          <title>Step 1</title>
          <p>Compression and reorganization of the existing GO numbers. The GO database (version 74.0 released 30 July 2009) contains many GO numbers. However, these numbers do not increase successively and orderly. For easier handling, some reorganization and compression procedure was taken to renumber them. For example, after such a procedure, the original GO numbers GO:0000001, GO:0000002, GO:0000003, GO:0000009, GO:00000011, GO:0000012, GO:0000015, …, GO:0090204 would become GO_compress: 00001, GO_compress: 00002, GO_compress: 00003, GO_compress: 00004, GO_compress: 00005, GO_compress: 00006, GO_compress: 00007, ……, GO_compress: 11118, respectively. The GO database obtained thru such a treatment is called GO_compress database, which contains 11,118 numbers increasing successively from 1 to the last one.</p>
        </sec>
        <sec id="s2a2">
          <title>Step 2</title>
          <p>Using <bold>Eq.5</bold> with <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e065" xlink:type="simple"/></inline-formula>, the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e066" xlink:type="simple"/></inline-formula> can be formulated as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e067" xlink:type="simple"/><label>(7)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e068" xlink:type="simple"/></inline-formula> <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e069" xlink:type="simple"/></inline-formula> are defined via the following steps.</p>
        </sec>
        <sec id="s2a3">
          <title>Step 3</title>
          <p>Use BLAST <xref ref-type="bibr" rid="pone.0020592-Schaffer1">[27]</xref> to search the homologous proteins of the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e070" xlink:type="simple"/></inline-formula> from the Swiss-Prot database (version 55.3), with the expect value <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e071" xlink:type="simple"/></inline-formula> for the BLAST parameter.</p>
        </sec>
        <sec id="s2a4">
          <title>Step 4</title>
          <p>Those proteins which have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e072" xlink:type="simple"/></inline-formula> pairwise sequence identity with the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e073" xlink:type="simple"/></inline-formula> are collected into a set, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e074" xlink:type="simple"/></inline-formula>, called the “homology set” of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e075" xlink:type="simple"/></inline-formula>. All the elements in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e076" xlink:type="simple"/></inline-formula> can be deemed as the “representative proteins” of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e077" xlink:type="simple"/></inline-formula>, sharing some similar attributes such as structural conformations and biological functions <xref ref-type="bibr" rid="pone.0020592-Loewenstein1">[28]</xref>, <xref ref-type="bibr" rid="pone.0020592-Gerstein1">[29]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou10">[30]</xref>. Because they were retrieved from the Swiss-Prot database, these representative proteins must each have their own accession numbers.</p>
        </sec>
        <sec id="s2a5">
          <title>Step 5</title>
          <p>Search each of these accession numbers collected in Step 4 against the GO database at <ext-link ext-link-type="uri" xlink:href="http://www.ebi.ac.uk/GOA/" xlink:type="simple">http://www.ebi.ac.uk/GOA/</ext-link> to find the corresponding GO numbers <xref ref-type="bibr" rid="pone.0020592-Camon2">[31]</xref>.</p>
        </sec>
        <sec id="s2a6">
          <title>Step 6</title>
          <p>Based on the results obtained in Step 5, the elements in <bold>Eq.7</bold> can be written as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e078" xlink:type="simple"/><label>(8)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e079" xlink:type="simple"/></inline-formula> is the number of representative proteins in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e080" xlink:type="simple"/></inline-formula>, and<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e081" xlink:type="simple"/><label>(9)</label></disp-formula></p>
          <p>As we can see from <bold>Eq.7</bold>, the GO formulation derived from the above steps consists of 11,118 real numbers rather than only the elements 0 and 1 as in the GO formulation adopted in <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</p>
          <p>Note that the GO formulation of <bold>Eq.6</bold> may become a naught vector or meaningless under any of the following situations: <bold>(1)</bold> the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e082" xlink:type="simple"/></inline-formula> does not have significant homology to any protein in the Swiss-Prot database, i.e., <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e083" xlink:type="simple"/></inline-formula> meaning the homology set <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e084" xlink:type="simple"/></inline-formula> is an empty one; <bold>(2)</bold> its representative proteins do not contain any useful GO information for statistical prediction based on a given training dataset.</p>
          <p>Under such a circumstance, let us consider using the sequential evolution formulation to represent the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e085" xlink:type="simple"/></inline-formula>, as described below.</p>
        </sec>
      </sec>
      <sec id="s2b">
        <title>2. SeqEvo (Sequential Evolution) Formulation</title>
        <p>Biology is a natural science with historic dimension. All biological species have developed continuously starting out from a very limited number of ancestral species. It is true for protein sequence as well <xref ref-type="bibr" rid="pone.0020592-Chou10">[30]</xref>. Their evolution involves changes of single residues, insertions and deletions of several residues <xref ref-type="bibr" rid="pone.0020592-Chou11">[32]</xref>, gene doubling, and gene fusion. With these changes accumulated for a long period of time, many similarities between initial and resultant amino acid sequences are gradually eliminated, but the corresponding proteins may still share many common attributes, such as having basically the same biological function and residing in a same subcellular location.</p>
        <p>To incorporate the sequential evolution information into the PseAAC of <bold>Eq.4</bold>, here let us use the information of the PSSM (Position-Specific Scoring Matrix) <xref ref-type="bibr" rid="pone.0020592-Schaffer1">[27]</xref>, as described below.</p>
        <sec id="s2b1">
          <title>Step 1</title>
          <p>According to <xref ref-type="bibr" rid="pone.0020592-Schaffer1">[27]</xref>, the sequential evolution information of protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e086" xlink:type="simple"/></inline-formula> can be expressed by a <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e087" xlink:type="simple"/></inline-formula> matrix as given by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e088" xlink:type="simple"/><label>(10)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e089" xlink:type="simple"/></inline-formula> is the length of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e090" xlink:type="simple"/></inline-formula> (counted in the total number of its constituent amino acids as shown in <bold>Eq.1</bold>), <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e091" xlink:type="simple"/></inline-formula> represents the score of the amino acid residue in the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e092" xlink:type="simple"/></inline-formula> position of the protein sequence being changed to amino acid type <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e093" xlink:type="simple"/></inline-formula> during the evolutionary process. Here, the numerical codes 1, 2, …, 20 are used to denote the 20 native amino acid types according to the alphabetical order of their single character codes. The <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e094" xlink:type="simple"/></inline-formula> scores in <bold>Eq.10</bold> were generated by using PSI-BLAST <xref ref-type="bibr" rid="pone.0020592-Schaffer1">[27]</xref> to search the UniProtKB/Swiss-Prot database (Release 2010_04 of 23-Mar-2010) through three iterations with 0.001 as the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e095" xlink:type="simple"/></inline-formula>-value cutoff for multiple sequence alignment against the sequence of the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e096" xlink:type="simple"/></inline-formula>. However, according to the formulation of <bold>Eq.10</bold>, proteins with different lengths will correspond to column-different matrices causing difficulty for developing a predictor able to uniformly cover proteins of any length. To make the descriptor become a size-uniform matrix, let us consider the following steps.</p>
        </sec>
        <sec id="s2b2">
          <title>Step 2</title>
          <p>Use the elements in<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e097" xlink:type="simple"/></inline-formula> of Eq.10 to define a new matrix <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e098" xlink:type="simple"/></inline-formula> as formulated by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e099" xlink:type="simple"/><label>(11)</label></disp-formula>with<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e100" xlink:type="simple"/><label>(12)</label></disp-formula>where<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e101" xlink:type="simple"/><label>(13)</label></disp-formula>is the mean for <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e102" xlink:type="simple"/></inline-formula> and<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e103" xlink:type="simple"/><label>(14)</label></disp-formula>is the corresponding standard deviation.</p>
        </sec>
        <sec id="s2b3">
          <title>Step 3</title>
          <p>Introduce a new matrix generated by multiplying <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e104" xlink:type="simple"/></inline-formula> with its own transpose matrix <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e105" xlink:type="simple"/></inline-formula>; i.e.,<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e106" xlink:type="simple"/><label>(15)</label></disp-formula>which contains <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e107" xlink:type="simple"/></inline-formula> elements. Since <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e108" xlink:type="simple"/></inline-formula> is a symmetric matrix, we only need the information of its 210 elements, of which 20 are the diagonal elements and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e109" xlink:type="simple"/></inline-formula> are the lower triangular elements, to formulate the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e110" xlink:type="simple"/></inline-formula>; i.e., the general PseAAC form of <bold>Eq.5</bold> can now be formulated as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e111" xlink:type="simple"/><label>(16)</label></disp-formula>where the components <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e112" xlink:type="simple"/></inline-formula> are respectively taken from the 210 diagonal and lower triangular elements of <bold>Eq.15</bold> by following a given order, say from left to right and from the 1<sup>st</sup> row to the last as illustrated by following equation<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e113" xlink:type="simple"/><label>(17)</label></disp-formula>where the numbers in parentheses indicate the order of elements taken from <bold>Eq.15</bold> for <bold>Eq.16</bold>.</p>
        </sec>
      </sec>
      <sec id="s2c">
        <title>3. The Self-consistency Formulation Principle</title>
        <p>Regardless of using which formulation to represent protein samples, the following self-consistency principle must be observed during the course of prediction: if the query protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e114" xlink:type="simple"/></inline-formula> was defined in the form of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e115" xlink:type="simple"/></inline-formula> (see <bold>Eq.7</bold>), then all the protein samples used to train the prediction engine should also be expressed in the GO formulation; if the query protein was defined in the form of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e116" xlink:type="simple"/></inline-formula> (see <bold>Eq.16</bold>), then all the training data should be expressed in the SeqEvo formulation as well.</p>
        <p>Below, let us consider the algorithm or operation engine for conducting the prediction.</p>
      </sec>
      <sec id="s2d">
        <title>4. Multi-Label KNN (K-Nearest Neighbor) Classifier</title>
        <p>In this study, let us introduce a novel classifier, called the multi-label KNN or abbreviated as ML-KNN classifier, to predict the subcellular localization for the systems that contain both single-location and multiple-location proteins.</p>
        <p>Suppose the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e117" xlink:type="simple"/></inline-formula> subset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e118" xlink:type="simple"/></inline-formula> of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e119" xlink:type="simple"/></inline-formula> (<bold>Eq.1</bold>) contains <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e120" xlink:type="simple"/></inline-formula> Gram-negative proteins, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e121" xlink:type="simple"/></inline-formula> is the<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e122" xlink:type="simple"/></inline-formula> one in that subset. Thus, we have<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e123" xlink:type="simple"/><label>(18)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e124" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e125" xlink:type="simple"/></inline-formula> have the same forms as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e126" xlink:type="simple"/></inline-formula>(<bold>Eq.7</bold>), and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e127" xlink:type="simple"/></inline-formula>(<bold>Eq.16</bold>), respectively; the only difference is that the corresponding constituent elements are derived from the amino acid sequence of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e128" xlink:type="simple"/></inline-formula> instead of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e129" xlink:type="simple"/></inline-formula>.</p>
        <p>In sequence analysis, there are many different scales to define the distance between two proteins, such as Euclidean distance, Hamming distance <xref ref-type="bibr" rid="pone.0020592-Mardia1">[33]</xref>, and Mahalanobis distance <xref ref-type="bibr" rid="pone.0020592-Chou5">[18]</xref>, <xref ref-type="bibr" rid="pone.0020592-Mahalanobis1">[34]</xref>, <xref ref-type="bibr" rid="pone.0020592-Pillai1">[35]</xref>. In <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, the distance between <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e130" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e131" xlink:type="simple"/></inline-formula> was defined by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e132" xlink:type="simple"/></inline-formula>. However, we have observed that when the GO descriptor was formulated with real numbers, better outcomes would be resulted by using the Euclidean metric; i.e., the distance between <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e133" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e134" xlink:type="simple"/></inline-formula> should be defined here by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e135" xlink:type="simple"/><label>(19)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e136" xlink:type="simple"/></inline-formula> represents the module of the vector difference between <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e137" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e138" xlink:type="simple"/></inline-formula> in the Euclidean space. According to <bold>Eq.19</bold>, when <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e139" xlink:type="simple"/></inline-formula> we have <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e140" xlink:type="simple"/></inline-formula>, indicating the distance between these two protein sequences is zero and hence they have perfect or 100% similarity.</p>
        <p>Suppose <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e141" xlink:type="simple"/></inline-formula> are the <italic>K</italic> nearest neighbor proteins to the protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e142" xlink:type="simple"/></inline-formula> that forms a set denoted by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e143" xlink:type="simple"/></inline-formula>, which is a subset of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e144" xlink:type="simple"/></inline-formula>; i.e.,<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e145" xlink:type="simple"/></inline-formula>. Based on the <italic>K</italic> nearest neighbor proteins in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e146" xlink:type="simple"/></inline-formula>, let us define an accumulation-layer (AL) scale, given by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e147" xlink:type="simple"/><label>(20)</label></disp-formula>where<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e148" xlink:type="simple"/><label>(21)</label></disp-formula>where<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e149" xlink:type="simple"/><label>(22)</label></disp-formula>and<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e150" xlink:type="simple"/><label>(23)</label></disp-formula>Note that <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e151" xlink:type="simple"/></inline-formula> because a protein may belong to one or more subcellular location sites in the current system.</p>
        <p>Now, for a query protein <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e152" xlink:type="simple"/></inline-formula>, its subcellular location(s) will be predicted according to the following steps.</p>
        <sec id="s2d1">
          <title>Step 1</title>
          <p>The number of how many different subcellular locations it belongs to will be determined by its nearest neighbor protein in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e153" xlink:type="simple"/></inline-formula>. For example, suppose <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e154" xlink:type="simple"/></inline-formula> is the nearest protein to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e155" xlink:type="simple"/></inline-formula> in <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e156" xlink:type="simple"/></inline-formula>. If <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e157" xlink:type="simple"/></inline-formula> has only one subcellular location, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e158" xlink:type="simple"/></inline-formula> will also have only one location; if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e159" xlink:type="simple"/></inline-formula> has two subcellular locations, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e160" xlink:type="simple"/></inline-formula> will also have two locations; and so forth. In general, if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e161" xlink:type="simple"/></inline-formula> belongs to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e162" xlink:type="simple"/></inline-formula> different location sites, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e163" xlink:type="simple"/></inline-formula> will be predicted to have the same number, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e164" xlink:type="simple"/></inline-formula>, of subcellular locations as well, as can be formulated by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e165" xlink:type="simple"/><label>(24)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e166" xlink:type="simple"/></inline-formula> is an integer <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e167" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e168" xlink:type="simple"/></inline-formula> represents the number of different subcellular locations to which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e169" xlink:type="simple"/></inline-formula> belongs, and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e170" xlink:type="simple"/></inline-formula> the number of different subcellular locations to which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e171" xlink:type="simple"/></inline-formula> belongs.</p>
        </sec>
        <sec id="s2d2">
          <title>Step 2</title>
          <p>However, the concrete location site(s) to which <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e172" xlink:type="simple"/></inline-formula> belongs will not be determined by the location site(s) of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e173" xlink:type="simple"/></inline-formula>, but by the element(s) in <bold>Eq.20</bold> that has (have) the highest score(s), as can be expressed by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e174" xlink:type="simple"/></inline-formula>, the subscript(s) of <bold>Eq.1</bold>. For example, if <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e175" xlink:type="simple"/></inline-formula> is found belonging to only one location <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e176" xlink:type="simple"/></inline-formula> in Step 1, and the highest score in <bold>Eq.20</bold> is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e177" xlink:type="simple"/></inline-formula>, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e178" xlink:type="simple"/></inline-formula> will be predicted as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e179" xlink:type="simple"/></inline-formula> meaning that it belongs to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e180" xlink:type="simple"/></inline-formula> or resides at “cytoplasm” (cf. <xref ref-type="table" rid="pone-0020592-t001"><bold>Table 1</bold></xref>). If <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e181" xlink:type="simple"/></inline-formula> is found belonging to two locations <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e182" xlink:type="simple"/></inline-formula>, and the first two highest scores in <bold>Eq.20</bold> are <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e183" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e184" xlink:type="simple"/></inline-formula>, then <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e185" xlink:type="simple"/></inline-formula> will be predicted as <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e186" xlink:type="simple"/></inline-formula> meaning that it belongs to <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e187" xlink:type="simple"/></inline-formula> and <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e188" xlink:type="simple"/></inline-formula> or resides simultaneously at “cell inner membrane” and “periplasm”. And so forth. In other words, the concrete predicted subcellular location(s) can be formulated as<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e189" xlink:type="simple"/><label>(25)</label></disp-formula>where the operator “<inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e190" xlink:type="simple"/></inline-formula>” means identifying the <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e191" xlink:type="simple"/></inline-formula> highest scores for the elements in the brackets right after it, followed by taking their <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e192" xlink:type="simple"/></inline-formula> subscripts.</p>
          <p>The entire classifier thus established is called <bold>iLoc-Gneg</bold>, which can be used to predict the subcellular localization of both singleplex and multiplex Gram-negative bacterial proteins. To provide an intuitive picture, a flowchart is provided in <xref ref-type="fig" rid="pone-0020592-g002"><bold>Fig. 2</bold></xref> to illustrate the prediction process of <bold>iLoc-Gneg</bold>.</p>
          <fig id="pone-0020592-g002" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pone.0020592.g002</object-id>
            <label>Figure 2</label>
            <caption>
              <title>A flowchart to show the prediction process of iLoc-Gneg.</title>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.g002" xlink:type="simple"/>
          </fig>
        </sec>
      </sec>
      <sec id="s2e">
        <title>5. Protocol Guide</title>
        <p>For user's convenience, a web-server for <bold>iLoc-Gneg</bold> was established. Below, let us give a step-by-step guide on how to use it to get the desired results.</p>
        <sec id="s2e1">
          <title>Step 1</title>
          <p>Open the web server at site <ext-link ext-link-type="uri" xlink:href="http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg" xlink:type="simple">http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg</ext-link> and you will see the top page of the predictor on your computer screen, as shown in <xref ref-type="fig" rid="pone-0020592-g003"><bold>Fig. 3</bold></xref>. Click on the <underline>Read Me</underline> button to see a brief introduction about <bold>iLoc-Gneg</bold> predictor and the caveat when using it.</p>
          <fig id="pone-0020592-g003" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pone.0020592.g003</object-id>
            <label>Figure 3</label>
            <caption>
              <title>A semi-screenshot to show the top page of the iLoc-Gneg web-server.</title>
              <p>Its website address is at <ext-link ext-link-type="uri" xlink:href="http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg" xlink:type="simple">http://icpr.jci.edu.cn/bioinfo/iLoc-Gneg</ext-link>.</p>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.g003" xlink:type="simple"/>
          </fig>
        </sec>
        <sec id="s2e2">
          <title>Step 2</title>
          <p>Either type or copy and paste the query protein sequence into the input box at the center of <xref ref-type="fig" rid="pone-0020592-g003"><bold>Fig. 3</bold></xref>. The input sequence should be in the FASTA format. A sequence in FASTA format consists of a single initial line beginning with a greater-than symbol (“&gt;”) in the first column, followed by lines of sequence data. The words right after the “&gt;” symbol in the single initial line are optional and only used for the purpose of identification and description. All lines should be no longer than 120 characters and usually do not exceed 80 characters. The sequence ends if another line starting with a “&gt;” appears; this indicates the start of another sequence. Example sequences in FASTA format can be seen by clicking on the <underline>Example</underline> button right above the input box. For more information about FASTA format, visit <ext-link ext-link-type="uri" xlink:href="http://en.wikipedia.org/wiki/Fasta_format" xlink:type="simple">http://en.wikipedia.org/wiki/Fasta_format</ext-link>. Different with <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, where only one query protein sequence at a time is allowed for each submission, now the maximum number of query proteins for each submission can be 10.</p>
        </sec>
        <sec id="s2e3">
          <title>Step 3</title>
          <p>Click on the <underline>Submit</underline> button to see the predicted result. For example, if you use the three query protein sequences in the <underline>Example</underline> window as the input, after clicking the <underline>Submit</underline> button, you will see <xref ref-type="fig" rid="pone-0020592-g004"><bold>Fig. 4</bold></xref> shown on your screen, indicating that the predicted result for the 1<sup>st</sup> query protein is “<bold>Cell outer membrane</bold>”, that for the 2<sup>nd</sup> one is “<bold>Cytoplasm; Periplasm</bold>”, and that for the 3<sup>rd</sup> one is “<bold>Cell inner membrane; Cytoplasm</bold>”. In other words, the 1<sup>st</sup> query protein (P0A3N8) is a single-location one residing at “cell outer membrane” only, the 2<sup>nd</sup> one (Q05097) can simultaneously reside in two different sites (“cytoplasm” and “periplasm”), and the 3<sup>rd</sup> one (P61380) can also simultaneously reside in two different sites (“cell inner membrane” and “cytoplasm”). All these results are exactly the same as observed by experiments as shown in the <underline><xref ref-type="supplementary-material" rid="pone.0020592.s001">Supporting Information S1</xref></underline>. It takes about 10 seconds for the above computation before the predicted results appear on your computer screen; the more number of query proteins and longer of each sequence, the more time it is usually needed.</p>
          <fig id="pone-0020592-g004" position="float">
            <object-id pub-id-type="doi">10.1371/journal.pone.0020592.g004</object-id>
            <label>Figure 4</label>
            <caption>
              <title>A semi-screenshot to show the output of iLoc-Gneg.</title>
              <p>The input was taken from the three protein sequences listed in the <underline>Example</underline> window of the <bold>iLoc-Gneg</bold> web-server (cf. <xref ref-type="fig" rid="pone-0020592-g003">Fig. 3</xref>).</p>
            </caption>
            <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.g004" xlink:type="simple"/>
          </fig>
        </sec>
        <sec id="s2e4">
          <title>Step 4</title>
          <p>As shown on the lower panel of <xref ref-type="fig" rid="pone-0020592-g003"><bold>Fig. 3</bold></xref>, you may also choose the batch prediction by entering your e-mail address and your desired batch input file (in FASTA format) via the “Browse” button. To see the sample of batch input file, click on the button <underline>Batch-example</underline>. The maximum number of the query proteins for each batch input file is 50. After clicking the button <underline>Batch-submit</underline>, you will see “Your batch job is under computation; once the results are available, you will be notified by e-mail.” Note that if you submit a batch input file from an Apple computer, although it looks like in the FASTA format, your input might change to non-FASTA format in the server end and cause errors. Under such a circumstance, the safest way is to submit your input file with a pdf format.</p>
        </sec>
        <sec id="s2e5">
          <title>Step 5</title>
          <p>Click on the <underline>Citation</underline> button to find the relevant papers that document the detailed development and algorithm of <bold>iLoc-Gneg</bold>.</p>
        </sec>
        <sec id="s2e6">
          <title>Step 6</title>
          <p>Click on the <underline>Data</underline> button to download the benchmark datasets used to train and test the <bold>iLoc-Gneg</bold> predictor.</p>
        </sec>
        <sec id="s2e7">
          <title>Caveat</title>
          <p>To obtain the predicted result with the expected success rate, the entire sequence of the query protein rather than its fragment should be used as an input. A sequence with less than 50 amino acid residues is generally deemed as a fragment. Also, if the query Gram-negative protein is known not one of the 8 locations as shown in <xref ref-type="fig" rid="pone-0020592-g001"><bold>Fig. 1</bold></xref>, stop the prediction because the result thus obtained will not make any sense.</p>
        </sec>
      </sec>
    </sec>
    <sec id="s3">
      <title>Results and Discussion</title>
      <p>In statistical prediction, it would be meaningless to simply report a success rate of a predictor without specifying what method and benchmark dataset were used to test its accuracy <xref ref-type="bibr" rid="pone.0020592-Chou3">[14]</xref>. As is well known, the following three methods are often used to examine the quality of a predictor: independent dataset test, subsampling test, and jackknife test <xref ref-type="bibr" rid="pone.0020592-Chou12">[36]</xref>. Owing to that subsampling test and jackknife test can be performed with one benchmark dataset and that independent dataset test can be treated as a special case of subsampling test, one benchmark dataset would suffice to serve all the three kinds of cross-validation. However, as demonstrated by Eq.1 of <xref ref-type="bibr" rid="pone.0020592-Chou13">[37]</xref> and elucidated in <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>, among the three cross-validation methods, the jackknife test is deemed the least arbitrary that can always yield a unique result for a given benchmark dataset and hence has been widely recognized and increasingly used to examine the power of various predictors (see, e.g., <xref ref-type="bibr" rid="pone.0020592-Cai1">[38]</xref>, <xref ref-type="bibr" rid="pone.0020592-Jahandideh1">[39]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chen1">[40]</xref>, <xref ref-type="bibr" rid="pone.0020592-Kannan1">[41]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chen2">[42]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chen3">[43]</xref>, <xref ref-type="bibr" rid="pone.0020592-Ding1">[44]</xref>, <xref ref-type="bibr" rid="pone.0020592-Du1">[45]</xref>, <xref ref-type="bibr" rid="pone.0020592-Fang1">[46]</xref>, <xref ref-type="bibr" rid="pone.0020592-Gao1">[47]</xref>, <xref ref-type="bibr" rid="pone.0020592-Jahandideh2">[48]</xref>, <xref ref-type="bibr" rid="pone.0020592-Jahandideh3">[49]</xref>, <xref ref-type="bibr" rid="pone.0020592-Li1">[50]</xref>, <xref ref-type="bibr" rid="pone.0020592-Lin1">[51]</xref>, <xref ref-type="bibr" rid="pone.0020592-Masso1">[52]</xref>, <xref ref-type="bibr" rid="pone.0020592-Mohabatkar1">[53]</xref>, <xref ref-type="bibr" rid="pone.0020592-Zou1">[54]</xref>, <xref ref-type="bibr" rid="pone.0020592-Sahu1">[55]</xref>, <xref ref-type="bibr" rid="pone.0020592-Chou14">[56]</xref>). Accordingly, in this study, the jackknife test will be adopted to evaluate the power of <bold>iLoc-Gneg</bold> as well.</p>
      <p>However, even if using the jackknife test to examine the accuracy, a same predictor may still yield obviously different success rates when tested by different benchmark datasets. This is because the more stringent of a benchmark dataset in excluding homologous sequences, the more difficult for a predictor to achieve a high success rate. Also, the more number of subsets (subcellular locations) a benchmark dataset covers, the more difficult to achieve a high overall success rate, as elaborated in a recent review <xref ref-type="bibr" rid="pone.0020592-Chou3">[14]</xref>.</p>
      <p>As mentioned in the Materials section, the benchmark dataset used in this study is <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e193" xlink:type="simple"/></inline-formula> (cf. <underline><xref ref-type="supplementary-material" rid="pone.0020592.s001">Supporting Information S1</xref></underline>), which is the same benchmark dataset constructed in <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> for <bold>Gneg-mPLoc</bold>.</p>
      <p>Actually, for such a dataset containing both single-location and multiple-location Gram-negative proteins distributed among 8 subcellular location sites, so far only one existing predictor, i.e., <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, had the capacity to deal with it. Therefore, to demonstrate the power of the current predictor, it would suffice to just compare <bold>iLoc-Gneg</bold> with <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</p>
      <p>Listed in <xref ref-type="table" rid="pone-0020592-t002"><bold>Table 2</bold></xref> are the results obtained with <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> and <bold>iLoc-Gneg</bold> on the aforementioned benchmark dataset <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e194" xlink:type="simple"/></inline-formula> by the jackknife test. As we can see from <xref ref-type="table" rid="pone-0020592-t002"><bold>Table 2</bold></xref>, for such a stringent and complicated benchmark dataset, the overall success rate achieved by <bold>iLoc-Gneg</bold> is over 91.4%, which is about 6% higher than that by <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>.</p>
      <p>Note that during the course of the jackknife test by <bold>Gneg-mPLoc</bold> and <bold>iLoc-Gneg</bold>, the false positives (over-predictions) and false negatives (under-predictions) were also taken into account to reduce the scores in calculating the overall success rate. As for the detailed process of how to count the over-predictions and under-predictions for a system containing both single-location and multiple-location proteins, see Eqs.43–48 and <xref ref-type="fig" rid="pone-0020592-g004">Fig. 4</xref> in a comprehensive review <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>.</p>
      <p>To provide a more intuitive and easier-to-understand measurement, let us introduce a new scale, the so-called “absolute true” success rate, to reflect the accuracy of a predictor, as defined by<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e195" xlink:type="simple"/><label>(26)</label></disp-formula>where <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e196" xlink:type="simple"/></inline-formula> represents the absolute true rate, <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e197" xlink:type="simple"/></inline-formula> the number of total proteins investigated, and<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e198" xlink:type="simple"/><label>(27)</label></disp-formula>According to the above definition, for a protein belonging to, say, two subcellular locations, if only one of the two is correctly predicted, or the predicted result contains a location not belonging to the two, the prediction score will be counted as 0. In other words, when and only when all the subcellular locations of a query protein are exactly predicted without any underprediction or overprediction, can the prediction be scored with 1. Therefore, the absolute true scale is much more strict and harsh than the scale used previously <xref ref-type="bibr" rid="pone.0020592-Chou1">[2]</xref>, <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> in measuring the success rate. However, even if using such a stringent criterion on the same benchmark dataset by the jackknife test, the overall absolute true success rate achieved by <bold>iLoc-Gneg</bold> was 1252/1392 = 89.9%.</p>
      <p>Why can <bold>iLoc-Gneg</bold> enhance the success rate so remarkably? One of the key reasons is that the GO formulation for protein samples in <bold>iLoc-Gneg</bold> contains more information than that in <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, as elaborated as follows. For example, for the protein with the access number “P0A8U0” as denoted by <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e199" xlink:type="simple"/></inline-formula>, according to Steps 3 and 4 in the Section of “GO (Gene Ontology) Formulation”, we found 47 proteins that were homologous to it; i.e., <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e200" xlink:type="simple"/></inline-formula>. Each of the 47 homologous proteins hit GO:0005886 (or GO_compress:00277) and GO:0016020 (or GO_compress:00830), and hence the two GO numbers were hit by a total of 47 times. Only one of the 47 proteins hit GO:0005737 (or GO_compress: 00269). Substituting these data into Eqs.8–9, we have<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e201" xlink:type="simple"/><label>(28)</label></disp-formula>In contrast, if the same protein was represented according to the formulation in <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>, it would be<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.e202" xlink:type="simple"/><label>(29)</label></disp-formula>It can be seen by a comparison of Eq.28 with Eq.29 that although the elements in the 269<sup>th</sup>, 277<sup>th</sup>, and 830<sup>th</sup> components are all not zero in both formulations, the differences of their weights are completely ignored in Eq.29 as formulated in <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref>. That is also why, when the sequence of <inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e203" xlink:type="simple"/></inline-formula> was inputted into <bold>iLoc-Gneg</bold> and <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> as a query protein for prediction, the former could accurately predict its both location sites (“cell inner membrane” and “cytoplasm”), while the latter could predict only one site (“cell inner membrane”) but miss the site of “cytoplasm”.</p>
      <sec id="s3a">
        <title>Conclusions</title>
        <p>Prediction of protein subcellular localization is a challenging problem, particularly when the system concerned contains both singleplex and multiplex proteins. The reasons why <bold>iLoc-Gneg</bold> can achieve higher success rates than <bold>Gneg-mPLoc</bold> are as follows. <bold>(1)</bold> The GO formulation used to represent protein samples in <bold>iLoc-Gneg</bold> is formed by the probabilities of hits (cf. Eqs.8–9) and hence contains more information than that in <bold>Gneg-mPLoc</bold> <xref ref-type="bibr" rid="pone.0020592-Shen1">[11]</xref> where only the number “0” or “1” was used regardless how many hits were found to the corresponding component in the GO formulation. <bold>(2)</bold> The accumulation-layer scale has been introduced in <bold>iLoc-Gneg</bold> that is more natural and effective for dealing with proteins having both single and multiple subcellular locations.</p>
      </sec>
    </sec>
    <sec id="s4">
      <title>Supporting Information</title>
      <supplementary-material id="pone.0020592.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0020592.s001" xlink:type="simple">
        <label>Supporting Information S1</label>
        <caption>
          <p><bold>This benchmark dataset </bold><inline-formula><inline-graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0020592.e204" xlink:type="simple"/></inline-formula><bold> includes 1,456 locative protein sequences (1,392 different proteins), classified into 8 Gram-negative subcellular locations.</bold> Among the 1,392 different proteins, 1,328 belong to one location; and 64 to two locations. Both the accession numbers and sequences are given. None of the proteins has ≥25% sequence identity to any other in the same subset (subcellular location). See the text of the paper for further explanation.</p>
          <p>(PDF)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>The authors wish to thank the two anonymous reviewers for their valuable comments, which are very helpful for strengthening the presentation of this paper.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pone.0020592-Nakai1">
        <label>1</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nakai</surname><given-names>K</given-names></name></person-group>             <year>2000</year>             <article-title>Protein sorting signals and prediction of subcellular localization.</article-title>             <source>Advances in Protein Chemistry</source>             <volume>54</volume>             <fpage>277</fpage>             <lpage>344</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou1">
        <label>2</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2007</year>             <article-title>Review: Recent progresses in protein subcellular location prediction.</article-title>             <source>Analytical Biochemistry</source>             <volume>370</volume>             <fpage>1</fpage>             <lpage>16</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Nakai2">
        <label>3</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nakai</surname><given-names>K</given-names></name><name name-style="western"><surname>Kanehisa</surname><given-names>M</given-names></name></person-group>             <year>1991</year>             <article-title>Expert system for predicting protein localization sites in Gram-negative bacteria.</article-title>             <source>Proteins: Structure, Function and Genetics</source>             <volume>11</volume>             <fpage>95</fpage>             <lpage>110</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Nakai3">
        <label>4</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nakai</surname><given-names>K</given-names></name><name name-style="western"><surname>Horton</surname><given-names>P</given-names></name></person-group>             <year>1999</year>             <article-title>PSORT: a program for detecting sorting signals in proteins and predicting their subcellular localization.</article-title>             <source>Trends in Biochemical Science</source>             <volume>24</volume>             <fpage>34</fpage>             <lpage>36</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Gardy1">
        <label>5</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gardy</surname><given-names>JL</given-names></name><name name-style="western"><surname>Spencer</surname><given-names>C</given-names></name><name name-style="western"><surname>Wang</surname><given-names>K</given-names></name><name name-style="western"><surname>Ester</surname><given-names>M</given-names></name><name name-style="western"><surname>Tusnady</surname><given-names>GE</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>PSORT-B: Improving protein subcellular localization prediction for Gram-negative bacteria.</article-title>             <source>Nucleic Acids Research</source>             <volume>31</volume>             <fpage>3613</fpage>             <lpage>3617</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Gardy2">
        <label>6</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gardy</surname><given-names>JL</given-names></name><name name-style="western"><surname>Laird</surname><given-names>MR</given-names></name><name name-style="western"><surname>Chen</surname><given-names>F</given-names></name><name name-style="western"><surname>Rey</surname><given-names>S</given-names></name><name name-style="western"><surname>Walsh</surname><given-names>CJ</given-names></name><etal/></person-group>             <year>2005</year>             <article-title>PSORTb v.2.0: expanded prediction of bacterial protein subcellular localization and insights gained from comparative proteome analysis.</article-title>             <source>Bioinformatics</source>             <volume>21</volume>             <fpage>617</fpage>             <lpage>623</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou2">
        <label>7</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2006</year>             <article-title>Large-scale predictions of Gram-negative bacterial protein subcellular locations.</article-title>             <source>Journal of Proteome Research</source>             <volume>5</volume>             <fpage>3420</fpage>             <lpage>3428</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Smith1">
        <label>8</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Smith</surname><given-names>C</given-names></name></person-group>             <year>2008</year>             <article-title>Subcellular targeting of proteins and drugs.</article-title>             <comment><ext-link ext-link-type="uri" xlink:href="http://www.biocompare.com/Articles/FeaturedArticle/976/Subcellular-Targeting-Of-Proteins-And-Drugs.html" xlink:type="simple">http://www.biocompare.com/Articles/FeaturedArticle/976/Subcellular-Targeting-Of-Proteins-And-Drugs.html</ext-link></comment>          </element-citation>
      </ref>
      <ref id="pone.0020592-Glory1">
        <label>9</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Glory</surname><given-names>E</given-names></name><name name-style="western"><surname>Murphy</surname><given-names>RF</given-names></name></person-group>             <year>2007</year>             <article-title>Automated subcellular location determination and high-throughput microscopy.</article-title>             <source>Dev Cell</source>             <volume>12</volume>             <fpage>7</fpage>             <lpage>16</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Millar1">
        <label>10</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Millar</surname><given-names>AH</given-names></name><name name-style="western"><surname>Carrie</surname><given-names>C</given-names></name><name name-style="western"><surname>Pogson</surname><given-names>B</given-names></name><name name-style="western"><surname>Whelan</surname><given-names>J</given-names></name></person-group>             <year>2009</year>             <article-title>Exploring the function-location nexus: using multiple lines of evidence in defining the subcellular location of plant proteins.</article-title>             <source>Plant Cell</source>             <volume>21</volume>             <fpage>1625</fpage>             <lpage>1631</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Shen1">
        <label>11</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>2010</year>             <article-title>Gneg-mPLoc: A top-down strategy to enhance the quality of predicting subcellular localization of Gram-negative bacterial proteins.</article-title>             <source>Journal of Theoretical Biology</source>             <volume>264</volume>             <fpage>326</fpage>             <lpage>333</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Ashburner1">
        <label>12</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ashburner</surname><given-names>M</given-names></name><name name-style="western"><surname>Ball</surname><given-names>CA</given-names></name><name name-style="western"><surname>Blake</surname><given-names>JA</given-names></name><name name-style="western"><surname>Botstein</surname><given-names>D</given-names></name><name name-style="western"><surname>Butler</surname><given-names>H</given-names></name><etal/></person-group>             <year>2000</year>             <article-title>Gene ontology: tool for the unification of biology.</article-title>             <source>Nature Genetics</source>             <volume>25</volume>             <fpage>25</fpage>             <lpage>29</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Camon1">
        <label>13</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Camon</surname><given-names>E</given-names></name><name name-style="western"><surname>Magrane</surname><given-names>M</given-names></name><name name-style="western"><surname>Barrell</surname><given-names>D</given-names></name><name name-style="western"><surname>Lee</surname><given-names>V</given-names></name><name name-style="western"><surname>Dimmer</surname><given-names>E</given-names></name><etal/></person-group>             <year>2004</year>             <article-title>The Gene Ontology Annotation (GOA) Database: sharing knowledge in Uniprot with Gene Ontology.</article-title>             <source>Nucleic Acids Res</source>             <volume>32</volume>             <fpage>D262</fpage>             <lpage>266</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou3">
        <label>14</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>2011</year>             <article-title>Some remarks on protein attribute prediction and pseudo amino acid composition (50th Anniversary Year Review).</article-title>             <source>Journal of Theoretical Biology</source>             <volume>273</volume>             <fpage>236</fpage>             <lpage>247</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou4">
        <label>15</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2009</year>             <article-title>Review: recent advances in developing web-servers for predicting protein attributes.</article-title>             <source>Natural Science</source>             <volume>2</volume>             <fpage>63</fpage>             <lpage>92</lpage>             <comment>(openly accessible at <ext-link ext-link-type="uri" xlink:href="http://www.scirp.org/journal/NS/" xlink:type="simple">http://www.scirp.org/journal/NS/</ext-link>)</comment>          </element-citation>
      </ref>
      <ref id="pone.0020592-Altschul1">
        <label>16</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Altschul</surname><given-names>SF</given-names></name></person-group>             <year>1997</year>             <article-title>Evaluating the statistical significance of multiple distinct local alignments.</article-title>             <person-group person-group-type="editor"><name name-style="western"><surname>Suhai</surname><given-names>S</given-names></name></person-group>             <source>Theoretical and Computational Methods in Genome Research</source>             <publisher-loc>New York</publisher-loc>             <publisher-name>Plenum</publisher-name>             <fpage>1</fpage>             <lpage>14</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Wootton1">
        <label>17</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Wootton</surname><given-names>JC</given-names></name><name name-style="western"><surname>Federhen</surname><given-names>S</given-names></name></person-group>             <year>1993</year>             <article-title>Statistics of local complexity in amino acid sequences and sequence databases.</article-title>             <source>Comput Chem</source>             <volume>17</volume>             <fpage>149</fpage>             <lpage>163</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou5">
        <label>18</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Zhang</surname><given-names>CT</given-names></name></person-group>             <year>1994</year>             <article-title>Predicting protein folding types by distance functions that make allowances for amino acid interactions.</article-title>             <source>Journal of Biological Chemistry</source>             <volume>269</volume>             <fpage>22014</fpage>             <lpage>22020</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Nakashima1">
        <label>19</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nakashima</surname><given-names>H</given-names></name><name name-style="western"><surname>Nishikawa</surname><given-names>K</given-names></name></person-group>             <year>1994</year>             <article-title>Discrimination of intracellular and extracellular proteins using amino acid composition and residue-pair frequencies.</article-title>             <source>J Mol Biol</source>             <volume>238</volume>             <fpage>54</fpage>             <lpage>61</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou6">
        <label>20</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>1995</year>             <article-title>A novel approach to predicting protein structural classes in a (20-1)-D amino acid composition space.</article-title>             <source>Proteins: Structure, Function &amp; Genetics</source>             <volume>21</volume>             <fpage>319</fpage>             <lpage>344</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Cedano1">
        <label>21</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Cedano</surname><given-names>J</given-names></name><name name-style="western"><surname>Aloy</surname><given-names>P</given-names></name><name name-style="western"><surname>P'erez-Pons</surname><given-names>JA</given-names></name><name name-style="western"><surname>Querol</surname><given-names>E</given-names></name></person-group>             <year>1997</year>             <article-title>Relation between amino acid composition and cellular location of proteins.</article-title>             <source>J Mol Biol</source>             <volume>266</volume>             <fpage>594</fpage>             <lpage>600</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Reinhardt1">
        <label>22</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Reinhardt</surname><given-names>A</given-names></name><name name-style="western"><surname>Hubbard</surname><given-names>T</given-names></name></person-group>             <year>1998</year>             <article-title>Using neural networks for prediction of the subcellular location of proteins.</article-title>             <source>Nucleic Acids Research</source>             <volume>26</volume>             <fpage>2230</fpage>             <lpage>2236</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou7">
        <label>23</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Elrod</surname><given-names>DW</given-names></name></person-group>             <year>1999</year>             <article-title>Protein subcellular location prediction.</article-title>             <source>Protein Engineering</source>             <volume>12</volume>             <fpage>107</fpage>             <lpage>118</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Zhou1">
        <label>24</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>GP</given-names></name><name name-style="western"><surname>Doctor</surname><given-names>K</given-names></name></person-group>             <year>2003</year>             <article-title>Subcellular location prediction of apoptosis proteins.</article-title>             <source>PROTEINS: Structure, Function, and Genetics</source>             <volume>50</volume>             <fpage>44</fpage>             <lpage>48</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou8">
        <label>25</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>2001</year>             <article-title>Prediction of protein cellular attributes using pseudo amino acid composition.</article-title>             <source>PROTEINS: Structure, Function, and Genetics (Erratum: ibid, 2001, Vol 44, 60)</source>             <volume>43</volume>             <fpage>246</fpage>             <lpage>255</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou9">
        <label>26</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2008</year>             <article-title>Cell-PLoc: A package of Web servers for predicting subcellular localization of proteins in various organisms.</article-title>             <source>Nature Protocols</source>             <volume>3</volume>             <fpage>153</fpage>             <lpage>162</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Schaffer1">
        <label>27</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Schaffer</surname><given-names>AA</given-names></name><name name-style="western"><surname>Aravind</surname><given-names>L</given-names></name><name name-style="western"><surname>Madden</surname><given-names>TL</given-names></name><name name-style="western"><surname>Shavirin</surname><given-names>S</given-names></name><name name-style="western"><surname>Spouge</surname><given-names>JL</given-names></name><etal/></person-group>             <year>2001</year>             <article-title>Improving the accuracy of PSI-BLAST protein database searches with composition-based statistics and other refinements.</article-title>             <source>Nucleic Acids Res</source>             <volume>29</volume>             <fpage>2994</fpage>             <lpage>3005</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Loewenstein1">
        <label>28</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Loewenstein</surname><given-names>Y</given-names></name><name name-style="western"><surname>Raimondo</surname><given-names>D</given-names></name><name name-style="western"><surname>Redfern</surname><given-names>OC</given-names></name><name name-style="western"><surname>Watson</surname><given-names>J</given-names></name><name name-style="western"><surname>Frishman</surname><given-names>D</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Protein function annotation by homology-based inference.</article-title>             <source>Genome Biol</source>             <volume>10</volume>             <fpage>207</fpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Gerstein1">
        <label>29</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gerstein</surname><given-names>M</given-names></name><name name-style="western"><surname>Thornton</surname><given-names>JM</given-names></name></person-group>             <year>2003</year>             <article-title>Sequences and topology.</article-title>             <source>Curr Opin Struct Biol</source>             <volume>13</volume>             <fpage>341</fpage>             <lpage>343</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou10">
        <label>30</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>2004</year>             <article-title>Review: Structural bioinformatics and its impact to biomedical science.</article-title>             <source>Current Medicinal Chemistry</source>             <volume>11</volume>             <fpage>2105</fpage>             <lpage>2134</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Camon2">
        <label>31</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Camon</surname><given-names>E</given-names></name><name name-style="western"><surname>Magrane</surname><given-names>M</given-names></name><name name-style="western"><surname>Barrell</surname><given-names>D</given-names></name><name name-style="western"><surname>Binns</surname><given-names>D</given-names></name><name name-style="western"><surname>Fleischmann</surname><given-names>W</given-names></name><etal/></person-group>             <year>2003</year>             <article-title>The Gene Ontology Annotation (GOA) project: implementation of GO in SWISS-PROT, TrEMBL, and InterPro.</article-title>             <source>Genome Res</source>             <volume>13</volume>             <fpage>662</fpage>             <lpage>672</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou11">
        <label>32</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name></person-group>             <year>1995</year>             <article-title>The convergence-divergence duality in lectin domains of the selectin family and its implications.</article-title>             <source>FEBS Letters</source>             <volume>363</volume>             <fpage>123</fpage>             <lpage>126</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Mardia1">
        <label>33</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Mardia</surname><given-names>KV</given-names></name><name name-style="western"><surname>Kent</surname><given-names>JT</given-names></name><name name-style="western"><surname>Bibby</surname><given-names>JM</given-names></name></person-group>             <year>1979</year>             <source>Multivariate Analysis: Chapter 11 Discriminant Analysis; Chapter 12 Multivariate analysis of variance; Chapter 13 cluster analysis (pp. 322–381)</source>             <publisher-loc>London</publisher-loc>             <publisher-name>Academic Press</publisher-name>             <fpage>322</fpage>             <lpage>381</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Mahalanobis1">
        <label>34</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Mahalanobis</surname><given-names>PC</given-names></name></person-group>             <year>1936</year>             <article-title>On the generalized distance in statistics.</article-title>             <source>Proc Natl Inst Sci India</source>             <volume>2</volume>             <fpage>49</fpage>             <lpage>55</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Pillai1">
        <label>35</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Pillai</surname><given-names>KCS</given-names></name></person-group>             <year>1985</year>             <article-title>Mahalanobis D2.</article-title>             <person-group person-group-type="editor"><name name-style="western"><surname>Kotz</surname><given-names>S</given-names></name><name name-style="western"><surname>Johnson</surname><given-names>NL</given-names></name></person-group>             <source>Encyclopedia of Statistical Sciences</source>             <publisher-loc>New York</publisher-loc>             <publisher-name>John Wiley &amp; Sons</publisher-name>             <fpage>176</fpage>             <lpage>181</lpage>             <comment>This reference also presents a brief biography of Mahalanobis who was a man of great originality and who made considerable contributions to statistics</comment>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou12">
        <label>36</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Zhang</surname><given-names>CT</given-names></name></person-group>             <year>1995</year>             <article-title>Review: Prediction of protein structural classes.</article-title>             <source>Critical Reviews in Biochemistry and Molecular Biology</source>             <volume>30</volume>             <fpage>275</fpage>             <lpage>349</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou13">
        <label>37</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2010</year>             <article-title>Cell-PLoc 2.0: An improved package of web-servers for predicting subcellular localization of proteins in various organisms.</article-title>             <source>Natural Science</source>             <volume>2</volume>             <fpage>1090</fpage>             <lpage>1103</lpage>             <comment>(openly accessible at <ext-link ext-link-type="uri" xlink:href="http://www.scirp.org/journal/NS/" xlink:type="simple">http://www.scirp.org/journal/NS/</ext-link>)</comment>          </element-citation>
      </ref>
      <ref id="pone.0020592-Cai1">
        <label>38</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Cai</surname><given-names>YD</given-names></name><name name-style="western"><surname>He</surname><given-names>J</given-names></name><name name-style="western"><surname>Li</surname><given-names>X</given-names></name><name name-style="western"><surname>Feng</surname><given-names>K</given-names></name><name name-style="western"><surname>Lu</surname><given-names>L</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>Predicting protein subcellular locations with feature selection and analysis.</article-title>             <source>Protein Pept Lett</source>             <volume>17</volume>             <fpage>464</fpage>             <lpage>472</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Jahandideh1">
        <label>39</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Jahandideh</surname><given-names>S</given-names></name><name name-style="western"><surname>Hoseini</surname><given-names>S</given-names></name><name name-style="western"><surname>Jahandideh</surname><given-names>M</given-names></name><name name-style="western"><surname>Hoseini</surname><given-names>A</given-names></name><name name-style="western"><surname>Disfani</surname><given-names>FM</given-names></name></person-group>             <year>2009</year>             <article-title>Gamma-turn types prediction in proteins using the two-stage hybrid neural discriminant model.</article-title>             <source>Journal of Theoretical Biology</source>             <volume>259</volume>             <fpage>517</fpage>             <lpage>522</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chen1">
        <label>40</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names></name><name name-style="western"><surname>Chen</surname><given-names>L</given-names></name><name name-style="western"><surname>Zou</surname><given-names>X</given-names></name><name name-style="western"><surname>Cai</surname><given-names>P</given-names></name></person-group>             <year>2009</year>             <article-title>Prediction of protein secondary structure content by using the concept of Chou's pseudo amino acid composition and support vector machine.</article-title>             <source>Protein &amp; Peptide Letters</source>             <volume>16</volume>             <fpage>27</fpage>             <lpage>31</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Kannan1">
        <label>41</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kannan</surname><given-names>S</given-names></name><name name-style="western"><surname>Hauth</surname><given-names>AM</given-names></name><name name-style="western"><surname>Burger</surname><given-names>G</given-names></name></person-group>             <year>2008</year>             <article-title>Function prediction of hypothetical proteins without sequence similarity to proteins of known function.</article-title>             <source>Protein &amp; Peptide Letters</source>             <volume>15</volume>             <fpage>1107</fpage>             <lpage>1116</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chen2">
        <label>42</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names></name><name name-style="western"><surname>Chen</surname><given-names>LX</given-names></name><name name-style="western"><surname>Zou</surname><given-names>XY</given-names></name><name name-style="western"><surname>Cai</surname><given-names>PX</given-names></name></person-group>             <year>2008</year>             <article-title>Predicting protein structural class based on multi-features fusion.</article-title>             <source>Journal of Theoretical Biology</source>             <volume>253</volume>             <fpage>388</fpage>             <lpage>392</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chen3">
        <label>43</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>K</given-names></name><name name-style="western"><surname>Kurgan</surname><given-names>LA</given-names></name><name name-style="western"><surname>Ruan</surname><given-names>J</given-names></name></person-group>             <year>2008</year>             <article-title>Prediction of protein structural class using novel evolutionary collocation-based sequence representation.</article-title>             <source>J Comput Chem</source>             <volume>29</volume>             <fpage>1596</fpage>             <lpage>1604</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Ding1">
        <label>44</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>H</given-names></name><name name-style="western"><surname>Luo</surname><given-names>L</given-names></name><name name-style="western"><surname>Lin</surname><given-names>H</given-names></name></person-group>             <year>2009</year>             <article-title>Prediction of cell wall lytic enzymes using Chou's amphiphilic pseudo amino acid composition.</article-title>             <source>Protein &amp; Peptide Letters</source>             <volume>16</volume>             <fpage>351</fpage>             <lpage>355</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Du1">
        <label>45</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>P</given-names></name><name name-style="western"><surname>Cao</surname><given-names>S</given-names></name><name name-style="western"><surname>Li</surname><given-names>Y</given-names></name></person-group>             <year>2009</year>             <article-title>SubChlo: predicting protein subchloroplast locations with pseudo-amino acid composition and the evidence-theoretic K-nearest neighbor (ET-KNN) algorithm.</article-title>             <source>Journal of Theoretical Biolology</source>             <volume>261</volume>             <fpage>330</fpage>             <lpage>335</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Fang1">
        <label>46</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>Y</given-names></name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names></name><name name-style="western"><surname>Feng</surname><given-names>Y</given-names></name><name name-style="western"><surname>Li</surname><given-names>M</given-names></name></person-group>             <year>2008</year>             <article-title>Predicting DNA-binding proteins: approached from Chou's pseudo amino acid composition and other specific sequence features.</article-title>             <source>Amino Acids</source>             <volume>34</volume>             <fpage>103</fpage>             <lpage>109</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Gao1">
        <label>47</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>QB</given-names></name><name name-style="western"><surname>Jin</surname><given-names>ZC</given-names></name><name name-style="western"><surname>Ye</surname><given-names>XF</given-names></name><name name-style="western"><surname>Wu</surname><given-names>C</given-names></name><name name-style="western"><surname>He</surname><given-names>J</given-names></name></person-group>             <year>2009</year>             <article-title>Prediction of nuclear receptors with optimal pseudo amino acid composition.</article-title>             <source>Analytical Biochemistry</source>             <volume>387</volume>             <fpage>54</fpage>             <lpage>59</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Jahandideh2">
        <label>48</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Jahandideh</surname><given-names>S</given-names></name><name name-style="western"><surname>Abdolmaleki</surname><given-names>P</given-names></name><name name-style="western"><surname>Jahandideh</surname><given-names>M</given-names></name><name name-style="western"><surname>Asadabadi</surname><given-names>EB</given-names></name></person-group>             <year>2007</year>             <article-title>Novel two-stage hybrid neural discriminant model for predicting proteins structural classes.</article-title>             <source>Biophys Chem</source>             <volume>128</volume>             <fpage>87</fpage>             <lpage>93</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Jahandideh3">
        <label>49</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Jahandideh</surname><given-names>S</given-names></name><name name-style="western"><surname>Sarvestani</surname><given-names>AS</given-names></name><name name-style="western"><surname>Abdolmaleki</surname><given-names>P</given-names></name><name name-style="western"><surname>Jahandideh</surname><given-names>M</given-names></name><name name-style="western"><surname>Barfeie</surname><given-names>M</given-names></name></person-group>             <year>2007</year>             <article-title>gamma-Turn types prediction in proteins using the support vector machines.</article-title>             <source>J Theor Biol</source>             <volume>249</volume>             <fpage>785</fpage>             <lpage>790</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Li1">
        <label>50</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>FM</given-names></name><name name-style="western"><surname>Li</surname><given-names>QZ</given-names></name></person-group>             <year>2008</year>             <article-title>Predicting protein subcellular location using Chou's pseudo amino acid composition and improved hybrid approach.</article-title>             <source>Protein &amp; Peptide Letters</source>             <volume>15</volume>             <fpage>612</fpage>             <lpage>616</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Lin1">
        <label>51</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>H</given-names></name></person-group>             <year>2008</year>             <article-title>The modified Mahalanobis discriminant for predicting outer membrane proteins by using Chou's pseudo amino acid composition.</article-title>             <source>Journal of Theoretical Biology</source>             <volume>252</volume>             <fpage>350</fpage>             <lpage>356</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Masso1">
        <label>52</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Masso</surname><given-names>M</given-names></name><name name-style="western"><surname>Vaisman</surname><given-names>II</given-names></name></person-group>             <year>2010</year>             <article-title>Knowledge-based computational mutagenesis for predicting the disease potential of human non-synonymous single nucleotide polymorphisms.</article-title>             <source>Journal of Theoretical Biology</source>             <volume>266</volume>             <fpage>560</fpage>             <lpage>568</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Mohabatkar1">
        <label>53</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Mohabatkar</surname><given-names>H</given-names></name></person-group>             <year>2010</year>             <article-title>Prediction of cyclin proteins using Chou's pseudo amino acid composition.</article-title>             <source>Protein &amp; Peptide Letters</source>             <volume>17</volume>             <fpage>1207</fpage>             <lpage>1214</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Zou1">
        <label>54</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Zou</surname><given-names>D</given-names></name><name name-style="western"><surname>He</surname><given-names>Z</given-names></name><name name-style="western"><surname>He</surname><given-names>J</given-names></name><name name-style="western"><surname>Xia</surname><given-names>Y</given-names></name></person-group>             <year>2011</year>             <article-title>Supersecondary structure prediction using Chou's pseudo amino acid composition.</article-title>             <source>Journal of Computational Chemistry</source>             <volume>32</volume>             <fpage>271</fpage>             <lpage>278</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Sahu1">
        <label>55</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Sahu</surname><given-names>SS</given-names></name><name name-style="western"><surname>Panda</surname><given-names>G</given-names></name></person-group>             <year>2010</year>             <article-title>A novel feature representation method based on Chou's pseudo amino acid composition for protein structural class prediction.</article-title>             <source>Computational Biology and Chemistry</source>             <volume>34</volume>             <fpage>320</fpage>             <lpage>327</lpage>          </element-citation>
      </ref>
      <ref id="pone.0020592-Chou14">
        <label>56</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>KC</given-names></name><name name-style="western"><surname>Shen</surname><given-names>HB</given-names></name></person-group>             <year>2010</year>             <article-title>Plant-mPLoc: A Top-Down Strategy to Augment the Power for Predicting Plant Protein Subcellular Localization.</article-title>             <source>PLoS ONE</source>             <volume>5</volume>             <fpage>e11335</fpage>          </element-citation>
      </ref>
    </ref-list>
    
  </back>
</article>