<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
  <front>
    <journal-meta><journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="pmc">plosone</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS ONE</journal-title></journal-title-group><issn pub-type="epub">1932-6203</issn><publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher></journal-meta>
    <article-meta><article-id pub-id-type="publisher-id">PONE-D-12-07697</article-id><article-id pub-id-type="doi">10.1371/journal.pone.0040155</article-id><article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Biochemistry</subject>
            <subj-group>
              <subject>Glycobiology</subject>
              <subj-group>
                <subject>Glycoproteins</subject>
              </subj-group>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Computational biology</subject>
            <subj-group>
              <subject>Sequence analysis</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Microbiology</subject>
            <subj-group>
              <subject>Archaeans</subject>
              <subject>Bacterial pathogens</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Proteomics</subject>
            <subj-group>
              <subject>Protein engineering</subject>
              <subject>Sequence analysis</subject>
              <subject>Spectrometric identification of proteins</subject>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Microbiology</subject>
          <subject>Computational Biology</subject>
          <subject>Biochemistry</subject>
        </subj-group>
      </article-categories><title-group><article-title>GlycoPP: A Webserver for Prediction of N- and O-Glycosites in Prokaryotic Protein Sequences</article-title><alt-title alt-title-type="running-head">Prokaryotic Glycosylated-Residue Prediction</alt-title></title-group><contrib-group>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Chauhan</surname>
            <given-names>Jagat S.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Bhat</surname>
            <given-names>Adil H.</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Raghava</surname>
            <given-names>Gajendra P. S.</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Rao</surname>
            <given-names>Alka</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
      </contrib-group><aff id="aff1"><label>1</label><addr-line>Bioinformatics Centre, Institute of Microbial Technology, Council of Scientific and Industrial Research, Chandigarh, India</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>Protein Science and Engineering, Institute of Microbial Technology, Council of Scientific and Industrial Research, Chandigarh, India</addr-line>       </aff><contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Burchell</surname>
            <given-names>Joy Marilyn</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group><aff id="edit1">King’s College London, United Kingdom</aff><author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">raoalka@imtech.res</email> (AR); <email xlink:type="simple">raghava@imtech.res.in</email> (GPSR)</corresp>
        <fn fn-type="con">
          <p>Conceived and designed the experiments: AR GPSR. Performed the experiments: JSC. Analyzed the data: AR GPSR. Contributed reagents/materials/analysis tools: AHB. Wrote the paper: JSC AHB AR.</p>
        </fn>
      <fn fn-type="conflict">
        <p>The authors have declared that no competing interests exist.</p>
      </fn></author-notes><pub-date pub-type="collection">
        <year>2012</year>
      </pub-date><pub-date pub-type="epub">
        <day>9</day>
        <month>7</month>
        <year>2012</year>
      </pub-date><volume>7</volume><issue>7</issue><elocation-id>e40155</elocation-id><history>
        <date date-type="received">
          <day>14</day>
          <month>3</month>
          <year>2012</year>
        </date>
        <date date-type="accepted">
          <day>1</day>
          <month>6</month>
          <year>2012</year>
        </date>
      </history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2012</copyright-year><copyright-holder>Chauhan et al</copyright-holder><license><license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions><abstract>
        <p>Glycosylation is one of the most abundant post-translational modifications (PTMs) required for various structure/function modulations of proteins in a living cell. Although elucidated recently in prokaryotes, this type of PTM is present across all three domains of life. In prokaryotes, two types of protein glycan linkages are more widespread namely, N- linked, where a glycan moiety is attached to the amide group of Asn, and O- linked, where a glycan moiety is attached to the hydroxyl group of Ser/Thr/Tyr. For their biologically ubiquitous nature, significance, and technology applications, the study of prokaryotic glycoproteins is a fast emerging area of research. Here we describe new Support Vector Machine (SVM) based algorithms (models) developed for predicting glycosylated-residues (glycosites) with high accuracy in prokaryotic protein sequences. The models are based on binary profile of patterns, composition profile of patterns, and position-specific scoring matrix profile of patterns as training features. The study employ an extensive dataset of 107 N-linked and 116 O-linked glycosites extracted from 59 experimentally characterized glycoproteins of prokaryotes. This dataset includes validated N-glycosites from phyla <italic>Crenarchaeota</italic>, <italic>Euryarchaeota</italic> (domain Archaea), <italic>Proteobacteria</italic> (domain Bacteria) and validated O-glycosites from phyla <italic>Actinobacteria</italic>, <italic>Bacteroidetes</italic>, <italic>Firmicutes</italic> and <italic>Proteobacteria</italic> (domain Bacteria). In view of the current understanding that glycosylation occurs on folded proteins in bacteria, hybrid models have been developed using information on predicted secondary structures and accessible surface area in various combinations with training features. Using these models, N-glycosites and O-glycosites could be predicted with an accuracy of 82.71% (MCC 0.65) and 73.71% (MCC 0.48), respectively. An evaluation of the best performing models with 28 independent prokaryotic glycoproteins confirms the suitability of these models in predicting N- and O-glycosites in potential glycoproteins from aforementioned organisms, with reasonably high confidence. A web server GlycoPP, implementing these models is available freely at <ext-link ext-link-type="uri" xlink:href="http:/www.imtech.res.in/raghava/glycopp/" xlink:type="simple">http:/www.imtech.res.in/raghava/glycopp/</ext-link>.</p>
      </abstract><funding-group><funding-statement>AR and GPSR acknowledge the financial support from Council of Scientific and Industrial Research (CSIR) and Department of Biotechnology (DBT), Government of India, respectively. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts>
        <page-count count="13"/>
      </counts></article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>Glycosylation is a recently identified post-translational modification of proteins in prokaryotes: Archaea and Bacteria <xref ref-type="bibr" rid="pone.0040155-Messner1">[1]</xref>, <xref ref-type="bibr" rid="pone.0040155-AbuQarn1">[2]</xref>. A glycan moiety is attached enzymatically to a protein by the process of glycosylation. Glycosylation is known to influence biological properties like activity, solubility, folding, conformation, stability, half-life, and/or immunogenicity of different cellular proteins thereby modulating the structure/function of these proteins for a variety of cellular/extracellular functions in a living cell <xref ref-type="bibr" rid="pone.0040155-Lechner1">[3]</xref>–<xref ref-type="bibr" rid="pone.0040155-Varki1">[5]</xref>. Owing to their involvement in host-pathogen interactions, immunogenicity and in many other important cellular functions, a number of bacterial and archaeal glycoproteins have been characterized experimentally <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>–<xref ref-type="bibr" rid="pone.0040155-Jennings1">[10]</xref>. Determination of glycosite(s) is one important aspect of glycoprotein characterization. Analysis of glysosites and their neighboring sequence and structural contexts may also provide important evolutionary insights and understanding of acceptor specificities of the protein glycosylating enzymes called glycosyltransferases (GTs) and oligosaccharyltransferases (OSTs), <xref ref-type="bibr" rid="pone.0040155-AbuQarn1">[2]</xref>. The experimental characterization of glycosite(s) and the glycoproteins, however, could be difficult, technically demanding, and time-consuming owing to the labile nature of modification involved as well as lack of high-senstivity yet cost-effective methods for glycoprotein detection. Therefore, the computational algorithms/models to predict glycosites in protein sequences are very useful in complementing and facilitating such studies. A number of such algorithms have been developed to predict glycosites in eukaryotic glycoproteins using different tools of machine learning like Neural Network based (NetOglyc), <xref ref-type="bibr" rid="pone.0040155-Hansen1">[11]</xref>, <xref ref-type="bibr" rid="pone.0040155-Julenius1">[12]</xref> Support Vector Machine (SVM) based (NetNglyc), <xref ref-type="bibr" rid="pone.0040155-Gupta1">[13]</xref>, Ensemble of SVMs (EnsembleGly), <xref ref-type="bibr" rid="pone.0040155-Caragea1">[14]</xref> and Random Forest based <xref ref-type="bibr" rid="pone.0040155-Hamby1">[15]</xref>. All these existing tools are trained on eukaryotic glycoprotein sequences. However, for non-availability of equivalent methods, these tools are routinely used for analyzing glycoproteomics data and potential glycosite analysis in prokaryotic glycoproteins for both N- and O- type of glycosylation <xref ref-type="bibr" rid="pone.0040155-Hanna1">[16]</xref>–<xref ref-type="bibr" rid="pone.0040155-Ghoshal1">[19]</xref>. In similar context, Dell and co-workers have discussed the unsuitability of these existing glycosite prediction tools in correctly predicting glycosites (especially O-glycosites), in most families of characterized prokaryotic glycoproteins that included pilins, flagellins, autotransporters and serine-rich proteins <xref ref-type="bibr" rid="pone.0040155-Dell1">[20]</xref>. In this study using a dataset of experimentally validated 107 N-linked and 116 O-linked glycosites from archaeal and bacterial glycoproteins, we have found that indeed these tools (as detailed in <xref ref-type="table" rid="pone-0040155-t001">Table 1</xref>), fail to provide reliable predictions for glycosites in prokaryotic glycoproteins. Furthermore, protein glycosylation in prokaryotes is much more versatile than in eukaryotes in terms of both mechanisms involved and the types of glycans and linkages present as discussed in references <xref ref-type="bibr" rid="pone.0040155-Dell1">[20]</xref>–<xref ref-type="bibr" rid="pone.0040155-Marino1">[22]</xref> &amp; <xref ref-type="table" rid="pone-0040155-t002">Table 2</xref>. Among archaea N-glycosylation is believed to be widespread yet experimental evidence exists only in case of phyla <italic>Crenarchaeota</italic> and <italic>Euryarchaeota</italic> where it is mediated by an enzyme AglB and its homologues and sugar is transferred on to NX(S/T)(where X≠P) acceptor sequon in an <italic>en-bloc</italic> fashion. Similarly, in bacteria N-glycosylation is known and experimentally validated only in a few organisms belonging to phylum <italic>Proteobacteria</italic>. In <italic>Proteobacteria</italic> both sequentially (in cytoplasm, ex. <italic>Haemophilus influenzae</italic>) and <italic>en-bloc</italic> glycosylated (in periplasm, ex. <italic>Campylobacter jejuni</italic>) proteins have been characterized in different organisms. Similarly, experimental data on O-glycosites is available only from four bacterial phyla namely, <italic>Actinobacteria</italic>, <italic>Bacteroidetes</italic>, <italic>Firmicutes</italic> and <italic>Proteobacteria</italic> out of eleven bacterial phyla where glycoproteins are known to exist. Interesting novel “conserved sequences of amino acids around glycosites (sequons)” like (D/E)X<sub>1</sub>NX(S/T)(where X<sub>1</sub> &amp; X≠P) for N-glycosylation and D(S/T)(A/I/L/V/M/T) for O-glycosylation have been elucidated within these glycoproteins that are not yet seen in eukaryotes <xref ref-type="bibr" rid="pone.0040155-Kowarik1">[23]</xref>, <xref ref-type="bibr" rid="pone.0040155-Fletcher1">[24]</xref>. A tool to map such sequons in amino acid sequence(s) of protein/proteomes has recently been made available by our group <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>. Further, an analysis of amino acid sequences surrounding N-glycosites of available archaeal glycoproteins (10 at that time) by Abu-Qarn and co-workers has also shown that archaeal N-glycosites are rarely surrounded by aromatic residues that are in abundance at positions –2 and –1 preceding glycosylated Asn at postion 0 in eukaryotic N-glycosites <xref ref-type="bibr" rid="pone.0040155-AbuQarn2">[25]</xref>–<xref ref-type="bibr" rid="pone.0040155-Petrescu1">[27]</xref>. For these reasons, the development of separate and new algorithms for prediction of glycosites in prokaryotes is of high interest and need <xref ref-type="bibr" rid="pone.0040155-Dell1">[20]</xref>.</p>
      <table-wrap id="pone-0040155-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t001</object-id><label>Table 1</label><caption>
          <title>An evaluation of performances of some of the well-known models for glycosylation prediction on prokaryotic glycoproteins.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t001" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Type of Glycosylation</td>
              <td align="left" colspan="1" rowspan="1">Prediction Tools</td>
              <td align="left" colspan="1" rowspan="1">Threshold</td>
              <td align="left" colspan="1" rowspan="1">Sensitivity (%)</td>
              <td align="left" colspan="1" rowspan="1">Specificity (%)</td>
              <td align="left" colspan="1" rowspan="1">Accuracy (%)</td>
              <td align="left" colspan="1" rowspan="1">MCC (%)</td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="1" rowspan="1">N-linked</td>
              <td align="left" colspan="1" rowspan="1">NetNglyc<sup>1</sup></td>
              <td align="left" colspan="1" rowspan="1">0.5</td>
              <td align="left" colspan="1" rowspan="1">81.75</td>
              <td align="left" colspan="1" rowspan="1">10.16</td>
              <td align="left" colspan="1" rowspan="1">34.41</td>
              <td align="left" colspan="1" rowspan="1">−0.11</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.6</td>
              <td align="left" colspan="1" rowspan="1">50.79</td>
              <td align="left" colspan="1" rowspan="1">42.68</td>
              <td align="left" colspan="1" rowspan="1">45.43</td>
              <td align="left" colspan="1" rowspan="1">−0.06</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.7</td>
              <td align="left" colspan="1" rowspan="1">15.87</td>
              <td align="left" colspan="1" rowspan="1">76.02</td>
              <td align="left" colspan="1" rowspan="1">55.65</td>
              <td align="left" colspan="1" rowspan="1">−0.09</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.9</td>
              <td align="left" colspan="1" rowspan="1">0.79</td>
              <td align="left" colspan="1" rowspan="1">98.37</td>
              <td align="left" colspan="1" rowspan="1">65.32</td>
              <td align="left" colspan="1" rowspan="1">−0.03</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">EnsembleGly<sup>3</sup></td>
              <td align="left" colspan="1" rowspan="1">0.3</td>
              <td align="left" colspan="1" rowspan="1">92.86</td>
              <td align="left" colspan="1" rowspan="1">0.41</td>
              <td align="left" colspan="1" rowspan="1">31.72</td>
              <td align="left" colspan="1" rowspan="1">−0.2</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.5</td>
              <td align="left" colspan="1" rowspan="1">90.48</td>
              <td align="left" colspan="1" rowspan="1">1.63</td>
              <td align="left" colspan="1" rowspan="1">31.72</td>
              <td align="left" colspan="1" rowspan="1">−0.18</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.7</td>
              <td align="left" colspan="1" rowspan="1">79.37</td>
              <td align="left" colspan="1" rowspan="1">10.98</td>
              <td align="left" colspan="1" rowspan="1">34.14</td>
              <td align="left" colspan="1" rowspan="1">−0.13</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">0.9</td>
              <td align="left" colspan="1" rowspan="1">51.59</td>
              <td align="left" colspan="1" rowspan="1">47.15</td>
              <td align="left" colspan="1" rowspan="1">48.66</td>
              <td align="left" colspan="1" rowspan="1">−0.01</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">O-linked</td>
              <td align="left" colspan="1" rowspan="1">NetOglyc<sup>2</sup></td>
              <td align="left" colspan="1" rowspan="1">0</td>
              <td align="left" colspan="1" rowspan="1">8.38</td>
              <td align="left" colspan="1" rowspan="1">95.64</td>
              <td align="left" colspan="1" rowspan="1">87.96</td>
              <td align="left" colspan="1" rowspan="1">0.05</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1"/>
              <td align="left" colspan="1" rowspan="1">EnsembleGly<sup>3</sup></td>
              <td align="left" colspan="1" rowspan="1">0</td>
              <td align="left" colspan="1" rowspan="1">9.5</td>
              <td align="left" colspan="1" rowspan="1">93.37</td>
              <td align="left" colspan="1" rowspan="1">86</td>
              <td align="left" colspan="1" rowspan="1">0.03</td>
            </tr>
          </tbody>
        </table></alternatives><table-wrap-foot>
          <fn id="nt101">
            <label/>
            <p>Footnotes: (1: <ext-link ext-link-type="uri" xlink:href="http://www.cbs.dtu.dk/services/NetNGlyc/" xlink:type="simple">http://www.cbs.dtu.dk/services/NetNGlyc/</ext-link>, 2: <ext-link ext-link-type="uri" xlink:href="http://www.cbs.dtu.dk/services/NetOGlyc-3.0/" xlink:type="simple">http://www.cbs.dtu.dk/services/NetOGlyc-3.0/</ext-link>, 3: <ext-link ext-link-type="uri" xlink:href="http://turing.cs.iastate.edu/EnsembleGly/" xlink:type="simple">http://turing.cs.iastate.edu/EnsembleGly/</ext-link>).</p>
          </fn>
        </table-wrap-foot></table-wrap>
      <table-wrap id="pone-0040155-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t002</object-id><label>Table 2</label><caption>
          <title>Experimentally characterized glycan linkages at known glycosites of bacteria and archaea.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t002" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Sugar linkage</td>
              <td align="left" colspan="1" rowspan="1">Class</td>
              <td align="left" colspan="1" rowspan="1">Example glycoproteins</td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="3" rowspan="1">
                <bold>N-LINKED GLYCANS/ARCHAEA</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Glc-Asn</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Halobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Flagellin, Slg</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">βGalNAc- Asn,</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Halobacteria, Methanococci, Methanobacteria Thermoprotei</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Flagellin, Slg, Cytochrome subunit</td>
            </tr>
            <tr>
              <td align="left" colspan="3" rowspan="1">
                <bold>N-LINKED GLYCANS/BACTERIA</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Bac-Asn</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Epsilonproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">AcrA, PEB3, CgpA, HisJ, ZnuA, jlpA etc.</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">GlcNAc-Asn</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Deltaproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">HmcA</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Hexose-Asn, dihexose-Asn, Glu-Asn,Gal-Asn</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Gammaproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Adhesins</td>
            </tr>
            <tr>
              <td align="left" colspan="3" rowspan="1">
                <bold>O-LINKED GLYCANS/BACTERIA</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Man-Ser/Thr</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Actinobacteria, Flavobacteria, Sphingobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Glycosidases, Cell surface lipoproteins,Secreted antigens, Superoxide dismutase, Heparinase, Chondroitinase etc.</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Fucose</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Bacteroidia</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Putative cell division proteins, exported proteins, outer membrane proteins etc.</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">β-GalNAc-Ser/Thr</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Bacilli</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Slg</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">β-D-Gal-Ser/Thr</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Bacilli</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Slg, SgsE, SgtA etc.</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">β-GlcNAc-Ser/Thr, HexNAc</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Bacilli</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Glycocin F, Flagellin</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Bac/DATDH-Ser</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Betaproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Pilin, CcoP, CycB etc.</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">FucNAc-Ser</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Gammaproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Pilin</td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Rha-Ser/Thr, Deoxyhexose-Ser</td>
              <td align="left" colspan="1" rowspan="1">
                <italic>Gammaproteobacteria</italic>
              </td>
              <td align="left" colspan="1" rowspan="1">Flagellin</td>
            </tr>
          </tbody>
        </table></alternatives><table-wrap-foot>
          <fn id="nt102">
            <label/>
            <p>Footnotes: Detailed information about attached glycan and glycoproteins can be obtained from <ext-link ext-link-type="uri" xlink:href="http://www.proglycprot.org" xlink:type="simple">www.proglycprot.org</ext-link>).</p>
          </fn>
        </table-wrap-foot></table-wrap>
      <p>Therefore in this study, we have attempted to analyze the sequence context, predicted secondary structure and surface accessibility of the experimentally verified glycosites in the largest available dataset of 107 N-linked glycosylated-residues (N-glycosites) and 116 O-linked glycosylated-residues (O-glycosites) from 59 prokaryotic glycoproteins retrieved from our recently published database of experimentally characterized prokaryotic glycoproteins, ProGlycProt <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>. In this study, we have developed a number of SVM models using three types of features namely, binary profile of patterns (BPP), composition profile of patterns (CPP), and PSI-BLAST generated PSSM profile of patterns (PPP) to recognize and differentiate glycosylated sequence contexts from non-glycosylated contexts in prokaryotic glycoproteins. For the reasons that mere presence of a consensus-sequon/pattern may not always be sufficient for glycosylation to occur and that the glycosites are predominantly situated on loops/accessible portions of folded proteins in prokaryotes, we have employed predicted secondary structure and surface accessibility features in combination with BPP, CPP and PPP for developing hybrid models (<xref ref-type="table" rid="pone-0040155-t003">Table 3</xref>), <xref ref-type="bibr" rid="pone.0040155-Nothaft1">[21]</xref>, <xref ref-type="bibr" rid="pone.0040155-Kowarik2">[28]</xref>. The best performing and significantly accurate models were then evaluated against an independent dataset of experimentally validated glycosites and finally implemented via web server GlycoPP (<xref ref-type="fig" rid="pone-0040155-g001">Figure 1</xref>) made available through open access at <ext-link ext-link-type="uri" xlink:href="http:/www.imtech.res.in/raghava/glycopp/" xlink:type="simple">http:/www.imtech.res.in/raghava/glycopp/</ext-link>.</p>
      <fig id="pone-0040155-g001" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0040155.g001</object-id>
        <label>Figure 1</label>
        <caption>
          <title>GlycoPP websever Schema.</title>
          <p>A flowchart of methodologies employed for development of GlycoPP webserver for prediction of N &amp; O-glycosites in prokaryotic protein sequences.</p>
        </caption>
        <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.g001" xlink:type="simple"/>
      </fig>
      <table-wrap id="pone-0040155-t003" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t003</object-id><label>Table 3</label><caption>
          <title>An analysis of experimentally observed secondary structures in prokaryotic glycosites.</title>
        </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t003-3" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t003" xlink:type="simple"/><table>
          <colgroup span="1">
            <col align="left" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
            <col align="center" span="1"/>
          </colgroup>
          <thead>
            <tr>
              <td align="left" colspan="1" rowspan="1">Protein Name (Source organism)</td>
              <td align="left" colspan="1" rowspan="1">PDB ID</td>
              <td align="left" colspan="1" rowspan="1">Presence of glycanin structure</td>
              <td align="left" colspan="1" rowspan="1">Validated Glycositesin full length protein sequence</td>
              <td align="left" colspan="1" rowspan="1">Position of Glycositesin PDB entry sequence</td>
              <td align="left" colspan="1" rowspan="1">SS</td>
            </tr>
          </thead>
          <tbody>
            <tr>
              <td align="left" colspan="6" rowspan="1">
                <bold>N-Glycosylated Proteins</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Tetrabrachion(Staphylothermus marinus)</td>
              <td align="left" colspan="1" rowspan="1">1YBK,1FE6</td>
              <td align="left" colspan="1" rowspan="1">–</td>
              <td align="left" colspan="1" rowspan="1">N44, N605, N641,N685, N708, N1279,N1402</td>
              <td align="left" colspan="1" rowspan="1">N44 (N1279*)</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>H<sup>1</sup></bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Chondroitinase ABC(Proteus vulgaris)</td>
              <td align="left" colspan="1" rowspan="1">1HN0</td>
              <td align="left" colspan="1" rowspan="1">–</td>
              <td align="left" colspan="1" rowspan="1">N282, N338, N345,N515, N675, N856,N963</td>
              <td align="left" colspan="1" rowspan="1">N282, N963 &amp; N675N338, N345 &amp; N515N856</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup> H<sup>1</sup> B<sup>1</sup></bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">PotD (Escherichia coli)</td>
              <td align="left" colspan="1" rowspan="1">1POT,1POY</td>
              <td align="left" colspan="1" rowspan="1">–</td>
              <td align="left" colspan="1" rowspan="1">N26, N62</td>
              <td align="left" colspan="1" rowspan="1">N26N62</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>3</sup> FR<sup>1</sup> (at beginning of helix)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">AcrA (Campylobacter jejuni)</td>
              <td align="left" colspan="1" rowspan="1">2K32,2K33(NMR)</td>
              <td align="left" colspan="1" rowspan="1">Heptasaccharide</td>
              <td align="left" colspan="1" rowspan="1">N123, N273</td>
              <td align="left" colspan="1" rowspan="1">N42 (N123*)</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup></bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">PEB3 (Campylobacter jejuni)</td>
              <td align="left" colspan="1" rowspan="1">2HXW</td>
              <td align="left" colspan="1" rowspan="1">-</td>
              <td align="left" colspan="1" rowspan="1">N90</td>
              <td align="left" colspan="1" rowspan="1">N90</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup> (between helices)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">HmcA (Desulfovibrio gigas)</td>
              <td align="left" colspan="1" rowspan="1">1Z1N</td>
              <td align="left" colspan="1" rowspan="1">Trisaccharide (NAG,NAA,any epimer of NAG),</td>
              <td align="left" colspan="1" rowspan="1">N290</td>
              <td align="left" colspan="1" rowspan="1">N261</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>2</sup> (between beta-sheets)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="6" rowspan="1">
                <bold>O-Glycosylated Proteins</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Chondroitinase-AC(Pedobacter heparinus)</td>
              <td align="left" colspan="1" rowspan="1">1CB8,1HM2,1HM3,1HMU,1HMW</td>
              <td align="left" colspan="1" rowspan="1">Tetrasaccharide Man-(Rha)-GlcUA-Xyl,</td>
              <td align="left" colspan="1" rowspan="1">S328, S455</td>
              <td align="left" colspan="1" rowspan="1">S328 S455</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup> (just after helix)</bold>
                <bold>FR<sup>1</sup> (between beta-sheets)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Chondroitinase-B(Pedobacter heparinus)</td>
              <td align="left" colspan="1" rowspan="1">1DBG,1DBO,1OFL,1OFM</td>
              <td align="left" colspan="1" rowspan="1">Heptasaccharide galactose-β(1–4)[galactose-α(1–3)](2-O-Me)fucose-β(1–4)xylose-β (1–4)glucuronicacid-α(1–2)[rhamnose-α(1–4)]mannose-α(1-</td>
              <td align="left" colspan="1" rowspan="1">S234</td>
              <td align="left" colspan="1" rowspan="1">S234</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup> (between beta-strands)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Heparinase II(Pedobacter heparinus)</td>
              <td align="left" colspan="1" rowspan="1">2FUQ,2FUT</td>
              <td align="left" colspan="1" rowspan="1">Tetrasaccharide Man-(Rha)-GlcUA-Xyl (xylose-β(1–4)glucuronic acid-α(1–2)[rhamnose-α(1–4)]mannose-α(1-</td>
              <td align="left" colspan="1" rowspan="1">T134</td>
              <td align="left" colspan="1" rowspan="1">T134</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>H<sup>3</sup></bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Fimbrial protein(Neisseria gonorrhoeae)</td>
              <td align="left" colspan="1" rowspan="1">2HI2,2HIL,2PIL,1AY2</td>
              <td align="left" colspan="1" rowspan="1">Disaccharides α-D-galactopyranosyl-(1→3)-2,4-diacetamido-2,4-dideoxy-β-D-glucopyranoside(bacillosamine, Bac);Gal-DADDGlc; andGlcNAc-α1,3-Gal</td>
              <td align="left" colspan="1" rowspan="1">S70</td>
              <td align="left" colspan="1" rowspan="1">S63</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>H<sup>1</sup> (before helix)</bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Glycocin F(Lactobacillus plantarum)</td>
              <td align="left" colspan="1" rowspan="1">2KUY(NMR)</td>
              <td align="left" colspan="1" rowspan="1">Two N-Acetylglucosamines</td>
              <td align="left" colspan="1" rowspan="1">S39</td>
              <td align="left" colspan="1" rowspan="1">S18</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>3</sup></bold>
              </td>
            </tr>
            <tr>
              <td align="left" colspan="1" rowspan="1">Endo-β-N-acetylglucosaminidase F3 (Flavobacterium meningosepticum)</td>
              <td align="left" colspan="1" rowspan="1">1EOM,1EOK</td>
              <td align="left" colspan="1" rowspan="1">-</td>
              <td align="left" colspan="1" rowspan="1">T88</td>
              <td align="left" colspan="1" rowspan="1">T49</td>
              <td align="left" colspan="1" rowspan="1">
                <bold>FR<sup>1</sup></bold>
              </td>
            </tr>
          </tbody>
        </table></alternatives><table-wrap-foot>
          <fn id="nt103">
            <label/>
            <p>Footnotes: All crystal structures are obtained from <ext-link ext-link-type="uri" xlink:href="http://www.rcsb.org" xlink:type="simple">www.rcsb.org</ext-link>. All structures are at a resolution of 1.4 Å or above.</p>
          </fn>
          <fn id="nt104">
            <label/>
            <p>Symbols used: - : No sugar detected, *: Corresponding position in full length protein sequence, F: flexible Regions with turns/loops/coils/bends or no assigned secondary structure, H: helix, B: beta sheet, 1: Intra domain, 2: Interdomain, 3: no assigned domain.</p>
          </fn>
        </table-wrap-foot></table-wrap>
    </sec>
    <sec id="s2" sec-type="methods">
      <title>Methods</title>
      <sec id="s2a">
        <title>Dataset Generation</title>
        <sec id="s2a1">
          <title>Source of data</title>
          <p>The primary set consisted of 39 N-linked and 54 O-linked glycoproteins obtained from the first release (July_2011) of ProGlycProt database <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>. For the reason that number of experimentally validated proteins is not very high, all the available N-linked and O-linked glycoprotein entries in primary dataset have been taken in to account for this study. However, entries containing only cysteine-linked (S-linked) glycosites as well as all glyco-engineered protein/peptides have been excluded from the primary set resulting into a total of 38 N-linked and 48 O-linked glycoproteins for further consideration. Some of these glycoproteins are N- as well as O-glycosylated. These glycoproteins include a variety of important proteins like S-layer proteins, flagellar proteins, pili/fimbrial proteins, lectins, adhesions, glycosidases, Cytochrome hemoprotein, heparinase, Chondroitinase as well as several known-unknown cytoplasmic, membrane bound and exported proteins (<xref ref-type="table" rid="pone-0040155-t002">Table 2</xref>). These glycoproteins represent all types of known N-glycosylation in prokaryotes representing organisms from phylum <italic>Crenarchaeota</italic> and <italic>Euryarchaeota</italic> of Archaea and phylum <italic>Proteobacteria</italic> of Bacteria. Similarly, this dataset represents all available validated examples of O-glycosylated proteins from four phyla namely, <italic>Actinobacteria</italic>, <italic>Bacteroidetes</italic>, <italic>Firmicutes</italic> and <italic>Proteobacteria</italic> of Bacteria. In Archaea no experimentally validated data exists for O-glycosites, so far. Further, within these glycoproteins, at least 30 N-linked and 40 O-linked glycoproteins have less than 40% sequence similarity to each other as deduced from CD-HIT v 4.0 available at <ext-link ext-link-type="uri" xlink:href="http://www.bioinformatics.org/cd-hit/" xlink:type="simple">http://www.bioinformatics.org/cd-hit/</ext-link>. From the primary dataset, 59 glycoproteins (16 archaeal and 43 bacterial) with higher number of characterized glycosites were hand- picked to form main datasets whereas remaining 27 (5 archaeal and 22 bacterial) glycoproteins were used as independent datasets of N and O glycosites, separately.</p>
        </sec>
        <sec id="s2a2">
          <title>Main datasets</title>
          <p>The Main datasets represent the training datasets employed in profile generation and later machine learning. The datasets contain 28 N-linked (overall sequence similarity less than 70%) and 31 O-linked (overall sequence similarity less than 90%) glycoproteins from prokaryotes. Using CD-HIT it has been deduced that in the main datasets, at least 23 N-linked and 26 O-linked glycoproteins have less than 40% sequence similarity to each other. From this set of glycoproteins, all N-and O-glycosites were retrieved and segregated in to separate datasets. All probable or predicted glycosites were excluded. Finally, the main datasets contained well-annotated unambiguous 107 N-linked and 116 O-linked glycosites derived from 59 experimentally validated prokaryotic glycoproteins. The O-linked glycosites (116) exclusively consisted of bacterial glycosites for unavailability of experimentally validated archaeal O-glycosite(s) <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>. To our knowledge, these are the <bold>most extensive</bold> datasets of experimentally validated prokaryotic glycosites (and glycoproteins), employed to develop <bold>first</bold> glycosite prediction models trained on and for prokaryotic protein sequences. These datasets are further divided in to two subgroups as follows.</p>
          <p><bold>Balanced datasets</bold> derived from randomly selecting all positive instances (positive training datasets) and equal number of negative instances (negative training datasets) across the protein lengths. Balanced datasets are useful in accelerating the machine learning and in avoiding biases in machine learning that are common in case of realistic dataset.</p>
          <p><bold>Realistic datasets</bold> contained all glycosylated/positive (107 N-linked &amp; 116 O-linked) and all non-glycosylated/negative (995 N-linked &amp; 2018 O-linked) sites from glycoprotein sequences. Performance of SVM on realistic (unbalanced) datasets could provide more confidence in predictions from real-time data where usually the non-glycosylated residues are much more than the glycosylated ones in a protein sequence.</p>
        </sec>
        <sec id="s2a3">
          <title>Independent datasets</title>
          <p>The independent balanced datasets of 28 (10 N-linked &amp; 17 O-linked) glycoproteins with experimentally validated 19 N-glycosites and 61 O-glycosites (with equivalent numbers of non-glycosylated sites) were used as test datasets in this study for evaluating the models trained on main datasets. Within these at least 7 N-linked and 14 O-linked glycoproteins show less than 40% sequence similarity to each other.</p>
        </sec>
      </sec>
      <sec id="s2b">
        <title>Pattern Generation and Feature Calculations</title>
        <p>Various overlapping symmetrical sequence patterns of residues length 21 that included central glycosylated residues were constructed according to previous studies <xref ref-type="bibr" rid="pone.0040155-Caragea1">[14]</xref>, <xref ref-type="bibr" rid="pone.0040155-Hamby1">[15]</xref>. A sequence pattern was considered positive if central residue was glycosylated otherwise the same was assigned as a negative pattern. To generate a pattern corresponding to the terminal residues in a protein sequence of length L, dummy residues “X” in number (L-1)/2 were added at both the termini of the protein <xref ref-type="bibr" rid="pone.0040155-Chauhan1">[29]</xref>, <xref ref-type="bibr" rid="pone.0040155-Agarwal1">[30]</xref>.</p>
        <sec id="s2b1">
          <title>Binary profile of patterns (BPP)</title>
          <p>Fixed length of 21 residues in sequence patterns was converted into binary form according to the existing study <xref ref-type="bibr" rid="pone.0040155-Agarwal1">[30]</xref>. Each residue of patterns was represented by a vector of dimension 21 (e.g. Ala by 1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0; Cys by 0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0), which contained 20 amino acids and one dummy amino acid “X”.</p>
        </sec>
        <sec id="s2b2">
          <title>Composition profile of patterns (CPP)</title>
          <p>Composition profile of patterns is the percentage frequencies of each amino acid in a fixed length sequence pattern. The fractions of all 20 natural amino acids of fixed length sequence patterns were calculated using the following equation <xref ref-type="bibr" rid="pone.0040155-Agarwal1">[30]</xref>:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e001" xlink:type="simple"/></disp-formula>Where <italic>Comp(i)</italic> is the percent composition of amino acid residue of type <italic>i</italic>; <italic>Ri</italic> is number of amino acid residues of type <italic>i</italic>, and <italic>N</italic> is the total number of residues in the fixed length sequence pattern.</p>
        </sec>
        <sec id="s2b3">
          <title>PSSM profile of patterns (PPP)</title>
          <p>In addition to compositional information, PSSM provides important information of evolutionary significance about residue conservation at a given position in a protein sequence. The multiple sequence alignment information in the form of position specific scoring matrix (PSSM) has been used here to develop learning model where each glycosylated protein sequence was first searched against ‘SWISS-PROT’ database followed by generation of alignment profiles or position specific scoring matrices (PSSM) using PSI-BLAST v 2.2.20 program (<ext-link ext-link-type="uri" xlink:href="ftp://ftp.ncbi.nlm.nih.gov/blast/executables/blast/LATEST/" xlink:type="simple">ftp://ftp.ncbi.nlm.nih.gov/blast/executables/blast/LATEST/</ext-link>). Three iterations of PSI-BLAST were run for each protein with cut off e-value 0.001. We have normalized each value range between 0 to 1 using sigmoid function by following equation, where <italic>val</italic> is the PSSM score and <italic>Val</italic> is its normalized value <xref ref-type="bibr" rid="pone.0040155-Chauhan1">[29]</xref>–<xref ref-type="bibr" rid="pone.0040155-Kumar1">[31]</xref>:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e002" xlink:type="simple"/></disp-formula></p>
        </sec>
        <sec id="s2b4">
          <title>Secondary structure information</title>
          <p>For this study, the secondary structure (SS) information (coil/helix/sheets) for glycosylated residue and its sequence context was obtained using webserver PSIPRED v 3.21 available at <ext-link ext-link-type="uri" xlink:href="http://bioinfadmin.cs.ucl.ac.uk/downloads/psipred/" xlink:type="simple">http://bioinfadmin.cs.ucl.ac.uk/downloads/psipred/</ext-link><xref ref-type="bibr" rid="pone.0040155-McGuffin1">[32]</xref>.</p>
        </sec>
        <sec id="s2b5">
          <title>Surface accessibility information</title>
          <p>The accessible surface area (ASA) is the surface area of a protein that is accessible to another protein or ligand(s). For our analysis, the average accessible surface area values of each amino acid were predicted from Sarpred available at <ext-link ext-link-type="uri" xlink:href="http://www.imtech.res.in/raghava/sarpred/" xlink:type="simple">www.imtech.res.in/raghava/sarpred/</ext-link><xref ref-type="bibr" rid="pone.0040155-Garg1">[33]</xref>.</p>
        </sec>
      </sec>
      <sec id="s2c">
        <title>Support Vector Machine (SVM) Algorithm and Evaluation Models</title>
        <p>The SVM is a supervised machine-learning technique based on the structural risk minimization principle <xref ref-type="bibr" rid="pone.0040155-Joachims1">[34]</xref>. In this study, we have used freely available SVM<sup>light</sup> classifier v 6.01 (<ext-link ext-link-type="uri" xlink:href="http://svmlight.joachims.org/" xlink:type="simple">http://svmlight.joachims.org/</ext-link>) where we could adjust the parameters and kernel (linear, polynomial, radial basis function, sigmoid) functions. The advantage of SVM over other machine learning techniques is that it can be trained on small dataset (as in this study) with minimum over-optimization. SVM based approach has been successfully employed in developing both N- and O-glycosylation prediction tools for mammalian glycoproteins in past <xref ref-type="bibr" rid="pone.0040155-Gupta1">[13]</xref>, <xref ref-type="bibr" rid="pone.0040155-Caragea1">[14]</xref>. Using different sequence properties like identity and position of residues (BPP), percentage composition of residues (CPP), residue conservation information (PSSM) along with structural features like secondary structure and surface accessibility several SVM classifiers have been trained and optimized for this study. Our group has successfully used one or more of these features in predicting GTP interacting residues, Mannose interacting residues, in predicting Cyclin protein sequences and in identification of conformational B-cell Epitopes from primary sequences of proteins, previously <xref ref-type="bibr" rid="pone.0040155-Chauhan1">[29]</xref>, <xref ref-type="bibr" rid="pone.0040155-Agarwal1">[30]</xref>, <xref ref-type="bibr" rid="pone.0040155-Kalita1">[35]</xref>, <xref ref-type="bibr" rid="pone.0040155-Ansari1">[36]</xref>. In this study, a 5-fold cross-validation procedure has been used to develop the prediction model, where five subsets were constructed randomly from the main datasets. At a given point of time, the models were trained on four sets of the training dataset and the performance was measured on the remaining fifth set. This process is repeated five times in such a way that each set was used once for testing. The final performance was obtained by averaging the performances of all five sets. The models thus obtained were evaluated for performance using threshold dependent parameters namely, sensitivity (Sn), Specificity (Sp), Accuracy (Acc), Matthews correlation coefficient (MCC) as well as using threshold dependent parameters Area Under Curve (AUC) values.</p>
        <p>Evaluation parameters employed in this study are described briefly as below:</p>
        <p><bold>Sensitivity</bold> is the percentage of glycosites that are correctly predicted as glycosylated:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e003" xlink:type="simple"/></disp-formula></p>
        <p><bold>Specificity</bold> is the percentage of non-glycosylated sites that are correctly predicted as non-glycosylated:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e004" xlink:type="simple"/></disp-formula></p>
        <p><bold>Accuracy</bold> is the percentage of correct prediction out of total number of predictions:<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e005" xlink:type="simple"/></disp-formula></p>
        <p><bold>Matthews correlation coefficient</bold> (MCC) is a measure of both sensitivity and specificity. MCC value would range from 0 (indicating completely random prediction) to 1 (indicating perfect prediction):<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.e006" xlink:type="simple"/></disp-formula>[Where TP- true positive; FN- false negative; TN- true negative; FP- false positive]</p>
        <p>Threshold selection is important criteria for checking the consistency of prediction results. In our study, we have varied threshold in the range of –1 to +1, normally we selected “0” as default threshold to achieve balance between sensitivity and specificity.</p>
        <p>Area Under Curve (AUC) a threshold independent parameter describes inherent trade-off between sensitivity and specificity. Receiver Operating Characteristic (ROC) plots were drawn between TP rate (sensitivity) and FP rate (1-specificity) using R-package v 2.14.1 (<ext-link ext-link-type="uri" xlink:href="http://www.r-project.org/" xlink:type="simple">http://www.r-project.org/</ext-link>) to calculate AUC values. Finally, the best performing models in terms of accuracy &amp; MCC values were validated using an independent dataset of prokaryotic glycoproteins for final implementation at GlycoPP webserver (<xref ref-type="fig" rid="pone-0040155-g001">Figure 1</xref>).</p>
      </sec>
    </sec>
    <sec id="s3">
      <title>Results</title>
      <sec id="s3a">
        <title>Prediction Performance of Some of the Existing Tools on Prokaryotic Glycoproteins</title>
        <p>In order to evaluate the suitability of models trained on eukaryotic glycoproteins for predicting glycosites in prokaryotic proteins, the proteins of main datasets were run on three of the well-known prediction tools for prediction of N- and O-glycosites. Against the experimentally validated glycoproteins of prokaryotes, the performances of these tools were found very poor and are detailed in <xref ref-type="table" rid="pone-0040155-t001">Table 1</xref>. From this, we conclude that the methods that are trained using eukaryotic glycoprotein are not optimum for prediction of potential glycosites in bacterial and archaeal proteins. This also suggests that the sequence or structural contexts around prokaryotic glycosites could be different from what is known in eukaryotic glycosites. This is logical as several OSTs with novel mechanisms of sugar transfer on to the acceptor proteins are now known in bacteria as well as archaea. This prompted us to develop a number of new algorithms to recognize and differentiate glycosylated and unglycosylated sequence contexts of known glycosites of archaeal and bacterial proteins representing aforementioned six different phyla. These algorithms are trained using different input features and described in this study.</p>
      </sec>
      <sec id="s3b">
        <title>Sequence Context of Prokaryotic Glycosites</title>
        <p>In an attempt to understand the general preferences for different amino acids around prokaryotic glycosites as well as the differences from the corresponding sequences in eukaryotes, we have generated a number of one sample and two sample weblogos (<ext-link ext-link-type="uri" xlink:href="http://weblogo.berkeley.edu/&amp;" xlink:type="simple">http://weblogo.berkeley.edu/&amp;</ext-link> <ext-link ext-link-type="uri" xlink:href="http://www.twosamplelogo.org/" xlink:type="simple">http://www.twosamplelogo.org/</ext-link>) for N- and O-glycosites of archaeal and bacterial glycoproteins in an organism specific, phylum specifc as well as domain specific manner, respectively. The interesting existing knowledge as well as our statistically significant observations for the purposes of a prediction model are discussed here, briefly. Similar to eukaryotic glycoproteins, the minimal sequon NX(S/T)(where X≠P) is essential for N-glycosylation in prokaryotic glycoproteins. For example in all archaeal glycoproteins (<xref ref-type="supplementary-material" rid="pone.0040155.s001">Figure S1</xref>), <xref ref-type="bibr" rid="pone.0040155-AbuQarn2">[25]</xref>, in HmcA protein of <italic>Desulfovibrio</italic> <xref ref-type="bibr" rid="pone.0040155-Ielmini1">[37]</xref>, adhesin protein HMW1 of <italic>Haemophilus influenzae</italic> and <italic>Actinobacillus pleuropneumoniae</italic> (where glycosylation is sequential and mediated by a novel cytoplasmic glycosyltransferase, HMW1C of family GT41, <xref ref-type="supplementary-material" rid="pone.0040155.s002">Figure S2</xref>), <xref ref-type="bibr" rid="pone.0040155-Choi1">[38]</xref>. However, as known already, the sequon is extended as (D/E)X<sub>1</sub>NX(S/T)(where X<sub>1</sub> &amp; X≠P) but not stringent in case of PglB (OST of <italic>Campylobacter</italic>) mediated <italic>en bloc</italic> N-glycosylation in <italic>Campylobacter</italic> and <italic>Helicobacter</italic> (<xref ref-type="supplementary-material" rid="pone.0040155.s002">Figure S2</xref>), <xref ref-type="bibr" rid="pone.0040155-Jervis1">[39]</xref>. As discussed before, the first-ever defined sequon D(S/T)(A/I/L/V/M/T) for O-glycosites is indeed conserved across available glycoproteins from three representative classes including <italic>Bacteroidia</italic>, <italic>Flavobacteria</italic> and <italic>Sphingobacteria</italic> of phylum <italic>Bacteroidetes</italic> (<xref ref-type="supplementary-material" rid="pone.0040155.s003">Figure S3</xref>). Further, the two-sample logos comparing prokaryotic and eukaryotic N-and O- glycosites clearly illustrate the differences in the amino acid preferences around these glycosites (<xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>), indicating a necessity for independent prediction tool for prokaryotes. With respect to glycosylated Asn (if at position 0) the positions at -1 and -2 have previously been stated to be enriched in aromatic amino acids in eukaryotic N-glycosites <xref ref-type="bibr" rid="pone.0040155-AbuQarn2">[25]</xref>–<xref ref-type="bibr" rid="pone.0040155-Petrescu1">[27]</xref>. However, in prokaryotic N-glycosites instead we observe a marked preference for polar residues like Asp/Glu/Thr/Asn and lysine at different positions preceding glycosylated Asn (Logo C, <xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>). Similarly, at positions -2 and -6 occurrence of polar residues is higher around NX(S/T) motif in validated N-glycosites of prokaryotes in contrast to randomly selected equal number of NX(S/T) motifs with unglycosylated Asn from prokaryotic glycoproteins (Logo A, <xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>). An analysis of eukaryotic N-glycosites by Pertescu and co-workers had suggested a preference for small hydrophobic residue at positions +1 and large hydrophobic residue at +3 in eukaryotes, previously <xref ref-type="bibr" rid="pone.0040155-Petrescu1">[27]</xref>. Similarly, in case of prokaryotic glycoproteins hydrophobic residues are though present at +1 position yet preference for large or small residues are not very clear (<xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s001">Figure S1</xref>), <xref ref-type="bibr" rid="pone.0040155-AbuQarn2">[25]</xref>. Furthermore, increased instances of Pro near the glycosylated residues are not observed in bacterial and archaeal glycoproteins as found in eukaryotic glycoproteins. Instead Pro is one of the significantly depleted amino acids at +4 and +5 positions here <xref ref-type="bibr" rid="pone.0040155-Petrescu1">[27]</xref>. Likewise, sequence surrounding all prokaryotic O-glycosites (<xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>) is different in having higher instances of Gly, Ala, Val and a significant depletion of Pro at almost all positions (except in mannosylated glycoproteins of <italic>Mycobacterium spp,</italic> <xref ref-type="supplementary-material" rid="pone.0040155.s002">Figure S2</xref>), <xref ref-type="bibr" rid="pone.0040155-Dobos1">[8]</xref> in comparison to the eukaryotic mucin type O-glycosites that are rich in Ser, Ala and Pro (Logo D, <xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>), <xref ref-type="bibr" rid="pone.0040155-Julenius1">[12]</xref>. In prokaryotic O-glycosites, apart from this general presence of small hydrophobic amino acids around 10 residues on either sides of glycosylated Ser/Thr residues (at position 0), a marked preference for negatively charged Asp at -1 that in fact is a part of potential sequon for O-glycosites in <italic>Bacteroidetes</italic> is observed (Logo B, <xref ref-type="fig" rid="pone-0040155-g002">Figure 2</xref>).</p>
        <fig id="pone-0040155-g002" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pone.0040155.g002</object-id>
          <label>Figure 2</label>
          <caption>
            <title>Sequence contexts of prokaryotic glycosites.</title>
            <p>Two sample weblogos depicting enriched and depleted amino acids around prokaryotic N-glycosites (logo A) and prokaryotic O-glycosites (logo B) in comparison to the percentage of these amino acids around non-glycosylated prokaryotic N-glycosites and O-glycosites, respectively. Similarly, logos C and D provide an assessment of probabilities of amino acids around prokaryotic N- and O-glycosites in comparison to probabilities around eukaryotic N- and O-glycosites, respectively. The datasets for eukaryotic N- and O- glycosites for generation of weblogos is obtained from SWISS-PROT (2011 release).</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.g002" xlink:type="simple"/>
        </fig>
      </sec>
      <sec id="s3c">
        <title>Structural Features of Prokaryotic Glycosites</title>
        <p>Previous statistical analysis of all available crystal structures of eukaryotic glycoproteins by Petrescu et al. had suggested that the probability of finding N-glycosites was higher at positions where there was a secondary structure change <xref ref-type="bibr" rid="pone.0040155-Petrescu1">[27]</xref>. Upon analysis of 12 eukaryotic glycoproteins, Julenius et al had also concluded that O-glycosites mainly occurred in coil region of mucin type of O-glycosylated proteins <xref ref-type="bibr" rid="pone.0040155-Julenius1">[12]</xref>. The secondary structure and surface accessibility of a residue therefore are considered important criteria in prediction of glycosites in eukaryotes. Some of the existing eukaryotic glycosite prediction models have employed these features successfully <xref ref-type="bibr" rid="pone.0040155-Hansen1">[11]</xref>, <xref ref-type="bibr" rid="pone.0040155-Julenius1">[12]</xref>. Unlike eukaryotic N-glycosylation that is a co-translational event, the glycosylation is considered a true post-translational modification in bacteria where the folding state of a polypeptide/protein could dictate availability of a sequon/site for glycan attachment on to a protein <xref ref-type="bibr" rid="pone.0040155-Nothaft1">[21]</xref>. Although limited, yet most of the X-ray crystal structures and NMR structures of bacterial glycoproteins (as listed at <ext-link ext-link-type="uri" xlink:href="http://www.proglycprot.org/CrystalStructure.aspx" xlink:type="simple">http://www.proglycprot.org/CrystalStructure.aspx</ext-link>) show that the glycosylated residues are indeed primarily located in surface-exposed flexible loops/turns/bends that then should be accessible to bacterial OSTs/GTs. The structural contexts for 20 glycosites (13 N- &amp; 7 O-glycosites) extracted from available structures of N- and O-glycoproteins of Archaea and Bacteria, reveal that at least 65% (13 out of 20) of these glycosites are located in aforementioned flexible regions and primarily in intra-domain region (<xref ref-type="table" rid="pone-0040155-t003">Table 3</xref>). Incidentally, at least three of the N-glycosylated proteins namely, PotD of Escherichia coli, AcrA and PEB3 of <italic>Campylobacter jejuni</italic> are glycosylated (<italic>in vitro/in vivo</italic>) by OST of <italic>Campylobacter</italic> (PglB) that has previously been shown to transfer sugars post-translationally to locally flexible structures in folded proteins <xref ref-type="bibr" rid="pone.0040155-Nothaft1">[21]</xref>, <xref ref-type="bibr" rid="pone.0040155-Kowarik2">[28]</xref>, <xref ref-type="bibr" rid="pone.0040155-Rangarajan1">[40]</xref>. Similarly, glycosylated Ser/Thr residues in Endo-β-N-acetylglucosaminidase F3 (<italic>Flavobacterium meningosepticum</italic>), Chondroitinase-AC and Chondroitinase-B (<italic>Pedobacter heparinus</italic>) lie in the similar loops/bends in their respective crystal structures (<xref ref-type="table" rid="pone-0040155-t003">Table 3</xref>).</p>
        <p>Our analysis of the predicted secondary structure indicates that 55.92% of the validated glycosites are found in coils, 15.51% in helix and 28.57% in sheets whereas non-validated glycosites or their sequence contexts are found correspondingly less in coil (47.23%), more in helix (24.42%) and almost equally in sheets (28.35%). Similarly, 17.36% of validated O-glycosites are situated in helix, 62.63% in coils and and 20.4% in sheets in contrast to non-validated O-glycosites that are found more often in helix (22.99%) and less in coils (53.98) and almost equally in sheets (23.03), respectively (<xref ref-type="fig" rid="pone-0040155-g003">Figure 3</xref>). Similarly, predicted surface accessibility profile of glycosylated Asn residues suggest them to be much more surface accessible than the corresponding non-glycosylated sequence contexts as shown in <xref ref-type="fig" rid="pone-0040155-g004">Figure 4</xref>. The glycosylated Ser/Thr are again, more accessible (80%) compared to the non-glycosylated residues (60%). To summarize, most of the prokaryotic glycosites (both N as well as O) indeed seems to be present in flexible and exposed regions. Further, not only the central glycosylated-residues but also their surrounding residues are highly accessible and surface exposed.</p>
        <fig id="pone-0040155-g003" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pone.0040155.g003</object-id>
          <label>Figure 3</label>
          <caption>
            <title>Predicted secondary structures around prokaryotic glycosites.</title>
            <p>Average percentage of secondary structures predicted in and around N-glycosites (panel A) and O-glycosites (panel B) in prokaryotic glycoproteins. The graph indicates a general likelihood of locating a glycosylated residue in coils/turns in a protein.</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.g003" xlink:type="simple"/>
        </fig>
        <fig id="pone-0040155-g004" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pone.0040155.g004</object-id>
          <label>Figure 4</label>
          <caption>
            <title>Predicted Surface accessibility of prokaryotic glycosites.</title>
            <p>Average percentage of exposed and buried residues predicted in and around N-glycosites (panel A) and O-glycosites (panel B) in prokaryotic glycoproteins. The graph suggests higher accessibility of glycosylated residues on surface of a protein in comparison to non-glycosylated ones.</p>
          </caption>
          <graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.g004" xlink:type="simple"/>
        </fig>
      </sec>
      <sec id="s3d">
        <title>Prediction Performance of SVM Using Balanced Datasets</title>
        <p>SVM models based on BPP, CPP and PSSM profiles are well recognized for their notable performances in predicting a variety of motifs and interactions in biomolecules and have been used effectively in the past for glycosites predictions as well <xref ref-type="bibr" rid="pone.0040155-Chauhan1">[29]</xref>–<xref ref-type="bibr" rid="pone.0040155-Kumar1">[31]</xref>, <xref ref-type="bibr" rid="pone.0040155-Kalita1">[35]</xref>, <xref ref-type="bibr" rid="pone.0040155-Ansari1">[36]</xref>. Accordingly, we have generated several SVM models using BPP, CPP and PPP profiles as input features. The performance measures were calculated at different thresholds of SVM scores ranging from −1.0 to 1.0 and the best performing thresholds were selected for further optimization. The prediction of the N- glycosites were best achieved by the SVM models developed using BPP profile achieving 79.91% accuracy and 0.60 MCC (<xref ref-type="table" rid="pone-0040155-t004">Table 4</xref>) whereas O-glycosites were best predicted by SVM model developed using PPP with 74.57% accuracy and 0.49 MCC (<xref ref-type="table" rid="pone-0040155-t005">Table 5</xref>). As discussed before, sequence features namely secondary structure (SS) and accessible surface area (ASA) could play an important role in correct predictions of sites of glycoslation in a protein. Therefore, we have developed prediction models with these features in following three combinations: (i) composition profile of patterns with either secondary structure or surface accessibility or both (ii) Binary profile of patterns with either secondary structure or surface accessibility or both (iii) PPP with either secondary structure or surface accessibility or both. In general, inclusion of SS and SAS profiles in prediction models helped improvise predictions (<xref ref-type="table" rid="pone-0040155-t004">Table 4</xref>, <xref ref-type="table" rid="pone-0040155-t005">Table 5</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s004">Figure S4</xref>). The hybrid model of BPP+ASA proved as good as BPP+SS+ASA improving the maximum MCC of prediction from 0.60 to 0.65 and accuracy of prediction from 79.91% to 82.24% for N-glycosites (<xref ref-type="table" rid="pone-0040155-t004">Table 4</xref>). Similarly, predictions of O-glycosites were improvised slightly using the hybrid model (based on combination of PPP+SS+ASA profiles) giving MCC of 0.48 and accuracy value of 71.73% in comparison with PPP alone derived 73.28% accuracy and 0.47 MCC (<xref ref-type="table" rid="pone-0040155-t005">Table 5</xref>).</p>
        <table-wrap id="pone-0040155-t004" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t004</object-id><label>Table 4</label><caption>
            <title>Combined performance statistics of SVM employing solo features and hybrid approaches in predicting N-glycosites (using balanced dataset).</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t004-4" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t004" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1">Feature</td>
                <td align="left" colspan="1" rowspan="1">Sensitivity (%)</td>
                <td align="left" colspan="1" rowspan="1">Specificity (%)</td>
                <td align="left" colspan="1" rowspan="1">Accuracy (%)</td>
                <td align="left" colspan="1" rowspan="1">MCC (%)</td>
                <td align="left" colspan="1" rowspan="1">AUC (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP</td>
                <td align="left" colspan="1" rowspan="1">59.81</td>
                <td align="left" colspan="1" rowspan="1">64.49</td>
                <td align="left" colspan="1" rowspan="1">62.15</td>
                <td align="left" colspan="1" rowspan="1">0.24</td>
                <td align="left" colspan="1" rowspan="1">0.65019</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+SS</td>
                <td align="left" colspan="1" rowspan="1">63.55</td>
                <td align="left" colspan="1" rowspan="1">69.16</td>
                <td align="left" colspan="1" rowspan="1">66.36</td>
                <td align="left" colspan="1" rowspan="1">0.33</td>
                <td align="left" colspan="1" rowspan="1">0.68731</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+ASA</td>
                <td align="left" colspan="1" rowspan="1">71.03</td>
                <td align="left" colspan="1" rowspan="1">69.16</td>
                <td align="left" colspan="1" rowspan="1">70.09</td>
                <td align="left" colspan="1" rowspan="1">0.40</td>
                <td align="left" colspan="1" rowspan="1">0.77203</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+SS+ASA</td>
                <td align="left" colspan="1" rowspan="1">70.09</td>
                <td align="left" colspan="1" rowspan="1">67.29</td>
                <td align="left" colspan="1" rowspan="1">68.69</td>
                <td align="left" colspan="1" rowspan="1">0.37</td>
                <td align="left" colspan="1" rowspan="1">0.71159</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <bold>BPP</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>79.44</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>80.37</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>79.91</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.60</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.88322</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP+SS</td>
                <td align="left" colspan="1" rowspan="1">82.24</td>
                <td align="left" colspan="1" rowspan="1">80.37</td>
                <td align="left" colspan="1" rowspan="1">81.31</td>
                <td align="left" colspan="1" rowspan="1">0.63</td>
                <td align="left" colspan="1" rowspan="1">0.88453</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <bold>BPP+ASA</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>84.11</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>81.31</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>82.71</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.65</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.89807</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP+SS+ASA</td>
                <td align="left" colspan="1" rowspan="1">84.11</td>
                <td align="left" colspan="1" rowspan="1">80.37</td>
                <td align="left" colspan="1" rowspan="1">82.24</td>
                <td align="left" colspan="1" rowspan="1">0.65</td>
                <td align="left" colspan="1" rowspan="1">0.88497</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP</td>
                <td align="left" colspan="1" rowspan="1">76.42</td>
                <td align="left" colspan="1" rowspan="1">69.81</td>
                <td align="left" colspan="1" rowspan="1">73.11</td>
                <td align="left" colspan="1" rowspan="1">0.46</td>
                <td align="left" colspan="1" rowspan="1">0.76833</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP+SS</td>
                <td align="left" colspan="1" rowspan="1">75.70</td>
                <td align="left" colspan="1" rowspan="1">71.03</td>
                <td align="left" colspan="1" rowspan="1">73.36</td>
                <td align="left" colspan="1" rowspan="1">0.47</td>
                <td align="left" colspan="1" rowspan="1">0.78880</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP+ASA</td>
                <td align="left" colspan="1" rowspan="1">77.57</td>
                <td align="left" colspan="1" rowspan="1">71.03</td>
                <td align="left" colspan="1" rowspan="1">74.30</td>
                <td align="left" colspan="1" rowspan="1">0.49</td>
                <td align="left" colspan="1" rowspan="1">0.78636</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP+SS+ASA</td>
                <td align="left" colspan="1" rowspan="1">75.70</td>
                <td align="left" colspan="1" rowspan="1">71.96</td>
                <td align="left" colspan="1" rowspan="1">73.83</td>
                <td align="left" colspan="1" rowspan="1">0.48</td>
                <td align="left" colspan="1" rowspan="1">0.79334</td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt105">
              <label/>
              <p>Footnotes: BPP- Binary profile of patterns, CPP- Composition profile of patterns, PPP- PSSM profile of patterns, MCC- Matthews correlation coefficient, AUC- Area under curve, SS-secondary structure and ASA- Accessible surface area.</p>
            </fn>
          </table-wrap-foot></table-wrap>
        <table-wrap id="pone-0040155-t005" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t005</object-id><label>Table 5</label><caption>
            <title>Combined performance statistics of SVM classifiers employing solo features and hybrid approaches in predicting O-glycosites (using balanced dataset).</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t005-5" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t005" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="1" rowspan="1">Feature</td>
                <td align="left" colspan="1" rowspan="1">Sensitivity (%)</td>
                <td align="left" colspan="1" rowspan="1">Specificity (%)</td>
                <td align="left" colspan="1" rowspan="1">Accuracy (%)</td>
                <td align="left" colspan="1" rowspan="1">MCC (%)</td>
                <td align="left" colspan="1" rowspan="1">AUC (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP</td>
                <td align="left" colspan="1" rowspan="1">68.10</td>
                <td align="left" colspan="1" rowspan="1">72.41</td>
                <td align="left" colspan="1" rowspan="1">70.26</td>
                <td align="left" colspan="1" rowspan="1">0.41</td>
                <td align="left" colspan="1" rowspan="1">0.74071</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+SS</td>
                <td align="left" colspan="1" rowspan="1">70.69</td>
                <td align="left" colspan="1" rowspan="1">71.55</td>
                <td align="left" colspan="1" rowspan="1">71.12</td>
                <td align="left" colspan="1" rowspan="1">0.42</td>
                <td align="left" colspan="1" rowspan="1">0.75743</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+ASA</td>
                <td align="left" colspan="1" rowspan="1">67.24</td>
                <td align="left" colspan="1" rowspan="1">75.00</td>
                <td align="left" colspan="1" rowspan="1">71.12</td>
                <td align="left" colspan="1" rowspan="1">0.42</td>
                <td align="left" colspan="1" rowspan="1">0.75780</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">CPP+SS+ASA</td>
                <td align="left" colspan="1" rowspan="1">72.41</td>
                <td align="left" colspan="1" rowspan="1">75.00</td>
                <td align="left" colspan="1" rowspan="1">73.71</td>
                <td align="left" colspan="1" rowspan="1">0.47</td>
                <td align="left" colspan="1" rowspan="1">0.76955</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP</td>
                <td align="left" colspan="1" rowspan="1">66.38</td>
                <td align="left" colspan="1" rowspan="1">67.24</td>
                <td align="left" colspan="1" rowspan="1">66.81</td>
                <td align="left" colspan="1" rowspan="1">0.34</td>
                <td align="left" colspan="1" rowspan="1">0.73023</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP+SS</td>
                <td align="left" colspan="1" rowspan="1">69.83</td>
                <td align="left" colspan="1" rowspan="1">68.10</td>
                <td align="left" colspan="1" rowspan="1">68.97</td>
                <td align="left" colspan="1" rowspan="1">0.38</td>
                <td align="left" colspan="1" rowspan="1">0.74160</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP+ASA</td>
                <td align="left" colspan="1" rowspan="1">77.59</td>
                <td align="left" colspan="1" rowspan="1">61.21</td>
                <td align="left" colspan="1" rowspan="1">69.40</td>
                <td align="left" colspan="1" rowspan="1">0.39</td>
                <td align="left" colspan="1" rowspan="1">0.71143</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">BPP+SS+ASA</td>
                <td align="left" colspan="1" rowspan="1">65.52</td>
                <td align="left" colspan="1" rowspan="1">72.41</td>
                <td align="left" colspan="1" rowspan="1">68.97</td>
                <td align="left" colspan="1" rowspan="1">0.38</td>
                <td align="left" colspan="1" rowspan="1">0.73766</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <bold>PPP</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>75.00</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>71.55</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>73.28</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.47</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.81250</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP+SS</td>
                <td align="left" colspan="1" rowspan="1">73.28</td>
                <td align="left" colspan="1" rowspan="1">73.28</td>
                <td align="left" colspan="1" rowspan="1">73.28</td>
                <td align="left" colspan="1" rowspan="1">0.47</td>
                <td align="left" colspan="1" rowspan="1">0.76806</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">PPP+ASA</td>
                <td align="left" colspan="1" rowspan="1">74.14</td>
                <td align="left" colspan="1" rowspan="1">71.55</td>
                <td align="left" colspan="1" rowspan="1">72.84</td>
                <td align="left" colspan="1" rowspan="1">0.46</td>
                <td align="left" colspan="1" rowspan="1">0.77341</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <bold>PPP+SS+ASA</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>77.59</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>69.83</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>73.71</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.48</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.76925</bold>
                </td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt106">
              <label/>
              <p>Footnotes: BPP- Binary profile of patterns, CPP- Composition profile of patterns, PPP- PSSM profile of patterns, MCC- Matthews correlation coefficient, AUC- Area under curve, SS-secondary structure and ASA- Accessible surface area.</p>
            </fn>
          </table-wrap-foot></table-wrap>
      </sec>
      <sec id="s3e">
        <title>Prediction Performance of SVM Using Realistic Datasets</title>
        <p>For any machine learning technique, learning of datasets is very easy when both positive and negative instances are equal in number. Nevertheless, in case of glycoproteins, the negative instances could be much more than the positive instances in a protein sequence. Therefore, in order to judge accuracy and applicability of our SVM prediction schemes on realistic datasets of users, in parallel we have calculated the performances of aforementioned SVM models using realistic datasets. As was seen in case of models optimized with balanced datasets, BPP based SVM models performed better in case of realistic datasets and could achieve a maximum MCC 0.48 and 0.51 with accuracy value of 82.03% and 86.39% for prediction of N-glycosites using solo feature based and hybrid models (BPP+ASA), respectively. Similarly, O-glycosites could also be predicted with reasonably high accuracy of 70.24% and 89.69% (with corresponding maximum MCC values of 0.19 and 0.50) using CPP and CPP+ASA based models, respectively. Surprisingly, while using realistic datasets, predictions for O-glycosites were better with CPP based models in contrast to PPP based models that fared well in case of balanced datasets (<xref ref-type="table" rid="pone-0040155-t005">Table 5</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s005">Table S1</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s004">Figure S4</xref>). Infact inclusion of surface accessibility features in combination with CPP in O-glycosites prediction scheme could enhance maximum MCC value of prediction by 2.5 fold (<xref ref-type="supplementary-material" rid="pone.0040155.s005">Table S1</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s004">Figure S4</xref>) indicating that surface accessibility alone could indeed be a useful criterion in glycosites prediction models discussed here. Further, the observed poorer performance of SVM with realistic datasets than with the balanced datasets of course is due to the inherent learning biases of realistic datasets. However, overall our SVM models optimized with realistic datasets fared reasonably well in predicting both N- and O- glycosites from realistic datasets (<xref ref-type="supplementary-material" rid="pone.0040155.s005">Table S1</xref>).</p>
      </sec>
      <sec id="s3f">
        <title>Prediction Performance on Independent Datasets</title>
        <p>Finally, the performance of best-optimized models as discussed before (<xref ref-type="table" rid="pone-0040155-t004">Table 4</xref>, <xref ref-type="table" rid="pone-0040155-t005">Table 5</xref>) were evaluated and compared with performances of NetOGlyc v 3.0, NetNGlyc, EnsembleGly against an independent set of experimentally verified prokaryotic glycoproteins (<xref ref-type="table" rid="pone-0040155-t006">Table 6</xref>). Models developed and discussed in this study not only provided reasonably high accuracy (86.84% for N-glycosites with maximum MCC of 0.74 and 76.23% for O-glycosites with maximum MCC value of 0.53, respectively) but have convincingly outperformed performances of at least three of the well-known existing glycosites prediction tools as detailed in <xref ref-type="table" rid="pone-0040155-t006">Table 6</xref> in the context of prokaryotic glycosites prediction.</p>
        <table-wrap id="pone-0040155-t006" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0040155.t006</object-id><label>Table 6</label><caption>
            <title>Comparative performances of existing well-known glycosylation prediction tools and GlycoPP models on independent dataset of prokaryotic glycoproteins.</title>
          </caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0040155-t006-6" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.t006" xlink:type="simple"/><table>
            <colgroup span="1">
              <col align="left" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
              <col align="center" span="1"/>
            </colgroup>
            <thead>
              <tr>
                <td align="left" colspan="7" rowspan="1">Prediction of N-glycosites</td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Models (Threshold)</td>
                <td align="left" colspan="1" rowspan="1">NetNglyc<sup>1</sup> (0.5)</td>
                <td align="left" colspan="1" rowspan="1">EnsembleGly<sup>3</sup> (0.7)</td>
                <td align="left" colspan="1" rowspan="1">GlycoPP-BPP (−0.1)</td>
                <td align="left" colspan="1" rowspan="1">GlycoPP-CPP (0.3)</td>
                <td align="left" colspan="1" rowspan="1">GlycoPP-PPP (−0.2)</td>
                <td align="left" colspan="1" rowspan="1">GlycoPP-BPP+ASA</td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td align="left" colspan="1" rowspan="1">Sensitivity (%)</td>
                <td align="left" colspan="1" rowspan="1">88.89</td>
                <td align="left" colspan="1" rowspan="1">94.44</td>
                <td align="left" colspan="1" rowspan="1">89.47</td>
                <td align="left" colspan="1" rowspan="1">68.42</td>
                <td align="left" colspan="1" rowspan="1">78.95</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>89.47</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Specificity (%)</td>
                <td align="left" colspan="1" rowspan="1">25.00</td>
                <td align="left" colspan="1" rowspan="1">11.36</td>
                <td align="left" colspan="1" rowspan="1">73.68</td>
                <td align="left" colspan="1" rowspan="1">73.68</td>
                <td align="left" colspan="1" rowspan="1">73.68</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>84.21</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Accuracy (%)</td>
                <td align="left" colspan="1" rowspan="1">43.55</td>
                <td align="left" colspan="1" rowspan="1">35.48</td>
                <td align="left" colspan="1" rowspan="1">81.58</td>
                <td align="left" colspan="1" rowspan="1">71.05</td>
                <td align="left" colspan="1" rowspan="1">76.32</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>86.84</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">MCC (%)</td>
                <td align="left" colspan="1" rowspan="1">0.15</td>
                <td align="left" colspan="1" rowspan="1">0.09</td>
                <td align="left" colspan="1" rowspan="1">0.64</td>
                <td align="left" colspan="1" rowspan="1">0.42</td>
                <td align="left" colspan="1" rowspan="1">0.53</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.74</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="7" rowspan="1">
                  <bold>Prediction of O-glycosites</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">
                  <bold>Models</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>NetOGlyc<sup>2</sup> (0.1)</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>EnsembleGly<sup>3</sup> (0.3)</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>GlycoPP-BPP (0.2)</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>GlycoPP-CPP (0.2)</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>GlycoPP-PPP</bold>
                  <bold>(0)</bold>
                </td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>GlycoPP-PPP+ASA</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Sensitivity (%)</td>
                <td align="left" colspan="1" rowspan="1">100.00</td>
                <td align="left" colspan="1" rowspan="1">6.67</td>
                <td align="left" colspan="1" rowspan="1">72.13</td>
                <td align="left" colspan="1" rowspan="1">72.55</td>
                <td align="left" colspan="1" rowspan="1">77.05</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>81.97</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Specificity (%)</td>
                <td align="left" colspan="1" rowspan="1">3.19</td>
                <td align="left" colspan="1" rowspan="1">93.05</td>
                <td align="left" colspan="1" rowspan="1">73.77</td>
                <td align="left" colspan="1" rowspan="1">68.18</td>
                <td align="left" colspan="1" rowspan="1">70.49</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>70.49</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">Accuracy (%)</td>
                <td align="left" colspan="1" rowspan="1">8.27</td>
                <td align="left" colspan="1" rowspan="1">88.28</td>
                <td align="left" colspan="1" rowspan="1">72.95</td>
                <td align="left" colspan="1" rowspan="1">70.36</td>
                <td align="left" colspan="1" rowspan="1">73.77</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>76.23</bold>
                </td>
              </tr>
              <tr>
                <td align="left" colspan="1" rowspan="1">MCC (%)</td>
                <td align="left" colspan="1" rowspan="1">0.04</td>
                <td align="left" colspan="1" rowspan="1"><bold>−</bold>0.00</td>
                <td align="left" colspan="1" rowspan="1">0.46</td>
                <td align="left" colspan="1" rowspan="1">0.41</td>
                <td align="left" colspan="1" rowspan="1">0.48</td>
                <td align="left" colspan="1" rowspan="1">
                  <bold>0.53</bold>
                </td>
              </tr>
            </tbody>
          </table></alternatives><table-wrap-foot>
            <fn id="nt107">
              <label/>
              <p>Footnotes: 1: <ext-link ext-link-type="uri" xlink:href="http://www.cbs.dtu.dk/services/NetNGlyc/" xlink:type="simple">http://www.cbs.dtu.dk/services/NetNGlyc/</ext-link>, 2: <ext-link ext-link-type="uri" xlink:href="http://www.cbs.dtu.dk/services/NetOGlyc-3.0/" xlink:type="simple">http://www.cbs.dtu.dk/services/NetOGlyc-3.0/</ext-link>, 3: <ext-link ext-link-type="uri" xlink:href="http://turing.cs.iastate.edu/EnsembleGly/" xlink:type="simple">http://turing.cs.iastate.edu/EnsembleGly/</ext-link>, BPP- Binary profile of patterns, CPP- Composition profile of patterns, PPP- PSSM profile of patterns, MCC- Matthews correlation coefficient, AUC- Area under curve, SS-secondary structure and ASA- Accessible surface area.</p>
            </fn>
          </table-wrap-foot></table-wrap>
      </sec>
      <sec id="s3g">
        <title>Description of Web-server</title>
        <p>The overall best performing models described in <xref ref-type="table" rid="pone-0040155-t006">Table 6</xref> are implemented in the form of a web-server GlycoPP available freely at <ext-link ext-link-type="uri" xlink:href="http://www.imtech.res.in/raghava/glycopp/" xlink:type="simple">http://www.imtech.res.in/raghava/glycopp/</ext-link>. The common gateway interface of GlycoPP is written using CGI/PERL script. This server allows for prediction of N- and O-glycosites in prokaryotic protein sequences. Predictions can be performed by the users at any of the user-defined thresholds ranging from −1.0 to 1.0 for optimizing SVM scores. Input is acceptable as single or multiple sequences in standard FASTA format.</p>
      </sec>
    </sec>
    <sec id="s4">
      <title>Discussion</title>
      <p>In this study, we have developed new SVM based glycosites prediction models trained on and at least for N- and/or -O-glycosylated proteins belonging to six different archaeal and bacterial phyla namely, <italic>Crenarchaeota</italic>, <italic>Euryarchaeota</italic>, <italic>Actinobacteria</italic>, <italic>Bacteroidetes</italic>, <italic>Firmicutes</italic> and <italic>Proteobacteria</italic>. The overall best performing models are implemented at GlycoPP webserver available freely to the users (<xref ref-type="fig" rid="pone-0040155-g001">Figure 1</xref>). Our approach is similar to the existing models employed successfully for <italic>in silico</italic> identification of glycosites in eukaryotic glycoproteins <xref ref-type="bibr" rid="pone.0040155-Gupta1">[13]</xref>, <xref ref-type="bibr" rid="pone.0040155-Caragea1">[14]</xref>. The webserver allows users to identify probable sites of N- and O-glycosylation in proteins belonging to or to the similar bacteria or archaea as described above, much more confidently than possible with the existing tools of similar nature. In this study, we observed that BPP models (containing single sequence information) were more efficient in discrimination of N-glycosylated and non-glycosylated sequences irrespective of their training on balanced or realistic datasets for the presence of a defined consensus-sequon NX(S/T) in all N-glycosites. Whereas, in case of O-glycosylation, multiple sequence information based PPP models performed better as the sole classifying feature. Possibly, for the lack of a defined consensus-sequon for most O-glycosites (except in phylum <italic>Bacteroidetes</italic>), <xref ref-type="bibr" rid="pone.0040155-Bhat1">[6]</xref>, <xref ref-type="bibr" rid="pone.0040155-Fletcher1">[24]</xref>, PSSM derived profiles could well be more informative and useful for O-glycosites prediction. In our study, average surface accessibility emerged as a more useful criterion than secondary structure around glycosylated residues in most of our hybrid prediction approaches. The tool in its existing form would be useful for both single protein and proteome scale analysis. However, users are encouraged to supplement these results with other complementary evidences like presence of signal peptides, transmemebrane domains, sub-cellular localization of the proteins, presence of certain OSTs or GTs in the genome of the organism to indicate likely type and mode of glycosylation, known glycosylation in a close homologue and available experimental data on type of linkages, attached sugars etc., for best interpretation of the results obtained and also to decipher the biological significance of the same. The datasets used in this study are currently the largest and the most extensive available, yet inclusion of more validated sequences or features may further enhance the prediction accuracy, in future.</p>
      <p>Further, the preliminary information gleaned from various organism-, phylum- and domain- specific weblogos of prokaryotic glycoproteins, suggest that sequence context of bacterial and archaeal N-glycosites not only differs from eukaryotic ones but they may vary between archaea and bacteria as well (<xref ref-type="supplementary-material" rid="pone.0040155.s001">Figures S1</xref>). In view of the understanding that the archaeal OST could be evolutionarily closer to eukaryotic OST <xref ref-type="bibr" rid="pone.0040155-Maita1">[41]</xref>, it may be beneficial to develop prediction tools separately for archaea and bacteria in future, when sufficient experimental data is available. Similarly, the approach could be extended to different phyla under domain Bacteria where novel sequons for N- and O-glycosites seem to be conserved with in a phylum. For example, preference for an acidic residue at -2 position in sequon for N-glycosylation among epsilonbacteria like <italic>Campylobacter</italic> and <italic>Helicobacter</italic> and novel O-glycosylation sequon D(S/T)(A/I/L/V/M/T) in phylum <italic>Bacteroidetes</italic> (<xref ref-type="supplementary-material" rid="pone.0040155.s002">Figures S2</xref>, <xref ref-type="supplementary-material" rid="pone.0040155.s003">S3</xref>) indicate that glycan and/or acceptor sequence specificities of OSTs/GTs could be conserved within a close group of bacteria and archaea. Therefore, in future it will be desirable to develop tools where prediction could be made taking in to account the glycan and/or acceptor sequence specificities of such individual protein glycosyltransferases of prokaryotes. However, as most of the OSTs involved in <italic>en-bloc</italic> N- and O- glycosylation both in archaea and bacteria including AglB, PglB, PglL and their homologues have been shown to have relaxed glycan specificity, the correlation between acceptor sequence specificity and glycan specificity of these enzymes may not be straight (<xref ref-type="table" rid="pone-0040155-t002">Table 2</xref>), <xref ref-type="bibr" rid="pone.0040155-Ielmini1">[37]</xref>, <xref ref-type="bibr" rid="pone.0040155-Faridmoayer1">[42]</xref>, <xref ref-type="bibr" rid="pone.0040155-Calo1">[43]</xref>. In this context, it could be speculated that in prokaryotes a complex inter-play of available biosynthesis machinery of certain precursor sugars, corresponding glycans, presence of certain OSTs/GTs along with their fine tuned specificities or subtle preferences towards given glycans and/or acceptor sequences may define protein glycosylation under given conditions.</p>
      <sec id="s4a">
        <title>Supporting Information</title>
        <p>The various datasets used in this study are available in downloadable format at <ext-link ext-link-type="uri" xlink:href="http://www.imtech.res.in/raghava/glycopp/suppli.html" xlink:type="simple">http://www.imtech.res.in/raghava/glycopp/suppli.html</ext-link>.</p>
      </sec>
    </sec>
    <sec id="s5">
      <title>Supporting Information</title>
      <supplementary-material id="pone.0040155.s001" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.s001" xlink:type="simple">
        <label>Figure S1</label>
        <caption>
          <p>Weblogos for archaeal N-glycosites (panel A) and bacterial N-glycosites (panel B).</p>
          <p>(TIF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pone.0040155.s002" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.s002" xlink:type="simple">
        <label>Figure S2</label>
        <caption>
          <p>Weblogos depicting two sequons for bacterial N-glycosites: (D/E)X<sub>1</sub>NX(S/T) in <italic>Campylobacter</italic> (panel A) and NX(S/T) in <italic>Haemophilus</italic> (panel B). Panel D represents typical eukaryotic mucin like sequence context around O-glycosites of mycobacterial glycoproteins whereas O-glycosites in <italic>Campylobacter</italic> is Ser, Gly rich as shown in panel C.</p>
          <p>(TIF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pone.0040155.s003" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.s003" xlink:type="simple">
        <label>Figure S3</label>
        <caption>
          <p>Conserved sequon D(S/T)A/I/L/V/M/T at O-glycosites in glycoproteins belonging to all major representatives: <italic>Bacteroides</italic> (panel A). <italic>Flavobacterium</italic> (panel B) and <italic>Paedobacter</italic> (panel C) of phylum <italic>Bacteroidetes</italic> (panel D).</p>
          <p>(TIF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pone.0040155.s004" mimetype="image/tiff" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.s004" xlink:type="simple">
        <label>Figure S4</label>
        <caption>
          <p>ROC plots for various hybrid models for prediction of N-glycosites (panel A &amp; B) and O-glycosites (panel C &amp; D) using balanced datasets and realistic datasets, respectively. The Area Under Curve (AUC) depicts relative trade-offs between true positives and false positives.</p>
          <p>(TIF)</p>
        </caption>
      </supplementary-material>
      <supplementary-material id="pone.0040155.s005" mimetype="application/msword" position="float" xlink:href="info:doi/10.1371/journal.pone.0040155.s005" xlink:type="simple">
        <label>Table S1</label>
        <caption>
          <p>Combined prediction performance of SVM employing solo features and hybrid approaches (using realistic datasets).</p>
          <p>(DOC)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>JSC and AHB are thankful to Council of Scientific and Industrial Research (CSIR) for their Senior and Junior research fellowships, respectively.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pone.0040155-Messner1">
        <label>1</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Messner</surname><given-names>P</given-names></name></person-group>             <year>2004</year>             <article-title>Prokaryotic glycoproteins: unexplored but important.</article-title>             <source>Journal of Bacteriology</source>             <volume>186</volume>             <fpage>2517</fpage>             <lpage>2519</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-AbuQarn1">
        <label>2</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Abu-Qarn</surname><given-names>M</given-names></name><name name-style="western"><surname>Eichler</surname><given-names>J</given-names></name><name name-style="western"><surname>Sharon</surname><given-names>N</given-names></name></person-group>             <year>2008</year>             <article-title>Not just for Eukarya anymore: protein glycosylation in Bacteria and Archaea.</article-title>             <source>Curr Opin Struct Biol</source>             <volume>18</volume>             <fpage>544</fpage>             <lpage>550</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Lechner1">
        <label>3</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Lechner</surname><given-names>J</given-names></name><name name-style="western"><surname>Wieland</surname><given-names>F</given-names></name></person-group>             <year>1989</year>             <article-title>Structure and biosynthesis of prokaryotic glycoproteins.</article-title>             <source>Annu Rev Biochem</source>             <volume>58</volume>             <fpage>173</fpage>             <lpage>194</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Upreti1">
        <label>4</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Upreti</surname><given-names>RK</given-names></name><name name-style="western"><surname>Kumar</surname><given-names>M</given-names></name><name name-style="western"><surname>Shankar</surname><given-names>V</given-names></name></person-group>             <year>2003</year>             <article-title>Bacterial glycoproteins: functions, biosynthesis and applications.</article-title>             <source>Proteomics</source>             <volume>3</volume>             <fpage>363</fpage>             <lpage>379</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Varki1">
        <label>5</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Varki</surname><given-names>A</given-names></name></person-group>             <year>1993</year>             <article-title>Biological roles of oligosaccharides: all of the theories are correct.</article-title>             <source>Glycobiology</source>             <volume>3</volume>             <fpage>97</fpage>             <lpage>130</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Bhat1">
        <label>6</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Bhat</surname><given-names>AH</given-names></name><name name-style="western"><surname>Mondal</surname><given-names>H</given-names></name><name name-style="western"><surname>Chauhan</surname><given-names>JS</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name><name name-style="western"><surname>Methi</surname><given-names>A</given-names></name><etal/></person-group>             <year>2012</year>             <article-title>ProGlycProt: a repository of experimentally characterized prokaryotic glycoproteins.</article-title>             <source>Nucleic Acids Res</source>             <volume>40</volume>             <fpage>D388</fpage>             <lpage>393</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Benz1">
        <label>7</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Benz</surname><given-names>I</given-names></name><name name-style="western"><surname>Schmidt</surname><given-names>MA</given-names></name></person-group>             <year>2002</year>             <article-title>Never say never again: protein glycosylation in pathogenic bacteria.</article-title>             <source>Molecular Microbiology</source>             <volume>45</volume>             <fpage>267</fpage>             <lpage>276</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Dobos1">
        <label>8</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Dobos</surname><given-names>KM</given-names></name><name name-style="western"><surname>Khoo</surname><given-names>KH</given-names></name><name name-style="western"><surname>Swiderek</surname><given-names>KM</given-names></name><name name-style="western"><surname>Brennan</surname><given-names>PJ</given-names></name><name name-style="western"><surname>Belisle</surname><given-names>JT</given-names></name></person-group>             <year>1996</year>             <article-title>Definition of the full extent of glycosylation of the 45-kilodalton glycoprotein of Mycobacterium tuberculosis.</article-title>             <source>Journal of Bacteriology</source>             <volume>178</volume>             <fpage>2498</fpage>             <lpage>2506</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Roy1">
        <label>9</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Roy</surname><given-names>K</given-names></name><name name-style="western"><surname>Hamilton</surname><given-names>D</given-names></name><name name-style="western"><surname>Ostmann</surname><given-names>MM</given-names></name><name name-style="western"><surname>Fleckenstein</surname><given-names>JM</given-names></name></person-group>             <year>2009</year>             <article-title>Vaccination with EtpA glycoprotein or flagellin protects against colonization with enterotoxigenic Escherichia coli in a murine model.</article-title>             <source>Vaccine</source>             <volume>27</volume>             <fpage>4601</fpage>             <lpage>4608</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Jennings1">
        <label>10</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Jennings</surname><given-names>MP</given-names></name><name name-style="western"><surname>Jen</surname><given-names>FE</given-names></name><name name-style="western"><surname>Roddam</surname><given-names>LF</given-names></name><name name-style="western"><surname>Apicella</surname><given-names>MA</given-names></name><name name-style="western"><surname>Edwards</surname><given-names>JL</given-names></name></person-group>             <year>2011</year>             <article-title>Neisseria gonorrhoeae pilin glycan contributes to CR3 activation during challenge of primary cervical epithelial cells.</article-title>             <source>Cell Microbiol</source>             <volume>13</volume>             <fpage>885</fpage>             <lpage>896</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Hansen1">
        <label>11</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Hansen</surname><given-names>JE</given-names></name><name name-style="western"><surname>Lund</surname><given-names>O</given-names></name><name name-style="western"><surname>Tolstrup</surname><given-names>N</given-names></name><name name-style="western"><surname>Gooley</surname><given-names>AA</given-names></name><name name-style="western"><surname>Williams</surname><given-names>KL</given-names></name><etal/></person-group>             <year>1998</year>             <article-title>NetOglyc: prediction of mucin type O-glycosylation sites based on sequence context and surface accessibility.</article-title>             <source>Glycoconj J</source>             <volume>15</volume>             <fpage>115</fpage>             <lpage>130</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Julenius1">
        <label>12</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Julenius</surname><given-names>K</given-names></name><name name-style="western"><surname>Molgaard</surname><given-names>A</given-names></name><name name-style="western"><surname>Gupta</surname><given-names>R</given-names></name><name name-style="western"><surname>Brunak</surname><given-names>S</given-names></name></person-group>             <year>2005</year>             <article-title>Prediction, conservation analysis, and structural characterization of mammalian mucin-type O-glycosylation sites.</article-title>             <source>Glycobiology</source>             <volume>15</volume>             <fpage>153</fpage>             <lpage>164</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Gupta1">
        <label>13</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>R</given-names></name><name name-style="western"><surname>Brunak</surname><given-names>S</given-names></name></person-group>             <year>2002</year>             <article-title>Prediction of glycosylation across the human proteome and the correlation to protein function.</article-title>             <source>Pac Symp Biocomput</source>             <fpage>310</fpage>             <lpage>322</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Caragea1">
        <label>14</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Caragea</surname><given-names>C</given-names></name><name name-style="western"><surname>Sinapov</surname><given-names>J</given-names></name><name name-style="western"><surname>Silvescu</surname><given-names>A</given-names></name><name name-style="western"><surname>Dobbs</surname><given-names>D</given-names></name><name name-style="western"><surname>Honavar</surname><given-names>V</given-names></name></person-group>             <year>2007</year>             <article-title>Glycosylation site prediction using ensembles of Support Vector Machine classifiers.</article-title>             <source>BMC Bioinformatics</source>             <volume>8</volume>             <fpage>438</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Hamby1">
        <label>15</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Hamby</surname><given-names>SE</given-names></name><name name-style="western"><surname>Hirst</surname><given-names>JD</given-names></name></person-group>             <year>2008</year>             <article-title>Prediction of glycosylation sites using random forests.</article-title>             <source>BMC Bioinformatics</source>             <volume>9</volume>             <fpage>500</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Hanna1">
        <label>16</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Hanna</surname><given-names>ES</given-names></name><name name-style="western"><surname>Roque-Barreira</surname><given-names>MC</given-names></name><name name-style="western"><surname>Bernardes</surname><given-names>ES</given-names></name><name name-style="western"><surname>Panunto-Castelo</surname><given-names>A</given-names></name><name name-style="western"><surname>Sousa</surname><given-names>MV</given-names></name><etal/></person-group>             <year>2007</year>             <article-title>Evidence for glycosylation on a DNA-binding protein of <italic>Salmonella enterica</italic>.</article-title>             <source>Microb Cell Fact</source>             <volume>6</volume>             <fpage>11</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Herrmann1">
        <label>17</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Herrmann</surname><given-names>JL</given-names></name><name name-style="western"><surname>Delahay</surname><given-names>R</given-names></name><name name-style="western"><surname>Gallagher</surname><given-names>A</given-names></name><name name-style="western"><surname>Robertson</surname><given-names>B</given-names></name><name name-style="western"><surname>Young</surname><given-names>D</given-names></name></person-group>             <year>2000</year>             <article-title>Analysis of post-translational modification of mycobacterial proteins using a cassette expression system.</article-title>             <source>FEBS Lett</source>             <volume>473</volume>             <fpage>358</fpage>             <lpage>62</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Balonova1">
        <label>18</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Balonova</surname><given-names>L</given-names></name><name name-style="western"><surname>Hernychova</surname><given-names>L</given-names></name><name name-style="western"><surname>Mann</surname><given-names>BF</given-names></name><name name-style="western"><surname>Link</surname><given-names>M</given-names></name><name name-style="western"><surname>Bilkova</surname><given-names>Z</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>Multimethodological approach to identification of glycoproteins from the proteome of <italic>Francisella tularensis</italic>, an intracellular microorganism.</article-title>             <source>J Proteome Res</source>             <volume>9</volume>             <fpage>1995</fpage>             <lpage>2005</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Ghoshal1">
        <label>19</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ghoshal</surname><given-names>A</given-names></name><name name-style="western"><surname>Mukhopadhyay</surname><given-names>S</given-names></name><name name-style="western"><surname>Demine</surname><given-names>R</given-names></name><name name-style="western"><surname>Forgber</surname><given-names>M</given-names></name><name name-style="western"><surname>Jarmalavicius</surname><given-names>S</given-names></name><etal/></person-group>             <year>2009</year>             <article-title>Detection and characterization of a sialoglycosylated bacterial ABC-type phosphate transporter protein from patients with visceral leishmaniasis.</article-title>             <source>Glycoconj J</source>             <volume>26</volume>             <fpage>675</fpage>             <lpage>89</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Dell1">
        <label>20</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Dell</surname><given-names>A</given-names></name><name name-style="western"><surname>Galadari</surname><given-names>A</given-names></name><name name-style="western"><surname>Sastre</surname><given-names>F</given-names></name><name name-style="western"><surname>Hitchen</surname><given-names>P</given-names></name></person-group>             <year>2010</year>             <article-title>Similarities and differences in the glycosylation mechanisms in prokaryotes and eukaryotes.</article-title>             <source>Int J Microbiol</source>             <volume>2010</volume>             <fpage>148–178</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Nothaft1">
        <label>21</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Nothaft</surname><given-names>H</given-names></name><name name-style="western"><surname>Szymanski</surname><given-names>CM</given-names></name></person-group>             <year>2010</year>             <article-title>Protein glycosylation in bacteria: sweeter than ever.</article-title>             <source>Nat Rev Microbiol</source>             <volume>8</volume>             <fpage>765</fpage>             <lpage>778</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Marino1">
        <label>22</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Marino</surname><given-names>K</given-names></name><name name-style="western"><surname>Bones</surname><given-names>J</given-names></name><name name-style="western"><surname>Kattla</surname><given-names>JJ</given-names></name><name name-style="western"><surname>Rudd</surname><given-names>PM</given-names></name></person-group>             <year>2010</year>             <article-title>A systematic approach to protein glycosylation analysis: a path through the maze.</article-title>             <source>Nat Chem Biol</source>             <volume>6</volume>             <fpage>713</fpage>             <lpage>723</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Kowarik1">
        <label>23</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kowarik</surname><given-names>M</given-names></name><name name-style="western"><surname>Young</surname><given-names>NM</given-names></name><name name-style="western"><surname>Numao</surname><given-names>S</given-names></name><name name-style="western"><surname>Schulz</surname><given-names>BL</given-names></name><name name-style="western"><surname>Hug</surname><given-names>I</given-names></name><etal/></person-group>             <year>2006</year>             <article-title>Definition of the bacterial N-glycosylation site consensus sequence.</article-title>             <source>Embo Journal</source>             <volume>25</volume>             <fpage>1957</fpage>             <lpage>1966</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Fletcher1">
        <label>24</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Fletcher</surname><given-names>CM</given-names></name><name name-style="western"><surname>Coyne</surname><given-names>MJ</given-names></name><name name-style="western"><surname>Comstock</surname><given-names>LE</given-names></name></person-group>             <year>2011</year>             <article-title>Theoretical and experimental characterization of the scope of protein O-glycosylation in Bacteroides fragilis.</article-title>             <source>Journal of Biological Chemistry</source>             <volume>286</volume>             <fpage>3219</fpage>             <lpage>3226</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-AbuQarn2">
        <label>25</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Abu-Qarn</surname><given-names>M</given-names></name><name name-style="western"><surname>Eichler</surname><given-names>J</given-names></name></person-group>             <year>2007</year>             <article-title>An analysis of amino acid sequences surrounding archaeal glycoprotein sequons.</article-title>             <source>Archaea</source>             <volume>2</volume>             <fpage>73</fpage>             <lpage>81</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-BenDor1">
        <label>26</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ben-Dor</surname><given-names>S</given-names></name><name name-style="western"><surname>Esterman</surname><given-names>N</given-names></name><name name-style="western"><surname>Rubin</surname><given-names>E</given-names></name><name name-style="western"><surname>Sharon</surname><given-names>N</given-names></name></person-group>             <year>2004</year>             <article-title>Biases and complex patterns in the residues flanking protein N-glycosylation sites.</article-title>             <source>Glycobiology</source>             <volume>14</volume>             <fpage>95</fpage>             <lpage>101</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Petrescu1">
        <label>27</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Petrescu</surname><given-names>AJ</given-names></name><name name-style="western"><surname>Milac</surname><given-names>AL</given-names></name><name name-style="western"><surname>Petrescu</surname><given-names>SM</given-names></name><name name-style="western"><surname>Dwek</surname><given-names>RA</given-names></name><name name-style="western"><surname>Wormald</surname><given-names>MR</given-names></name></person-group>             <year>2004</year>             <article-title>Statistical analysis of the protein environment of N-glycosylation sites: implications for occupancy, structure, and folding.</article-title>             <source>Glycobiology</source>             <volume>14</volume>             <fpage>103</fpage>             <lpage>114</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Kowarik2">
        <label>28</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kowarik</surname><given-names>M</given-names></name><name name-style="western"><surname>Numao</surname><given-names>S</given-names></name><name name-style="western"><surname>Feldman</surname><given-names>MF</given-names></name><name name-style="western"><surname>Schulz</surname><given-names>BL</given-names></name><name name-style="western"><surname>Callewaert</surname><given-names>N</given-names></name><etal/></person-group>             <year>2006</year>             <article-title>N-linked glycosylation of folded proteins by the bacterial oligosaccharyltransferase.</article-title>             <source>Science</source>             <volume>314</volume>             <fpage>1148</fpage>             <lpage>1150</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Chauhan1">
        <label>29</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Chauhan</surname><given-names>JS</given-names></name><name name-style="western"><surname>Mishra</surname><given-names>NK</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name></person-group>             <year>2010</year>             <article-title>Prediction of GTP interacting residues, dipeptides and tripeptides in a protein from its evolutionary information.</article-title>             <source>BMC Bioinformatics</source>             <volume>11</volume>             <fpage>301</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Agarwal1">
        <label>30</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names></name><name name-style="western"><surname>Mishra</surname><given-names>NK</given-names></name><name name-style="western"><surname>Singh</surname><given-names>H</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name></person-group>             <year>2011</year>             <article-title>Identification of mannose interacting residues using local composition.</article-title>             <source>PLoS One</source>             <volume>6</volume>             <fpage>e24039</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Kumar1">
        <label>31</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kumar</surname><given-names>M</given-names></name><name name-style="western"><surname>Gromiha</surname><given-names>MM</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name></person-group>             <year>2008</year>             <article-title>Prediction of RNA binding sites in a protein using SVM and PSSM profile.</article-title>             <source>Proteins</source>             <volume>71</volume>             <fpage>189</fpage>             <lpage>194</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-McGuffin1">
        <label>32</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>McGuffin</surname><given-names>LJ</given-names></name><name name-style="western"><surname>Bryson</surname><given-names>K</given-names></name><name name-style="western"><surname>Jones</surname><given-names>DT</given-names></name></person-group>             <year>2000</year>             <article-title>The PSIPRED protein structure prediction server.</article-title>             <source>Bioinformatics</source>             <volume>16</volume>             <fpage>404</fpage>             <lpage>405</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Garg1">
        <label>33</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Garg</surname><given-names>A</given-names></name><name name-style="western"><surname>Kaur</surname><given-names>H</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name></person-group>             <year>2005</year>             <article-title>Real value prediction of solvent accessibility in proteins using multiple sequence alignment and secondary structure.</article-title>             <source>Proteins</source>             <volume>61</volume>             <fpage>318</fpage>             <lpage>324</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Joachims1">
        <label>34</label>
        <element-citation publication-type="other" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Joachims</surname><given-names>T</given-names></name></person-group>             <year>1999</year>             <article-title>Making large-Scale SVM Learning Practical In: Advances in Kernel Models - Support Vector Learning, B. Schölkopf and C. Burges and A. Smola (ed.), MIT-Press.</article-title>          </element-citation>
      </ref>
      <ref id="pone.0040155-Kalita1">
        <label>35</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Kalita</surname><given-names>MK</given-names></name><name name-style="western"><surname>Nandal</surname><given-names>UK</given-names></name><name name-style="western"><surname>Pattnaik</surname><given-names>A</given-names></name><name name-style="western"><surname>Sivalingam</surname><given-names>A</given-names></name><name name-style="western"><surname>Ramasamy</surname><given-names>G</given-names></name><etal/></person-group>             <year>2008</year>             <article-title>CyclinPred: a SVM-based method for predicting cyclin protein sequences.</article-title>             <source>PLoS One</source>             <volume>3</volume>             <fpage>e2605</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Ansari1">
        <label>36</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ansari</surname><given-names>HR</given-names></name><name name-style="western"><surname>Raghava</surname><given-names>GP</given-names></name></person-group>             <year>2010</year>             <article-title>Identification of conformational B-cell Epitopes in an antigen from its primary sequence.</article-title>             <source>Immunome Res</source>             <volume>6</volume>             <fpage>6</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Ielmini1">
        <label>37</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Ielmini</surname><given-names>MV</given-names></name><name name-style="western"><surname>Feldman</surname><given-names>MF</given-names></name></person-group>             <year>2011</year>             <article-title>Desulfovibrio desulfuricans PglB homolog possesses oligosaccharyltransferase activity with relaxed glycan specificity and distinct protein acceptor sequence requirements.</article-title>             <source>Glycobiology</source>             <volume>21</volume>             <fpage>734</fpage>             <lpage>742</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Choi1">
        <label>38</label>
        <element-citation publication-type="journal" xlink:type="simple">             <collab xlink:type="simple">Choi KJ, Grass S, Paek S, St Geme JW, 3rd, Yeo HJ</collab>             <year>2010</year>             <article-title>The Actinobacillus pleuropneumoniae HMW1C-like glycosyltransferase mediates N-linked glycosylation of the Haemophilus influenzae HMW1 adhesin.</article-title>             <source>PLoS One</source>             <volume>5</volume>             <fpage>e15888</fpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Jervis1">
        <label>39</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Jervis</surname><given-names>AJ</given-names></name><name name-style="western"><surname>Langdon</surname><given-names>R</given-names></name><name name-style="western"><surname>Hitchen</surname><given-names>P</given-names></name><name name-style="western"><surname>Lawson</surname><given-names>AJ</given-names></name><name name-style="western"><surname>Wood</surname><given-names>A</given-names></name><etal/></person-group>             <year>2010</year>             <article-title>Characterization of N-linked protein glycosylation in Helicobacter pullorum.</article-title>             <source>Journal of Bacteriology</source>             <volume>192</volume>             <fpage>5228</fpage>             <lpage>5236</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Rangarajan1">
        <label>40</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Rangarajan</surname><given-names>ES</given-names></name><name name-style="western"><surname>Bhatia</surname><given-names>S</given-names></name><name name-style="western"><surname>Watson</surname><given-names>DC</given-names></name><name name-style="western"><surname>Munger</surname><given-names>C</given-names></name><name name-style="western"><surname>Cygler</surname><given-names>M</given-names></name><etal/></person-group>             <year>2007</year>             <article-title>Structural context for protein N-glycosylation in bacteria: The structure of PEB3, an adhesin from Campylobacter jejuni.</article-title>             <source>Protein Sci</source>             <volume>16</volume>             <fpage>990</fpage>             <lpage>995</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Maita1">
        <label>41</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Maita</surname><given-names>N</given-names></name><name name-style="western"><surname>Nyirenda</surname><given-names>J</given-names></name><name name-style="western"><surname>Igura</surname><given-names>M</given-names></name><name name-style="western"><surname>Kamishikiryo</surname><given-names>J</given-names></name><name name-style="western"><surname>Kohda</surname><given-names>D</given-names></name></person-group>             <year>2010</year>             <article-title>Comparative structural biology of eubacterial and archaeal oligosaccharyltransferases.</article-title>             <source>Journal of Biological Chemistry</source>             <volume>285</volume>             <fpage>4941</fpage>             <lpage>4950</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Faridmoayer1">
        <label>42</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Faridmoayer</surname><given-names>A</given-names></name><name name-style="western"><surname>Fentabil</surname><given-names>MA</given-names></name><name name-style="western"><surname>Haurat</surname><given-names>MF</given-names></name><name name-style="western"><surname>Yi</surname><given-names>W</given-names></name><name name-style="western"><surname>Woodward</surname><given-names>R</given-names></name><etal/></person-group>             <year>2008</year>             <article-title>Extreme substrate promiscuity of the <italic>Neisseria</italic> oligosaccharyl transferase involved in protein O-glycosylation.</article-title>             <source>J Biol Chem</source>             <volume>283</volume>             <fpage>34596</fpage>             <lpage>34604</lpage>          </element-citation>
      </ref>
      <ref id="pone.0040155-Calo1">
        <label>43</label>
        <element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author"><name name-style="western"><surname>Calo</surname><given-names>D</given-names></name><name name-style="western"><surname>Kaminski</surname><given-names>L</given-names></name><name name-style="western"><surname>Eichler</surname><given-names>J</given-names></name></person-group>             <year>2010</year>             <article-title>Protein glycosylation in Archaea: sweet and extreme.</article-title>             <source>Glycobiology</source>             <volume>20</volume>             <fpage>1065</fpage>             <lpage>1076</lpage>          </element-citation>
      </ref>
    </ref-list>
    
  </back>
</article>