<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="en">
  <front>
    <journal-meta>
      <journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
      <journal-id journal-id-type="publisher-id">plos</journal-id>
      <journal-id journal-id-type="pmc">plosone</journal-id>
      <journal-title-group>
        <journal-title>PLoS ONE</journal-title>
      </journal-title-group>
      <issn pub-type="epub">1932-6203</issn>
      <publisher>
        <publisher-name>Public Library of Science</publisher-name>
        <publisher-loc>San Francisco, USA</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">PONE-D-12-14754</article-id>
      <article-id pub-id-type="doi">10.1371/journal.pone.0049425</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Article</subject>
        </subj-group>
        <subj-group subj-group-type="Discipline-v2">
          <subject>Biology</subject>
          <subj-group>
            <subject>Biochemistry</subject>
            <subj-group>
              <subject>Proteins</subject>
              <subj-group>
                <subject>Protein chemistry</subject>
                <subject>Protein structure</subject>
                <subject>Recombinant proteins</subject>
              </subj-group>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Computational biology</subject>
            <subj-group>
              <subject>Biological data management</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Microbiology</subject>
            <subj-group>
              <subject>Bacteriology</subject>
            </subj-group>
          </subj-group>
          <subj-group>
            <subject>Proteomics</subject>
            <subj-group>
              <subject>Proteomic databases</subject>
            </subj-group>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="Discipline">
          <subject>Microbiology</subject>
          <subject>Computational Biology</subject>
          <subject>Biochemistry</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>CyanoPhyChe: A Database for Physico-Chemical Properties, Structure and Biochemical Pathway Information of Cyanobacterial Proteins</article-title>
        <alt-title alt-title-type="running-head">A Database on Cyanobacterial Protein Properties</alt-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Arun</surname>
            <given-names>P. V. Parvati Sai</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Bakku</surname>
            <given-names>Ranjith Kumar</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Subhashini</surname>
            <given-names>Mranu</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Singh</surname>
            <given-names>Pankaj</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Prabhu</surname>
            <given-names>N. Prakash</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Suzuki</surname>
            <given-names>Iwane</given-names>
          </name>
          <xref ref-type="aff" rid="aff3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" xlink:type="simple">
          <name name-style="western">
            <surname>Prakash</surname>
            <given-names>Jogadhenu S. S.</given-names>
          </name>
          <xref ref-type="aff" rid="aff2">
            <sup>2</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">
            <sup>*</sup>
          </xref>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <addr-line>Department of Plant Sciences, School of Life Sciences, University of Hyderabad, Hyderabad, Andhra Pradesh, India</addr-line>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <addr-line>Department of Biotechnology, School of Life Sciences, University of Hyderabad, Hyderabad, Andhra Pradesh, India</addr-line>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <addr-line>Faculty of Life and Environmental Science, University of Tsukuba, Tsukuba, Japan</addr-line>
      </aff>
      <contrib-group>
        <contrib contrib-type="editor" xlink:type="simple">
          <name name-style="western">
            <surname>Sutherland-Smith</surname>
            <given-names>Andrew John</given-names>
          </name>
          <role>Editor</role>
          <xref ref-type="aff" rid="edit1"/>
        </contrib>
      </contrib-group>
      <aff id="edit1">
        <addr-line>Massey University, New Zealand</addr-line>
      </aff>
      <author-notes>
        <corresp id="cor1">* E-mail: <email xlink:type="simple">syamsunderp@yahoo.com</email></corresp>
        <fn fn-type="conflict">
          <p>The authors have the following interests: For this study the CyanoPhyChe database was developed. One can browse through it for physico-chemical properties, structure and biochemical pathway information of cyanobacterial proteins. It can be accessed from a local web server located in Bioinformatics Infrastructure Facility at School of life Sciences, University of Hyderabad, India using the following URL: <ext-link ext-link-type="uri" xlink:href="http://bif.uohyd.ac.in/cpc" xlink:type="simple">http://bif.uohyd.ac.in/cpc</ext-link>. There are no further patents, products in development or marketed products to declare. This does not alter the authors’ adherence to all the PLOS ONE policies on sharing data and materials, as detailed online in the guide for authors.</p>
        </fn>
        <fn fn-type="con">
          <p>Conceived and designed the experiments: NPP JSSP. Performed the experiments: PVPSA RKB PS MS. Analyzed the data: NPP IS JSSP. Contributed reagents/materials/analysis tools: RKB. Wrote the paper: RKB NPP JSSP.</p>
        </fn>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2012</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>21</day>
        <month>11</month>
        <year>2012</year>
      </pub-date>
      <volume>7</volume>
      <issue>11</issue>
      <elocation-id>e49425</elocation-id>
      <history>
        <date date-type="received">
          <day>24</day>
          <month>5</month>
          <year>2012</year>
        </date>
        <date date-type="accepted">
          <day>7</day>
          <month>10</month>
          <year>2012</year>
        </date>
      </history>
      <permissions>
        <copyright-year>2012</copyright-year>
        <copyright-holder>Arun et al</copyright-holder>
        <license xlink:type="simple">
          <license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>CyanoPhyChe is a user friendly database that one can browse through for physico-chemical properties, structure and biochemical pathway information of cyanobacterial proteins. We downloaded all the protein sequences from the cyanobacterial genome database for calculating the physico-chemical properties, such as molecular weight, net charge of protein, isoelectric point, molar extinction coefficient, canonical variable for solubility, grand average hydropathy, aliphatic index, and number of charged residues. Based on the physico-chemical properties, we provide the polarity, structural stability and probability of a protein entering in to an inclusion body (PEPIB). We used the data generated on physico-chemical properties, structure and biochemical pathway information of all cyanobacterial proteins to construct CyanoPhyChe. The data can be used for optimizing methods of expression and characterization of cyanobacterial proteins. Moreover, the ‘Search’ and data export options provided will be useful for proteome analysis. Secondary structure was predicted for all the cyanobacterial proteins using PSIPRED tool and the data generated is made accessible to researchers working on cyanobacteria. In addition, external links are provided to biological databases such as PDB and KEGG for molecular structure and biochemical pathway information, respectively. External links are also provided to different cyanobacterial databases. CyanoPhyChe can be accessed from the following URL: <ext-link ext-link-type="uri" xlink:href="http://bif.uohyd.ac.in/cpc" xlink:type="simple">http://bif.uohyd.ac.in/cpc</ext-link>.</p>
      </abstract>
      <funding-group>
        <funding-statement>This work was supported by a grant from the Department of Biotechnology (DBT), Project No: BT/PR13616/BRB/10/774/2010 to JSSP. This study utilized the DBT - sponsored Bioinformatics Infrastructure Facility (BIF), to School of Life Sciences, University of Hyderabad. The funders had no role in study design, data collection, and analysis, decision to publish or preparation of the manuscript.</funding-statement>
      </funding-group>
      <counts>
        <page-count count="7"/>
      </counts>
    </article-meta>
  </front>
  <body>
    <sec id="s1">
      <title>Introduction</title>
      <p>As proteins mediate and control coordinated biochemical transformations and cellular processes that are central to activity of life forms, characterization of proteins provide insights into the structure and function of a cell <xref ref-type="bibr" rid="pone.0049425-Mathura1">[1]</xref>, <xref ref-type="bibr" rid="pone.0049425-Creighton1">[2]</xref>. Pure form of a protein is needed for its characterization and can be extracted through expression and purification techniques. Instead of getting an active and soluble form of a protein, there are chances for a protein to enter into an inclusion body <xref ref-type="bibr" rid="pone.0049425-Kopito1">[3]</xref>, <xref ref-type="bibr" rid="pone.0049425-Fink1">[4]</xref>. Obtaining functionally active protein from an inclusion body requires denaturation of the protein, followed by refolding into its native form. This is a slow and difficult process which greatly reduces the net yield and activity of the protein <xref ref-type="bibr" rid="pone.0049425-Harrison1">[5]</xref>. Therefore, prevention of a protein to enter into inclusion body is better than resolving it. Solubility of a protein depends on its physico-chemical properties. For instance, length of a protein, composition and properties of amino acid residues in a protein influence its solubility <xref ref-type="bibr" rid="pone.0049425-Wilkinson1">[6]</xref>. Folding of an expressed protein also depends on the conditions employed during the process of expression and purification. It is possible to prevent the aggregation of a protein by providing suitable conditions based on its physico-chemical properties. The physico-chemical properties can be used to predict the nature of a protein and this information is useful for optimization of expression methods. These properties help to understand the native environment in which the protein will be in soluble and active form, thus aids the researchers in the characterization studies on the proteins of interest. In addition to the available traditional standard laboratory methods for determining physico-chemical properties of a protein, mathematical methods have also been in use for calculating the same based on primary sequence information <xref ref-type="bibr" rid="pone.0049425-Wilkinson1">[6]</xref>–<xref ref-type="bibr" rid="pone.0049425-Kyte1">[13]</xref>. Though different web based tools are available to determine physico-chemical properties of proteins <xref ref-type="bibr" rid="pone.0049425-Mathura1">[1]</xref>, <xref ref-type="bibr" rid="pone.0049425-Rice1">[14]</xref>–<xref ref-type="bibr" rid="pone.0049425-Li1">[17]</xref>, it is time consuming and difficult task for a naïve user in choosing a tool among the available pool.</p>
      <fig id="pone-0049425-g001" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0049425.g001</object-id>
        <label>Figure 1</label>
        <caption>
          <title>A snapshot of the homepage of CyanoPhyChe database.</title>
          <p>User can navigate for physico-chemical properties of any cyanobacterial protein by browsing through ‘Browse’ link located at the top left corner of the home page. Proteins with specific properties can be retrieved using ‘Search’ option. Links to external cyanobacterial databases are provided in the main page of the database.</p>
        </caption>
        <graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0049425.g001" position="float" xlink:type="simple"/>
      </fig>
      <p>Cyanobacteria are a group of oxygenic photosynthetic microorganisms that survived since life took its form on Earth. Because of their vital metabolic pathways and being contributors to global carbon and nitrogen budgets, they are one of the mostly studied microbes. At present, there are 38 cyanobacterial species for which total genome sequence information is available (<ext-link ext-link-type="uri" xlink:href="http://genome.kazusa.or.jp/cyanobase" xlink:type="simple">http://genome.kazusa.or.jp/cyanobase</ext-link>). The available genome sequence information can be used to generate huge data, by applying bioinformatic, functional and comparative genomic approaches, which would address several questions related to the evolution, adaptation, physiology and biochemistry of cyanobacteria. Nevertheless, there is no database which can provide physico-chemical properties of cyanobacterial proteins. This motivated us to make a database on physico-chemical properties of cyanobacterial proteins using different tools and formulae which are scientifically proven to be accurate <xref ref-type="bibr" rid="pone.0049425-Wilkinson1">[6]</xref>, <xref ref-type="bibr" rid="pone.0049425-Ikai1">[10]</xref>, <xref ref-type="bibr" rid="pone.0049425-Kyte1">[13]</xref>–<xref ref-type="bibr" rid="pone.0049425-Wishart1">[16]</xref>. Thus, in this report we provide a user friendly database in which one can easily search for the properties of a cyanobacterial protein(s) in question and get preliminary understanding about it. We used the genome data from the database of cyanobacteria for calculating the physico-chemical properties of all cyanobacterial proteins and generated a database called ‘CyanoPhyChe’. Researchers can use the physico-chemical properties of any cyanobacterial protein that is available in the database for their research. The information provided in the database aids researchers for choosing optimal conditions for expression, purification and characterization of a cyanobacterial protein. Moreover, user can export the physicochemical properties, predicted secondary structure, amino acid sequence and amino acid composition of selected cyanobacterial proteins for further analysis. External links are provided to make a direct access to other cyanobacterial databases available on the internet.</p>
      <fig id="pone-0049425-g002" position="float">
        <object-id pub-id-type="doi">10.1371/journal.pone.0049425.g002</object-id>
        <label>Figure 2</label>
        <caption>
          <title>A snapshot of an access page.</title>
          <p>This page contains a menu with physico-chemical properties, amino acid composition, biochemical pathway and structure information.</p>
        </caption>
        <graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0049425.g002" position="float" xlink:type="simple"/>
      </fig>
    </sec>
    <sec id="s2" sec-type="materials|methods">
      <title>Materials and Methods</title>
      <sec id="s2a">
        <title>Protein Sequence Data for Calculating Physico-chemical Properties</title>
        <p>Total 38 files with extension.faa, each containing all protein sequences of a cyanobacterial species, were downloaded from NCBI (<ext-link ext-link-type="uri" xlink:href="ftp://ftp.ncbi.nih.gov/genomes/Bacteria" xlink:type="simple">ftp://ftp.ncbi.nih.gov/genomes/Bacteria</ext-link>). A code was developed in PERL to separate all the protein sequences of a ‘.faa’ file into multiple FASTA files, each with an individual protein sequence. This primary seed data is used for calculating protein properties. Seed data contained total 1,26,610 proteins, covering 38 cyanobacteria.</p>
        <fig id="pone-0049425-g003" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pone.0049425.g003</object-id>
          <label>Figure 3</label>
          <caption>
            <title>A snapshot of physico-chemical properties of the selected proteins.</title>
            <p>Physico-chemical properties of (A) Sll0649, (B) Sll0088 and (C) Sll0359 proteins from <italic>Synechocystis</italic> sp. PCC6803. Based on the presented physico-chemical properties of, solubility, probability of the protein entering into an inclusion body (PEPIB), polarity and structural stability were calculated and indicated on a point scale for better visualization of its nature.</p>
          </caption>
          <graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0049425.g003" position="float" xlink:type="simple"/>
        </fig>
        <p>Certain physico-chemical properties were calculated using PEPSTATS tool, which is available in EMBOSS package, installed in a computer using Linux mint-12 operating system (<ext-link ext-link-type="uri" xlink:href="http://www.ebi.ac.uk/Tools/emboss/pepinfo/" xlink:type="simple">http://www.ebi.ac.uk/Tools/emboss/pepinfo/</ext-link>). PEPSTATS provides molecular weight, number of residues, isoelectric point (pI), molar extinction coefficient and amino acid composition of a protein <xref ref-type="bibr" rid="pone.0049425-Rice1">[14]</xref>. The output file generated by PEPSTATS was used as a secondary seed data for calculating other properties like, aliphatic index (AI), GRAVY, canonical variable for solubility (<italic>CV</italic><sub>sol</sub>) and probability of expressed protein entering into an inclusion body (PEPIB).</p>
        <fig id="pone-0049425-g004" position="float">
          <object-id pub-id-type="doi">10.1371/journal.pone.0049425.g004</object-id>
          <label>Figure 4</label>
          <caption>
            <title>A snapshot showing the CyanoPhyChe ‘Search’ page.</title>
            <p>A dropdown menu provided in the search page with various parameters can be used for retrieving cyanobacterial proteins in the database. Proteins of a cyanobacterium can be retrieved by gene name, locus ID, E.C. number and protein ID, and properties like molecular weight, number of residues, and pI value. User can also retrieve the cyanobacterial protein(s) by searching the database using combination of two or more properties listed under “Search with combination of properties” menu.</p>
          </caption>
          <graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0049425.g004" position="float" xlink:type="simple"/>
        </fig>
        <sec id="s2a1">
          <title>Aliphatic index and structural stability</title>
          <p>Stability of a protein can be calculated using aliphatic index. We used the formula that was developed for determining aliphatic index <xref ref-type="bibr" rid="pone.0049425-Ikai1">[10]</xref>, <xref ref-type="bibr" rid="pone.0049425-Argos1">[18]</xref>. The aliphatic index values of cyanobacterial proteins were normalized between 0 and 10 to predict the structural stability.</p>
        </sec>
        <sec id="s2a2">
          <title>Grand average value of hydropathy (GRAVY)</title>
          <p>An empirical formula for calculating hydropathy value for a protein was developed by Kyte and Doolittle (1982), wherein the hydrophilic and hydrophobic properties of each amino acid side chain in a protein are taken into consideration <xref ref-type="bibr" rid="pone.0049425-Kyte1">[13]</xref>. Positive hydropathy value is indicative of a polar protein and negative value indicates a non-polar protein.</p>
        </sec>
        <sec id="s2a3">
          <title>Canonical variable for solubility (CVsol) and PEPIB</title>
          <p>A mathematical formula, based on the amino acid composition and their properties, was derived to predict the solubility of a protein and its probability to enter into an inclusion body <xref ref-type="bibr" rid="pone.0049425-Wilkinson1">[6]</xref>. The solubility or insolubility of a protein can be predicted from the canonical variable, which is a composite parameter of cysteine fraction, proline fraction, turn forming residue fraction, approximate charge average, number of residues and hydrophilicity, according to Wilkinson and Harrison model <xref ref-type="bibr" rid="pone.0049425-Wilkinson1">[6]</xref>. The formula for calculating the canonical variable is given below.<disp-formula id="pone.0049425.e001"><graphic position="anchor" xlink:href="info:doi/10.1371/journal.pone.0049425.e001" xlink:type="simple"/></disp-formula>where,</p>
          <p><italic>n</italic> = number of amino acids in protein</p>
          <p><italic>N</italic>, <italic>G, P</italic>, <italic>S</italic> = number of Asn, Gly, Pro &amp; Ser residues respectively.</p>
          <p><italic>R</italic>, <italic>K</italic>, <italic>D</italic>, <italic>E</italic> = number of Arg, Lys, Asp &amp; Glu residues.</p>
          <p><italic>λ</italic><sub>1</sub> and <italic>λ</italic><sub>2</sub> =  Coefficients (15.43 and −29.56 respectively) and</p>
          <p>Canonical variable for solubility (CV<sub>sol</sub>) = (CV–CV′)</p>
          <p><italic>Canonical variable for solubility</italic> <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0049425.e002" xlink:type="simple"/></inline-formula>where, <italic>CV′</italic> = 1.71.</p>
          <p><italic>Probability of solubility or insolubility</italic> = <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0049425.e003" xlink:type="simple"/></inline-formula>.</p>
          <p>The formula for calculating probability of solubility or insolubility for any protein when expressed in <italic>E. coli</italic> was further evaluated by Harrison 2000 <xref ref-type="bibr" rid="pone.0049425-Harrison1">[5]</xref>. Also, a detailed description of canonical variable and its usage has been given by Koschorreck et al. 2005 <xref ref-type="bibr" rid="pone.0049425-Koschorreck1">[19]</xref>. Based on the sign of <italic>CV</italic><sub>sol</sub> value PEPIB was determined. If <italic>CV</italic><sub>sol</sub> value is positive, the above equation provides the probability of insolubility, which is considered as PEPIB. If <italic>CV</italic><sub>sol</sub> value is negative, the equation provides the probability of solubility <xref ref-type="bibr" rid="pone.0049425-Harrison1">[5]</xref>. PEPIB of these proteins were calculated by converting their probability of solubility in to the probability of insolubility, considering the fact that the sum of probability of solubility and insolubility must be one.</p>
        </sec>
      </sec>
      <sec id="s2b">
        <title>Secondary Structure Prediction</title>
        <p>Secondary structure prediction was done using PSIPRED tool <xref ref-type="bibr" rid="pone.0049425-McGuffin1">[15]</xref>. UNIREF90 database was used as target for PSI-Blast (<ext-link ext-link-type="uri" xlink:href="ftp://ftp.ebi.ac.uk/pub/databases/uniprot/uniref/" xlink:type="simple">ftp://ftp.ebi.ac.uk/pub/databases/uniprot/uniref/</ext-link>). A BASH shell program was designed to automate this tool for predicting secondary structure for cyanobacterial proteins.</p>
      </sec>
      <sec id="s2c">
        <title>Design of CyanoPhyChe Database and its Accessibility</title>
        <p>The CyanoPhyChe database was developed using MySQL database management system (MySQL Version 5.1.4.1 and php MyAdmin 3.2.4). Web interface was designed using HTML, Java script and PHP to retrieve and visualize the data. PDB IDs and biochemical pathway IDs were extracted from NCBI (<ext-link ext-link-type="uri" xlink:href="ftp://ftp.ncbi.nih.gov/genomes/Bacteria" xlink:type="simple">ftp://ftp.ncbi.nih.gov/genomes/Bacteria</ext-link>) and KEGG (<ext-link ext-link-type="uri" xlink:href="http://www.genome.jp/kegg/" xlink:type="simple">http://www.genome.jp/kegg/</ext-link>) databases respectively for all cyanobacterial proteins. The pathway IDs were provided with links using HTML and PHP scripting. CyanoPhyChe can be accessed from a local web server located in Bioinformatics Infrastructure Facility at School of life Sciences, University of Hyderabad, India using the following URL: <ext-link ext-link-type="uri" xlink:href="http://bif.uohyd.ac.in/cpc" xlink:type="simple">http://bif.uohyd.ac.in/cpc</ext-link>.</p>
      </sec>
    </sec>
    <sec id="s3">
      <title>Results and Discussion</title>
      <sec id="s3a">
        <title>Description of CyanoPhyChe</title>
        <p>The web interface contains ‘Home’, ‘Browse’, ‘Search’, ‘Help’ and ‘Contact’ links on the top left corner of the index page for browsing the database (<xref ref-type="fig" rid="pone-0049425-g001">Figure 1</xref>). A brief introduction about CyanoPhyChe is given in the home page. The physico-chemical properties, structural and biochemical pathway information of any cyanobacterial protein can be accessed either by ‘Browse’ or ‘Search’ link. Further, the webpage contains external links to other cyanobacterial databases such as cyanobase, cyanoclust, cyanoDB.cz and cyanoBIKE. This will assist the user to get a direct access to other related databases for additional information on cyanobacteria.</p>
      </sec>
      <sec id="s3b">
        <title>Browsing the Database</title>
        <p>As the user goes through the ‘Browse’ option, the list of cyanobacteria is displayed. Name of each cyanobacterium is further linked to a table that lists all the proteins coded by its genome, along with their protein ID, locus ID, gene name and function. Upon a single click on any protein, an access page appears that displays links to physico-chemical properties, amino acid composition, biochemical pathway and structure information of the selected protein (<xref ref-type="fig" rid="pone-0049425-g002">Figure 2</xref>). User can select more than one protein from the protein-list of a cyanobacterium and export the data in CSV format. In addition, the user can also download the secondary structure, protein sequence and amino acid composition of the selected proteins.</p>
        <sec id="s3b1">
          <title>Physico-chemical properties</title>
          <p>Browsing through the link ‘Physico-chemical properties’ leads to a table containing physico-chemical properties of the protein in question. In addition, four different property scales with gradient-colored arrows are provided to aid the user predict the protein’s nature, such as solubility, polarity and structural stability, at a glimpse (<xref ref-type="fig" rid="pone-0049425-g003">Figure 3</xref>). Scales are provided for the canonical variable (<italic>CV</italic><sub>sol</sub>), probability of an expressed protein entering into an inclusion body (PEPIB), GRAVY and structural stability (<xref ref-type="fig" rid="pone-0049425-g003">Figure 3</xref>). When the user browses through “Physico-chemical properties” of a protein for visualizing its properties, canonical variable for solubility (<italic>CV</italic><sub>sol</sub>), GRAVY value, PEPIB value, and structural stability are indicated on their respective scales (<xref ref-type="fig" rid="pone-0049425-g003">Figure 3</xref>). Property values of a selected protein are highlighted with blinking on the respective scales. <xref ref-type="fig" rid="pone-0049425-g003">Figure 3</xref> shows the determined physico-chemical properties of three selected proteins, Sll0649, Sll0088, and Sll0359 from <italic>Synechocystis</italic> sp. PCC 6803. PEPIB values calculated for Sll0649 and Sll0088 are 0.7 and 0.65, respectively. These values indicate that there is a high chance for these proteins to enter into inclusion body during their heterologous expression. To validate our predictions, we expressed these proteins to verify whether the calculated properties of the proteins can be relied upon and be considered for optimizing the conditions of expression, purification and characterization. Upon expression, most of the Sll0649 and Sll0088 were found in the inclusion body and a little was observed in the soluble fraction (<xref ref-type="supplementary-material" rid="pone.0049425.s001">Figure S1A and S1B</xref>). PEPIB value of Sll0359 protein is 0.1, hence it is predicted to be a soluble protein during its heterologous expression. As predicted, this protein was appeared in soluble fraction upon expression in <italic>E. coli</italic> (<xref ref-type="supplementary-material" rid="pone.0049425.s001">Figure S1C</xref>). These results are well in agreement with the calculated properties. The molar extinction coefficients calculated for the above proteins are 24750, 27310 and 7680 M<sup>−1</sup> cm<sup>−1</sup>, respectively This values can be used to determine the concentration of purified proteins by measuring their absorbance at 280 nm. Temperature is one of the factors that affect the structure of a protein. Studies show that there is a positive correlation between the structural stability and aliphatic amino acid content of proteins <xref ref-type="bibr" rid="pone.0049425-Ikai1">[10]</xref>, <xref ref-type="bibr" rid="pone.0049425-Argos1">[18]</xref>. The aliphatic index values of Sll0649, Sll0088 and Sll0359 are 95.1, 87.0 and 75.5. These values give the structural stability value of 4.8, 4.4 and 3.8 for these three proteins, respectively. This indicates that these proteins are moderately stable.</p>
          <p>The isoelectric point of a protein is an important property, because protein is least soluble at the pH near this point. There is a significant relation between pI of a protein and pH of the buffer being used for crystallization. The buffer pH equals to or very near to the pI value of a protein offers a reasonable probability of yielding crystals <xref ref-type="bibr" rid="pone.0049425-Kantardjieff1">[20]</xref>. The CyanoPhyChe database provides pI value for all the cyanobacterial proteins. Calculated pI of a cyanobacterial protein can be used to select a suitable buffer condition for its crystallization.</p>
        </sec>
        <sec id="s3b2">
          <title>Amino acid composition and structure information</title>
          <p>Amino acid composition and structural information of a cyanobacterial protein in question can be visualized by browsing through the links ‘Amino acid composition’ and ‘Structure information’, respectively. The ‘Structure information’ link navigates to the page containing predicted-secondary structure. In addition, an external link is provided to PDB database (<ext-link ext-link-type="uri" xlink:href="http://www.rcsb.org/pdb/home/home.do" xlink:type="simple">http://www.rcsb.org/pdb/home/home.do</ext-link>) for viewing the 3D structure, when it is available. If the structure is not available for a protein, then the user can look for homologous proteins whose structures are available in the PDB database, by clicking on the “BLAST” tab for building a homology model.</p>
        </sec>
        <sec id="s3b3">
          <title>Biochemical pathway in which a protein is involved</title>
          <p>If a protein is known to be involved in any biochemical pathway, an external link to KEGG database is provided under the pathway name and ID that navigates to and displays the pathway in which it is involved (<ext-link ext-link-type="uri" xlink:href="http://www.genome.jp/kegg/pathway.html" xlink:type="simple">http://www.genome.jp/kegg/pathway.html</ext-link>).</p>
        </sec>
      </sec>
      <sec id="s3c">
        <title>Search Options in the CyanoPhyChe Database</title>
        <p>Search page contains the list of cyanobacteria. User can choose one or more organisms from the list for finding proteins with desired properties. The selection leads to a query page, where search option to retrieve a protein or a group of proteins with specific properties from the selected cyanobacteria have been provided (<xref ref-type="fig" rid="pone-0049425-g004">Figure 4</xref>). User can search the database using identifiers, such as gene name, locus ID, E.C. number, function and protein ID. The search window allows the user to enter a string of values for the selected identifier to search for different proteins among the selected organisms. The user can also search for the proteins based on the properties like molecular weight, number of residues and pI value.</p>
        <p>An option is provided to search and display proteins of a cyanobacterium that fall within a given range of a property. For instance, searching the database to retrieve the proteins with pI values ranging from 8 to 9 in <italic>Synechocystis</italic> sp. PCC6803, displays the list of proteins fall within this range. User can also retrieve the cyanobacterial protein(s) by searching the database using a specific property value or combination of two or more properties listed under “Search with combination of properties” menu. This option is more useful for retrieving proteins with different combination of properties, such as molecular weight, number of residues and isoelectric point. Physicochemical properties of the listed proteins from a single or multiple cyanobacterial species, using above search criteria can be exported in the CSV format. Additionally, the user can also export the secondary structure, protein sequence and amino acid composition of the resulted proteins. These features of the database will be more useful to the researchers working on proteome analysis of any cyanobacterium.</p>
      </sec>
      <sec id="s3d">
        <title>Conclusions</title>
        <p>In summary, the database CyanoPhyChe is a collection of the calculated physico-chemical properties, solubility, and probability of an expressed protein entering into an inclusion body, structural stability, polarity and secondary structure of all cyanobacterial proteins. External links to PDB structure and KEGG pathway are provided in the database. Search option facilitates the retrieval of proteins of a particular cyanobacterium with specific property or combination of more than one property. The database also allows the user to export the retrieved data and encourages to use it for comparative studies. The data provided in the database can be used by the researchers, who are working on the cyanobacterial proteins for optimizing the methods employed for expression, purification, and characterization. The database is also useful for interpreting the results obtained from proteome analysis of cyanobacteria. CyanoPhyChe will be further updated with additional information on cellular localization of cyanobacterial proteins and physico-chemical properties of the proteins encoded by the plasmid-DNA of cyanobacteria in the upcoming versions. Further, the database will be constantly updated and curated by the authors, as and when new information is reported in the literature or communicated by the users.</p>
      </sec>
    </sec>
    <sec id="s4">
      <title>Supporting Information</title>
      <supplementary-material id="pone.0049425.s001" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xlink:href="info:doi/10.1371/journal.pone.0049425.s001" position="float" xlink:type="simple">
        <label>Figure S1</label>
        <caption>
          <p><bold>SDS-PAGE analyses of soluble and insoluble fractions of </bold><bold><italic>E. coli</italic></bold><bold> expressing the </bold><bold><italic>Synechocystis</italic></bold><bold> sp. PCC6803 proteins.</bold> Solubility of (A) Sll0649, (B) Sll0088 and (C) Sll0359 proteins. 0.4 mM IPTG (final concentration) was added to <italic>E. coli</italic> cells for inducing expression. The cells were harvested for separation of soluble and insoluble protein fractions, 2 hours after induced expression as described above. IS, Insoluble fraction; S, soluble fraction. Expressed protein bands are shown by open arrows.</p>
          <p>(DOCX)</p>
        </caption>
      </supplementary-material>
    </sec>
  </body>
  <back>
    <ack>
      <p>This study utilized the DBT - sponsored Bioinformatics Infrastructure Facility (BIF), School of Life Sciences, University of Hyderabad. We thank B. Radha Rani and Depak Singh for their help in protein expression. We also thank Mr.Goutham Pilla for his help in database improvement. We acknowledge the University Grants Commission-Special Assistance Programme-Centre for Advanced Study (UGC-SAP-CAS) programme of School of Life Sciences, University of Hyderabad for providing infrastructural facilities.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="pone.0049425-Mathura1">
        <label>1</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mathura</surname><given-names>VS</given-names></name>, <name name-style="western"><surname>Kolippakkam</surname><given-names>D</given-names></name> (<year>2005</year>) <article-title>APDbase: Amino acid Physico-chemical Properties Database</article-title>. <source>Bioinformation</source> <volume>1</volume>: <fpage>2</fpage>–<lpage>4</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Creighton1">
        <label>2</label>
        <mixed-citation publication-type="other" xlink:type="simple">Creighton TE (1993) Proteins: Structures and Molecular Properties (2nd edition), W. H. Freeman and company, New York, 105–137.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Kopito1">
        <label>3</label>
        <mixed-citation publication-type="other" xlink:type="simple">Kopito RR (2000) Aggresomes, inclusion bodies and protein aggregation. Trends Cell Biol 10, 524–530.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Fink1">
        <label>4</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Fink</surname><given-names>AL</given-names></name> (<year>1998</year>) <article-title>Protein aggregation: folding aggregates, inclusion bodies and amyloid</article-title>. <source>Folding and Design</source> <volume>3</volume>: <fpage>R9</fpage>–<lpage>R23</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Harrison1">
        <label>5</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Harrison</surname><given-names>RG</given-names></name> (<year>2000</year>) <article-title>Expression of soluble heterologus proteins via fusion with NusA protein</article-title>. <source>inNovations</source> <volume>11</volume>: <fpage>4</fpage>–<lpage>7</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Wilkinson1">
        <label>6</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wilkinson</surname><given-names>DL</given-names></name>, <name name-style="western"><surname>Harrison</surname><given-names>RG</given-names></name> (<year>1991</year>) <article-title>Predicting the solubility of recombinant proteins in Escherichia coli</article-title>. <source>Nat Biotech</source> <volume>9</volume>: <fpage>443</fpage>–<lpage>448</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-IdiculaThomas1">
        <label>7</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Idicula-Thomas</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Balaji</surname><given-names>PV</given-names></name> (<year>2004</year>) <article-title>Understanding the relationship between the primary structure of proteins and its propensity to be soluble on over expression in Escherichia coli</article-title>. <source>Protein Sci</source> <volume>14</volume>: <fpage>582</fpage>–<lpage>592</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Levene1">
        <label>8</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Levene</surname><given-names>PA</given-names></name>, <name name-style="western"><surname>Simms</surname><given-names>HS</given-names></name> (<year>1923</year>) <article-title>Calculation of iso-electric points</article-title>. <source>J Biol Chem</source> <volume>55</volume>: <fpage>801</fpage>–<lpage>813</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Gill1">
        <label>9</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gill</surname><given-names>SC</given-names></name>, <name name-style="western"><surname>Von Hippel</surname><given-names>PH</given-names></name> (<year>1989</year>) <article-title>Calculation of protein extinction coefficients from amino acid sequence data</article-title>. <source>Anal Biochem</source> <volume>182</volume>: <fpage>319</fpage>–<lpage>326</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Ikai1">
        <label>10</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ikai</surname><given-names>A</given-names></name> (<year>1980</year>) <article-title>Thermostability and aliphatic index of globular proteins</article-title>. <source>J Biochem</source> <volume>88</volume>: <fpage>1895</fpage>–<lpage>1898</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Chou1">
        <label>11</label>
        <mixed-citation publication-type="other" xlink:type="simple">Chou PY, Fasman GD (1977) Secondary structural prediction of proteins from their amino acid sequence. Trends Biochem Sci : 128–131.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Nakashima1">
        <label>12</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nakashima</surname><given-names>H</given-names></name>, <name name-style="western"><surname>Nishikawa</surname><given-names>K</given-names></name> (<year>1994</year>) <article-title>Discrimination of intercellular and extracellular proteins using amino acid composition and residue pair frequencies</article-title>. <source>J Mol Biol</source> <volume>238</volume>: <fpage>54</fpage>–<lpage>61</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Kyte1">
        <label>13</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kyte</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Doolittle</surname><given-names>R</given-names></name> (<year>1982</year>) <article-title>A simple method for displaying the hydropathic character of a protein</article-title>. <source>J Mol Biol</source> <volume>157</volume>: <fpage>105</fpage>–<lpage>132</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Rice1">
        <label>14</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rice</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Longden</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Bleasby</surname><given-names>A</given-names></name> (<year>2000</year>) <article-title>Emboss: The european molecular biology open software suite</article-title>. <source>Trends Genet</source> <volume>16</volume>: <fpage>276</fpage>–<lpage>277</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-McGuffin1">
        <label>15</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>McGuffin</surname><given-names>LJ</given-names></name>, <name name-style="western"><surname>Bryson</surname><given-names>K</given-names></name>, <name name-style="western"><surname>Jones</surname><given-names>DT</given-names></name> (<year>2000</year>) <article-title>The PSIPRED protein structure prediction server</article-title>. <source>Bioinoformatics Appl Note</source> <volume>16</volume>: <fpage>404</fpage>–<lpage>405</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Wishart1">
        <label>16</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wishart</surname><given-names>DS</given-names></name>, <name name-style="western"><surname>Arndt</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Berjanskii</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Guo</surname><given-names>AC</given-names></name>, <name name-style="western"><surname>Shi</surname><given-names>Y</given-names></name>, <etal>et al</etal>. (<year>2008</year>) <article-title>PPT-DB: the protein property prediction testing database</article-title>. <source>Nucleic Acids Res</source> <volume>36</volume>: <fpage>222</fpage>–<lpage>229</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Li1">
        <label>17</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Li</surname><given-names>ZR</given-names></name>, <name name-style="western"><surname>Lin</surname><given-names>HH</given-names></name>, <name name-style="western"><surname>Han</surname><given-names>LY</given-names></name>, <name name-style="western"><surname>Jiang</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Chen</surname><given-names>X</given-names></name>, <etal>et al</etal>. (<year>2006</year>) <article-title>PROFEAT: a web server for computing structural and physicochemical features of proteins and peptides from amino acid sequence</article-title>. <source>Nucleic Acids Res</source> <volume>34</volume>: <fpage>32</fpage>–<lpage>37</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Argos1">
        <label>18</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Argos</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Rossman</surname><given-names>MG</given-names></name>, <name name-style="western"><surname>Grau</surname><given-names>UM</given-names></name>, <name name-style="western"><surname>Zuber</surname><given-names>H</given-names></name>, <name name-style="western"><surname>Frank</surname><given-names>G</given-names></name>, <etal>et al</etal>. (<year>1979</year>) <article-title>Thermostability and protein structure</article-title>. <source>Biochemistry</source> <volume>18</volume>: <fpage>5698</fpage>–<lpage>5703</lpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Koschorreck1">
        <label>19</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Koschorreck</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Fischer</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Barth</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Pleiss</surname><given-names>J</given-names></name> (<year>2005</year>) <article-title>How to find soluble proteins: a comprehensive analysis of alpha/beta hydrolases for recombinant expression in <italic>E. coli.</italic></article-title>. <source>BMC Genomics</source> <volume>6</volume>: <fpage>49</fpage>.</mixed-citation>
      </ref>
      <ref id="pone.0049425-Kantardjieff1">
        <label>20</label>
        <mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kantardjieff</surname><given-names>KA</given-names></name>, <name name-style="western"><surname>Rupp</surname><given-names>B</given-names></name> (<year>2004</year>) <article-title>Protein isoelectric point as a predictor for increased crystallization screening efficiency</article-title>. <source>Bioinformatics</source> <volume>20</volume>: <fpage>2162</fpage>–<lpage>2168</lpage>.</mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>