<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="EN">
<front>
<journal-meta><journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id><journal-id journal-id-type="publisher-id">plos</journal-id><journal-id journal-id-type="pmc">plosone</journal-id><!--===== Grouping journal title elements =====--><journal-title-group><journal-title>PLoS ONE</journal-title></journal-title-group><issn pub-type="epub">1932-6203</issn><publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, USA</publisher-loc></publisher></journal-meta>
<article-meta><article-id pub-id-type="publisher-id">09-PONE-RA-14482R1</article-id><article-id pub-id-type="doi">10.1371/journal.pone.0013066</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="Discipline"><subject>Computational Biology</subject><subject>Computational Biology/Transcriptional Regulation</subject><subject>Genetics and Genomics/Bioinformatics</subject><subject>Genetics and Genomics/Functional Genomics</subject><subject>Genetics and Genomics/Gene Expression</subject><subject>Diabetes and Endocrinology/Obesity</subject></subj-group></article-categories><title-group><article-title>Ontology-Based Meta-Analysis of Global Collections of High-Throughput Public Data</article-title><alt-title alt-title-type="running-head">Public Data Meta-Analysis</alt-title></title-group><contrib-group>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Kupershmidt</surname><given-names>Ilya</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="corresp" rid="cor1"><sup>*</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Su</surname><given-names>Qiaojuan Jane</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Grewal</surname><given-names>Anoop</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Sundaresh</surname><given-names>Suman</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Halperin</surname><given-names>Inbal</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Flynn</surname><given-names>James</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Shekar</surname><given-names>Mamatha</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Wang</surname><given-names>Helen</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Park</surname><given-names>Jenny</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Cui</surname><given-names>Wenwu</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Wall</surname><given-names>Gregory D.</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Wisotzkey</surname><given-names>Robert</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Alag</surname><given-names>Satnam</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Akhtari</surname><given-names>Saeid</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Ronaghi</surname><given-names>Mostafa</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff3"><sup>3</sup></xref></contrib>
</contrib-group><aff id="aff1"><label>1</label><addr-line>NextBio, Cupertino, California, United States of America</addr-line>       </aff><aff id="aff2"><label>2</label><addr-line>Royal Institute of Technology (KTH), Stockholm, Sweden</addr-line>       </aff><aff id="aff3"><label>3</label><addr-line>Illumina, San Diego, California, United States of America</addr-line>       </aff><contrib-group>
<contrib contrib-type="editor" xlink:type="simple"><name name-style="western"><surname>Aziz</surname><given-names>Ramy K.</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/></contrib>
</contrib-group><aff id="edit1">Cairo University, Egypt</aff><author-notes>
<corresp id="cor1">* E-mail: <email xlink:type="simple">ilya@nextbio.com</email></corresp>
<fn fn-type="con"><p>Conceived and designed the experiments: IK. Performed the experiments: MS. Analyzed the data: IK QJS AG JF MS WC GDW MR. Contributed reagents/materials/analysis tools: IK QJS SS IH HW JP WC GDW RW SA SA MR. Wrote the paper: IK.</p></fn>
<fn fn-type="conflict"><p>All of the authors are employed by a commercial company, NextBio (with the exception of the last author, Mostafa Ronaghi, who is employed by Illumina). There are also a number of patents filed with respect to the technology and algorithms described in the article. NextBio also provides a commercial software platform in both free and paid versions. These competing interests do not alter the authors' adherence to all the PLoS ONE policies on sharing data and materials.</p></fn></author-notes><pub-date pub-type="collection"><year>2010</year></pub-date><pub-date pub-type="epub"><day>29</day><month>9</month><year>2010</year></pub-date><volume>5</volume><issue>9</issue><elocation-id>e13066</elocation-id><history>
<date date-type="received"><day>15</day><month>11</month><year>2009</year></date>
<date date-type="accepted"><day>28</day><month>7</month><year>2010</year></date>
</history><!--===== Grouping copyright info into permissions =====--><permissions><copyright-year>2010</copyright-year><copyright-holder>Kupershmidt et al</copyright-holder><license><license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution License, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions><abstract><sec>
<title>Background</title>
<p>The investigation of the interconnections between the molecular and genetic events that govern biological systems is essential if we are to understand the development of disease and design effective novel treatments. Microarray and next-generation sequencing technologies have the potential to provide this information. However, taking full advantage of these approaches requires that biological connections be made across large quantities of highly heterogeneous genomic datasets. Leveraging the increasingly huge quantities of genomic data in the public domain is fast becoming one of the key challenges in the research community today.</p>
</sec><sec>
<title>Methodology/Results</title>
<p>We have developed a novel data mining framework that enables researchers to use this growing collection of public high-throughput data to investigate any set of genes or proteins. The connectivity between molecular states across thousands of heterogeneous datasets from microarrays and other genomic platforms is determined through a combination of rank-based enrichment statistics, meta-analyses, and biomedical ontologies. We address data quality concerns through dataset replication and meta-analysis and ensure that the majority of the findings are derived using multiple lines of evidence. As an example of our strategy and the utility of this framework, we apply our data mining approach to explore the biology of brown fat within the context of the thousands of publicly available gene expression datasets.</p>
</sec><sec>
<title>Conclusions</title>
<p>Our work presents a practical strategy for organizing, mining, and correlating global collections of large-scale genomic data to explore normal and disease biology. Using a hypothesis-free approach, we demonstrate how a data-driven analysis across very large collections of genomic data can reveal novel discoveries and evidence to support existing hypothesis.</p>
</sec></abstract><funding-group><funding-statement>A major part of this work was funded by NextBio, which employs all authors (with the exception of Mostafa Ronaghi, employed by Illumina). This work was also supported in part by NIH grant R43 GM078602. The funders provided resources necessary for the development of the presented methods and technology and played a role in study design, data collection and analysis, decision to publish, and preparation of the manuscript.</funding-statement></funding-group><counts><page-count count="13"/></counts></article-meta>
</front>
<body><sec id="s1">
<title>Introduction</title>
<p>High-throughput technologies have become essential tools for biological researchers. The advent of “open biology” has led to an exponential growth of high-throughput data in publicly shared repositories, such as NCBI GEO, EBI Array Express, and the Stanford Microarray Database (SMD) <xref ref-type="bibr" rid="pone.0013066-GardinerGarden1">[1]</xref>. The billions of data points collected within these repositories provide an unprecedented opportunity for exploring and comparing molecular portraits of different biological states. However, the complex and heterogeneous nature of this exponentially growing amount of data has created a new and daunting challenge for a community wishing to explore it in a systematic and easy way.</p>
<p>A number of meta-analysis studies across multiple sets of gene expression data have led to important discoveries, such as: i) the identification of consistently and significantly deregulated genes in prostate cancer <xref ref-type="bibr" rid="pone.0013066-Rhodes1">[2]</xref>, ii) the derivation of candidate biological pathways that underlie mechanisms of carcinogenesis <xref ref-type="bibr" rid="pone.0013066-Ghosh1">[3]</xref>, and iii) the identification of lung adenocarcinoma genetic markers that correlated with patient survival <xref ref-type="bibr" rid="pone.0013066-Jiang1">[4]</xref>, among others <xref ref-type="bibr" rid="pone.0013066-Griffith1">[5]</xref>–<xref ref-type="bibr" rid="pone.0013066-Miller1">[10]</xref>. These studies typically focused on a single phenotype and identified significant differentially expressed sets of genes across multiple datasets. Conversely, an investigator-generated gene signature can be applied across large collections of high-throughput data to look for associations with various diseases, tissues, and treatments. In the landmark study by <italic>Lamb et al.</italic>, the connections between disease-associated gene expression “footprints” and gene expression profiles of different cell lines treated with diverse compounds were explored through meta-analysis of data generated on a highly standardized, single microarray platform <xref ref-type="bibr" rid="pone.0013066-Lamb1">[11]</xref>. The authors created a map that linked disease to relevant compounds by computing enrichment-based “connectivity” scores between their corresponding gene expression signatures. Public repositories, however, contain thousands of independent studies with highly heterogeneous data from different labs, platforms, and organisms. The high level of complexity makes it difficult to use by the broader scientific community.</p>
<p>Here we report the development of a novel strategy to explore the biological properties of gene sets found in global collections of public or proprietary large-scale experimental data. The size of gene sets queried can range from the tens (e.g., the results of qPCR experiments) to hundreds or even thousands (e.g., the gene signature results from microarray or next generation sequencing experiments). Using a unique combination of rank-based enrichment algorithms, ontologies, and meta-analysis techniques, we compute correlation scores between a given gene set and thousands of public studies. The output provides a ranked set of signatures and “meta-concepts” representing diseases, normal tissues, compound treatments, and genetic perturbations (gene mutations, knockouts, siRNA knockdowns) that have strong association with a gene set of interest. We applied our strategy to develop NextBio (<ext-link ext-link-type="uri" xlink:href="http://www.nextbio.com" xlink:type="simple">www.nextbio.com</ext-link>) – a data mining framework that integrates and correlates global public datasets with the user's own experimental data. As a demonstration, we used NextBio to detect and explore the biology of brown fat, to compare its expression profile to those of other tissues and cell types, and to discover connectivities with different disease states and chemical and genetic perturbations.</p>
</sec><sec id="s2">
<title>Results</title>
<sec id="s2a">
<title>Data pre-processing and correlation overview</title>
<p>Our data mining strategy can be divided into two parts. In the first part (<xref ref-type="fig" rid="pone-0013066-g001">Figure 1</xref>), semi-automated crawlers collected public data from diverse sources, such as NCBI GEO <xref ref-type="bibr" rid="pone.0013066-Edgar1">[12]</xref>, Array Express <xref ref-type="bibr" rid="pone.0013066-Brazma1">[13]</xref>, SMD <xref ref-type="bibr" rid="pone.0013066-Sherlock1">[14]</xref>, Broad Cancer Genomics <xref ref-type="bibr" rid="pone.0013066-Park1">[15]</xref>, Cancer Biomedical Informatics Grid (caBIG), and other repositories (<xref ref-type="supplementary-material" rid="pone.0013066.s001">Table S1</xref>). A data analysis step produced sets of differentially expressed gene signatures associated with each experimental or clinical comparison, such as disease versus normal (<xref ref-type="sec" rid="s4">Methods</xref>). In the final step of part one, all signatures were tagged with relevant ontology terms (<xref ref-type="fig" rid="pone-0013066-g001">Figure 1</xref>) that reflected associated tissue types, disease/phenotype, compound treatment, or genetic perturbation (e.g., gene mutation, knockout, siRNA knockdown). In the second part, rank-based enrichment statistics were applied to compute pairwise correlation scores between all signatures (<xref ref-type="fig" rid="pone-0013066-g002">Figures 2</xref> and <xref ref-type="fig" rid="pone-0013066-g003">3</xref>) followed by a meta-analysis to compute individual signature-ontology concept correlation scores (<xref ref-type="fig" rid="pone-0013066-g004">Figure 4</xref>).</p>
<fig id="pone-0013066-g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.g001</object-id><label>Figure 1</label><caption>
<title>Public data processing and analysis pipeline diagram.</title>
<p>The steps for turning public datasets into processed gene signatures include: raw data collection, sample annotation curation, data quality control, automated analysis, and manual tagging of resulting signatures with disease, tissue, compound ontology, and gene perturbation terms (tags). Curation of sample annotation includes a systematic analysis of all sample attributes that should be processed for differential expression. The data processing step converts original raw data into processed results – gene expression signatures representative of a given biological condition. The final tagging step ensures that key biological conditions associated with each signature are captured with standardized vocabulary terms, enabling downstream meta-analysis.</p>
</caption><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.g001" xlink:type="simple"/></fig><fig id="pone-0013066-g002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.g002</object-id><label>Figure 2</label><caption>
<title>Computing pairwise signature correlation scores.</title>
<p>The algorithm represented by this schematic computes an enrichment score and p-value between two ranked gene signatures. Dark red and blue colored boxes indicate genes present in both signatures; light red and blue colored boxes represent genes present in only one of the signatures. Dark lines connecting genes in each signature represent connections between genes with the same direction of regulation in both signatures. Light lines connect genes with opposite direction in two signatures.</p>
</caption><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.g002" xlink:type="simple"/></fig><fig id="pone-0013066-g003" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.g003</object-id><label>Figure 3</label><caption>
<title>Computing directionality and final correlation scores between two signatures.</title>
<p>The directional subsets are formed for both b1 and b2, and subset-subset enrichment scores are Computed for b1<sup>+</sup>b2<sup>+</sup>, b1<sup>+</sup>b2<sup>−</sup>, b1<sup>−</sup>b2<sup>+</sup>, and b1<sup>−</sup>b2<sup>−</sup>. Pairwise correlation scores for the directional subsets are positive where subsets are of the same direction and negative sign otherwise. The correlation scores of the subsets are summed up to give the final score for full set b1 versus full set b2.</p>
</caption><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.g003" xlink:type="simple"/></fig><fig id="pone-0013066-g004" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.g004</object-id><label>Figure 4</label><caption>
<title>Gene signature query against all other signatures within the system.</title>
<p>First, pairwise gene signature correlation scores (using rank-based enrichment statistics) are computed, followed by meta-analysis of individual score-tag pairs to compute overall tag scores. This two step process results in computation of direct correlations between user's defined signature and diverse biological conditions representing normal tissues and cell types, diseases, and compounds. Furthermore, overall positive or negative correlation between a signature and a concept is computed based on individual pairwise signature correlation scores. A positive correlation implies a similar up- and down-regulation of genes in each signature or signature-tag pair, while a negative correlation implies the opposite trend.</p>
</caption><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.g004" xlink:type="simple"/></fig>
<p>To ensure that data were comparable across different platforms and species, gene signature identifiers were translated using both a universal gene dictionary to map them to a standard NCBI gene reference and a cross-organism dictionary to assign them to precomputed ortholog clusters (<xref ref-type="sec" rid="s4">Methods</xref>). Gene annotations and probe definitions were regularly revised to ensure that they are up-to-date with manufacturer specifications. While issues have been reported with microarray spot definitions <xref ref-type="bibr" rid="pone.0013066-Harbig1">[16]</xref>–<xref ref-type="bibr" rid="pone.0013066-Yu1">[18]</xref>, we believe that our meta-analysis approach mitigated the effects of errors that occur on a single platform since the strength of an association was weighted by its consistency across multiple studies and platforms.</p>
<p>We collected and analyzed over 6,000 individual experiments from different public sources of large-scale experimental data. Within this collection were more than 140,000 individual samples profiled on gene expression microarrays. Out of 6,000 experiments, only 4,000 passed our extensive quality control (QC) criteria. Approximately 60% of the disqualified studies were excluded from processing due to insufficient replicates, lack of control samples, or unsupported platforms (e.g., platforms that do not cover more than half the number of genes for an organism). Approximately 25% of the disqualified studies were duplicated (e.g. part of an already processed super-series), and 15% were excluded for failing QC metrics during pre-processing and differential expression analysis (<xref ref-type="table" rid="pone-0013066-t001">Table 1</xref>, <xref ref-type="sec" rid="s4">Methods</xref>).</p>
<table-wrap id="pone-0013066-t001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t001</object-id><label>Table 1</label><caption>
<title>Summary of all data associated with normal tissues, diseases, drug treatments, and genetic perturbations.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t001-1" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t001" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">Concept Type</td>
<td align="left" colspan="1" rowspan="1">Total Studies</td>
<td align="left" colspan="1" rowspan="1">Total Signatures</td>
<td align="left" colspan="1" rowspan="1">Total Samples</td>
<td align="left" colspan="1" rowspan="1">Total Concepts</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">Normal Tissues</td>
<td align="left" colspan="1" rowspan="1">450</td>
<td align="left" colspan="1" rowspan="1">2,120</td>
<td align="left" colspan="1" rowspan="1">14,500</td>
<td align="left" colspan="1" rowspan="1">120</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Diseases</td>
<td align="left" colspan="1" rowspan="1">1,390</td>
<td align="left" colspan="1" rowspan="1">5,880</td>
<td align="left" colspan="1" rowspan="1">54,600</td>
<td align="left" colspan="1" rowspan="1">700</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Compounds</td>
<td align="left" colspan="1" rowspan="1">990</td>
<td align="left" colspan="1" rowspan="1">10,830</td>
<td align="left" colspan="1" rowspan="1">45,420</td>
<td align="left" colspan="1" rowspan="1">1,430</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Genetic Perturbations</td>
<td align="left" colspan="1" rowspan="1"><italic>1,235</italic></td>
<td align="left" colspan="1" rowspan="1"><italic>6,170</italic></td>
<td align="left" colspan="1" rowspan="1"><italic>16,400</italic></td>
<td align="left" colspan="1" rowspan="1">135</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1"><bold>Total</bold></td>
<td align="left" colspan="1" rowspan="1"><bold>4,065</bold></td>
<td align="left" colspan="1" rowspan="1"><bold>25,000</bold></td>
<td align="left" colspan="1" rowspan="1"><bold>130,920</bold></td>
<td align="left" colspan="1" rowspan="1"><bold>2,375</bold></td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt101"><label/><p>Total concepts count represents the number of specific ontology terms that are assigned as tags to signatures, or which represent “parent” ontology concepts. For example, when a gene signature is tagged with “heart ventricle”, it is automatically considered tagged with the parent term “heart” and both are considered in the counts shown. “Total studies” refers to the number of studies that contain gene signatures that contribute to a given concept based on their associated tags. “Total Signatures” and “Total Samples” refer to the number of gene signatures and individual samples contributing to a given concept type (e.g. disease).</p></fn></table-wrap-foot></table-wrap>
<p>After applying a statistical analysis to identify differentially expressed genes in each experiment, we obtained a total of 25,000 gene signatures (a typical study produces multiple results). Each signature was tagged with relevant ontology terms. This annotation step identified a total of 120 unique normal tissue concepts, 700 disease, 1,430 compound, and 135 genetic perturbation concepts (including gene mutations, knockouts, and siRNA knockdowns). The final dataset contained a high-dimensional space of gene signatures with ranked genes and associated ontology concepts (tags) for diseases, tissues, compounds, and genetic perturbations (<xref ref-type="table" rid="pone-0013066-t001">Table 1</xref>).</p>
</sec><sec id="s2b">
<title>Computing pairwise signature correlation (enrichment) scores</title>
<p>A number of factors must be considered when performing a comparative analysis of highly heterogeneous data from different sources, platforms, and technologies. We applied statistical methods to compensate for differences in platforms and their probe content, in organisms studied, and in signature sizes that could arise from choices of analysis stringency (e.g. p-value cutoffs). In addition, directional information (up- or down-regulation) is important for assessing connectivity between different gene sets derived from gene expression data. Our rank-based directional enrichment analysis enabled us to statistically assess pairwise correlations between any two gene signatures and to use this information to rank connectivity between different biological states.</p>
<p>We applied our algorithm to compute pairwise correlation scores between all signatures in our system (<xref ref-type="fig" rid="pone-0013066-g002">Figure 2</xref>). The magnitude of the pairwise correlation score reflected the similarity of the two signatures, which is measured by the extent that the genes in one signature set are enriched at the top ranks of the other signature set, and vice versa. Each signature consisted of a list of genes that passed a select fold change, p-value, or other test statistic threshold. Thresholds might vary between different researchers, types of studies, and analytical methods, and thereby result in different signature sizes. To capture the key enrichment signal, even among very short or very long signatures, we developed a non-parametric rank-based statistical approach.</p>
<p>The general design of the algorithm, which we call “Running Fisher” (see <xref ref-type="sec" rid="s4">Methods</xref>), is analogous to the Gene Set Enrichment Analysis (GSEA) method <xref ref-type="bibr" rid="pone.0013066-Lamb1">[11]</xref>, <xref ref-type="bibr" rid="pone.0013066-Subramanian1">[19]</xref>. As with GSEA, Running Fisher dynamically detects the most significant enrichment signal in a ranked signature, allowing the signature to contain a relatively more comprehensive collection of genes than would otherwise be required when using a stringent statistical cutoff. This “dynamic” enrichment detection approach overcomes the limitations of a more commonly used “selection” approach where a too stringent cutoff might lead to potential loss of significant information, and a too relaxed cutoff might include insignificant data into the evaluation <xref ref-type="bibr" rid="pone.0013066-Newton1">[20]</xref>. The Running Fisher algorithm differs from GSEA in the assessment of the statistical significance, where p-values are computed by a Fisher's exact test rather than by permutations (see <xref ref-type="sec" rid="s4">Methods</xref> for details). Overall, this approach provided us the flexibility to compute correlation scores for data of different sizes and filter thresholds, as well as the ability to use ranks in both query and target signatures.</p>
<p>The directional relationship between the two signatures was captured by the sign of the correlation score. The up-regulated genes and the down-regulated genes were separated into directional subsets, and correlation scores were computed for each directional subset from one signature against each subset from the other signature (<xref ref-type="fig" rid="pone-0013066-g003">Figure 3</xref>). A positive sign was given to a subset pair that changed expression in the same direction, and a negative sign was given to a subset pair that changed in opposite directions. The overall correlation score was the sum of directional subset scores, and the sign of the sum determined whether the two signatures were positively or negatively correlated (see <xref ref-type="sec" rid="s4">Methods</xref> for details). Using this strategy, we computed pairwise correlation scores between all 25,000 signatures, resulting in over 625 million pairwise scores.</p>
</sec><sec id="s2c">
<title>Meta-analysis to compute signature-ontology correlations</title>
<p>Currently, our system contains tens of thousands of datasets representing diverse types of biological conditions. With the development of new sequencing technologies we anticipate hundreds of thousands of public datasets to be available for the research community in the near future. To systematically interrogate these huge quantities of data, we have to abstract our analysis to the level of biological conditions those datasets represent. Researchers can then look at the connections between their own data and the potentially thousands of individual datasets with matching tissues, diseases, compounds, or genetic perturbations. Ontology-based meta-analysis is designed to accomplish that goal by computing an overall correlation score between a given gene set and an ontology concept (e.g. disease). The meta-analysis algorithm statistically assesses “reproducibility” of significant findings, thus minimizing the chance of random correlations and poor-quality data affecting the final results.</p>
<p>The meta-analysis algorithm aggregated scores for various ontology terms associated with the correlated signatures, weighted by the strength of the correlation score (<xref ref-type="fig" rid="pone-0013066-g004">Figure 4</xref>, <xref ref-type="sec" rid="s4">Methods</xref>). This was computed separately for tissue, disease, compound, and genetic perturbation categories. The algorithm considered any available hierarchical relationships of ontology terms and propagated enrichment scores to more general concepts accordingly. Concepts that may have had lower scores than their parent concepts were clustered under the parent. A ranked structure of the most relevant tissues, diseases, and compounds was thus pre-computed for each signature. The advantage of this strategy is that related ontology tags could be associated with each signature semantically. For example, “heart” and “left heart ventricle” can both contribute to the “heart” concept meta-analysis score since “heart” is the parent concept of “heart ventricle”.</p>
<p>As each new signature was added, the meta-analysis computations for existing signatures in the system were also updated. This ongoing process ensured that at a given time the most up-to-date results of the meta-analysis given the current state of the knowledge base were produced. When a query with a given set of genes was performed a total collection of meta-categories, as well as individual signatures were scanned to identify top-ranking normal tissues, diseases, and compounds (<xref ref-type="table" rid="pone-0013066-t001">Table 1</xref>).</p>
</sec><sec id="s2d">
<title>Use Case 1: Comparative Analysis across Normal Tissue and Cell Type Data</title>
<sec id="s2d1">
<title>Analysis of the molecular similarity between brown fat and other normal tissues</title>
<p>A large number of experiments in the public domain provide a great resource for exploring normal tissue biology. It is virtually impossible to create a single comprehensive dataset with gene expression profiles of all tissues and cell types of interest. However, normal tissue datasets from hundreds of independent studies can be scanned using gene sets of interest to identify similarities. This can further our understanding of the relationships between different tissues, different stem cell lineages, as well as mechanisms governing normal and aberrant developmental pathways.</p>
<p>We applied our strategy to investigate molecular properties of brown fat cells and to explore their similarity to a collection of other normal tissues. To achieve this we derived a brown fat tissue gene expression signature from the mouse tissue atlas dataset containing genome-wide gene expression profiles of 61unique tissues and organs (NCBI GEO Accession # GSE1133) <xref ref-type="bibr" rid="pone.0013066-Su1">[21]</xref>. Each gene was ranked according to its fold change relative to the median of all mouse tissues (see <xref ref-type="sec" rid="s4">Methods</xref>). As a result we obtained a tissue signature that consisted of 31,309 probe sets and their associated ranks.</p>
<p>We then used our rank-based enrichment analysis to compute pairwise correlation scores between brown fat and signatures across all studies contained in NextBio (<ext-link ext-link-type="uri" xlink:href="http://www.nextbio.com" xlink:type="simple">www.nextbio.com</ext-link>, <xref ref-type="table" rid="pone-0013066-t001">Table 1</xref>). A meta-analysis of the pairwise signature correlations then computed the correlation of the brown fat signature with the tags of all target signatures (<xref ref-type="fig" rid="pone-0013066-g005">Figure 5A</xref>). The majority of ontology concepts were associated with multiple signatures derived from different studies and organisms. As shown in <xref ref-type="table" rid="pone-0013066-t002">Table 2</xref>, skeletal muscle tissue produced the strongest positive correlation with the brown fat tissue signature. A positive correlation indicates that predominantly the same sets of genes are either up- or down-regulated in the query and target set of signatures (in this case a total of 4 signatures). Skeletal muscle had the top ranked correlation score to brown fat out of 120 total tissue concepts computed from over 2,000 signatures (<xref ref-type="table" rid="pone-0013066-t001">Table 1</xref>). Skeletal muscle data used in the meta-analysis was generated from mouse, rat, and human tissues, providing evidence that the results are consistent across species and platforms. Of interest, these data indicate that brown fat cells are more closely related to muscle than to white adipose tissue, which failed to produce significant correlation with muscle tissue concepts for the same query (<xref ref-type="supplementary-material" rid="pone.0013066.s002">Table S2</xref>). In a recently published study, Seale <italic>et al.</italic> demonstrated that brown fat cell precursors can turn into muscle cells upon the loss of PRDM16 protein <xref ref-type="bibr" rid="pone.0013066-Seale1">[22]</xref>. Other studies have also shown that brown fat and muscle tissue share important molecular characteristics, thus validating our approach <xref ref-type="bibr" rid="pone.0013066-Timmons1">[23]</xref>, <xref ref-type="bibr" rid="pone.0013066-Atit1">[24]</xref>.</p>
<fig id="pone-0013066-g005" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.g005</object-id><label>Figure 5</label><caption>
<title>Brown fat meta-analysis.</title>
<p>Diagram representing analyses of two different brown fat related signatures: (a) Brown fat tissue signature (relative to all other mouse tissues). (b) Signature of brown preadipocytes vs. white preadipocytes. After computing pairwise scores between query and all target signatures the meta-analysis of pairwise scores and their associated tags (associated disease, tissue, and compound terms) is performed. The final result produces a ranked set of tissues, diseases, and compounds with the most significant association to query signature.</p>
</caption><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.g005" xlink:type="simple"/></fig><table-wrap id="pone-0013066-t002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t002</object-id><label>Table 2</label><caption>
<title>Brown fat tissue signature query results.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t002-2" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t002" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">Rank</td>
<td align="left" colspan="1" rowspan="1">Normal tissue</td>
<td align="left" colspan="1" rowspan="1">Correlation Direction</td>
<td align="left" colspan="1" rowspan="1">Correlation Score</td>
<td align="left" colspan="1" rowspan="1"># Correlated/ Total Studies</td>
<td align="left" colspan="1" rowspan="1"># Total Correlated Signatures</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">1</td>
<td align="left" colspan="1" rowspan="1">Skeletal muscle tissue</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">100</td>
<td align="left" colspan="1" rowspan="1">4/4</td>
<td align="left" colspan="1" rowspan="1">4/4</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">2</td>
<td align="left" colspan="1" rowspan="1">Tongue</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">93.8</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
<td align="left" colspan="1" rowspan="1">5/5</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">3</td>
<td align="left" colspan="1" rowspan="1">Epidermis</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">93.3</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">4</td>
<td align="left" colspan="1" rowspan="1">Cardiac atrium</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">88.9</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">5</td>
<td align="left" colspan="1" rowspan="1">Duodenum</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">85.9</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt102"><label/><p>The gene expression signature from brown fat tissue was queried against all studies corresponding to normal normal tissues from different microarray platforms and organisms. The results for top five tissues with the biggest positive correlation to brown fat signature are shown. Additional results are shown in <xref ref-type="supplementary-material" rid="pone.0013066.s002">Tabs S2</xref> of the Supporting Information section.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s2e">
<title>Analysis of a brown preadipocyte signature</title>
<p>Seale <italic>et al.</italic> identified new progenitor cells that gave rise to brown fat and muscle cells but not to white fat cells <xref ref-type="bibr" rid="pone.0013066-Seale1">[22]</xref>. Currently, however, there is no gene expression data available for these new precursor cells. Given that, we decided to investigate the molecular properties of another brown fat cell precursor and its molecular similarity with normal tissues and cell types. We derived a brown preadipocytes signature consisting of 2,302 probesets (mouse MG_U74Av2 Affymetrix chip) by comparing gene expression of brown to white preadipocytes (GEO Accession # GSE7032, <xref ref-type="supplementary-material" rid="pone.0013066.s004">Table S4</xref>) <xref ref-type="bibr" rid="pone.0013066-Timmons1">[23]</xref>. Using this signature, we performed a correlation analysis across all normal tissues and cell types in NextBio (<xref ref-type="fig" rid="pone-0013066-g005">Figure 5B</xref>), and found that muscle stem cells were among the top five concepts with the strongest positive correlation scores to the brown preadipocytes signature (<xref ref-type="table" rid="pone-0013066-t003">Table 3</xref>). This result further contrasts the differences between brown and white fat cells and demonstrates the association of muscle to brown fat. We continued our analysis by looking for patterns as brown and white adipocytes undergo differentiation, and we continued to observe a pattern of positive correlation between brown adipocyte and muscle cell precursors. In accordance with results in Seale <italic>et al.</italic>, our meta-analysis approach suggests the possibility of the existence of a common precursor cell for these two cell types <xref ref-type="bibr" rid="pone.0013066-Seale1">[22]</xref>.</p>
<table-wrap id="pone-0013066-t003" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t003</object-id><label>Table 3</label><caption>
<title>Brown versus white preadipocytes signature correlation with normal tissues and cell types.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t003-3" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t003" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">Normal tissue</td>
<td align="left" colspan="1" rowspan="1">Correlation Direction</td>
<td align="left" colspan="1" rowspan="1">Correlation Score</td>
<td align="left" colspan="1" rowspan="1"># Correlated/ Total Studies</td>
<td align="left" colspan="1" rowspan="1"># Total Correlated Signatures</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">Embryonic tissue</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">89.0</td>
<td align="left" colspan="1" rowspan="1">7/7</td>
<td align="left" colspan="1" rowspan="1">17/17</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Muscle stem cell</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">88.7</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Hair follicle matrix</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">87.1</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">T-helper type 1 cells</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">82.2</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Embryonic Stem cells</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">82.1</td>
<td align="left" colspan="1" rowspan="1">15/18</td>
<td align="left" colspan="1" rowspan="1">26/32</td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt103"><label/><p>Query results for gene expression signature differentiating brown and white preadipocytes across normal tissue signatures. Positive correlation scores for the top four tissues whose expression signatures correlate with brown vs. white preadipocytes signature (for additional results see <xref ref-type="supplementary-material" rid="pone.0013066.s005">Table S5</xref>, Supporting Information section).</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2f">
<title>Use Case 2: Comparative Analysis across Disease Related Data</title>
<sec id="s2f1">
<title>Analysis of brown fat tissue and correlation with disease signatures</title>
<p>Using public data, we generated thousands of signatures representing 700 distinct disease states. This large collection of disease profiles provides a rich contextual framework with which to explore gene sets of interest. Analysis of tissue- and cell type-specific gene sets against these disease-state profiles can unveil abnormal cell- and tissue-specific programs that are involved in disease development. Analysis of gene sets derived from specific patient cohorts against specific disease signatures can classify a disease more precisely and help drive patient stratification and trial selection <xref ref-type="bibr" rid="pone.0013066-Bild1">[25]</xref>, <xref ref-type="bibr" rid="pone.0013066-Noushmehr1">[26]</xref>.</p>
<p>As an example, we investigated the relationship of brown fat tissue to all disease tissue signatures. For the purposes of this study we focused on disease concepts that had a negative correlation to brown fat signature. Among the top ranking disease states, we found obesity, quadriplegia, aging, Duchenne muscular dystrophy, and myocardial infarction (<xref ref-type="table" rid="pone-0013066-t004">Table 4</xref>). The negative correlation to obesity provided a positive control, as brown fat functions in energy expenditure and is associated with resistance to obesity in diverse mouse strains <xref ref-type="bibr" rid="pone.0013066-Almind1">[27]</xref>. Furthermore, the proportion of white to brown fat is significantly increased in obese versus normal subjects <xref ref-type="bibr" rid="pone.0013066-Farmer1">[28]</xref>.</p>
<table-wrap id="pone-0013066-t004" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t004</object-id><label>Table 4</label><caption>
<title>Correlation between brown fat and muscle tissue signatures with diseases.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t004-4" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t004" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">A. Query: Brown Fat</td>
<td align="left" colspan="1" rowspan="1">Rank</td>
<td align="left" colspan="1" rowspan="1">Disease</td>
<td align="left" colspan="1" rowspan="1">Correlation Direction</td>
<td align="left" colspan="1" rowspan="1">Correlation Score</td>
<td align="left" colspan="1" rowspan="1"># Correlated/ Total Studies</td>
<td align="left" colspan="1" rowspan="1"># Total Correlated Signatures</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">Brown fat</td>
<td align="left" colspan="1" rowspan="1">1</td>
<td align="left" colspan="1" rowspan="1">Obesity</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">100</td>
<td align="left" colspan="1" rowspan="1">8/14</td>
<td align="left" colspan="1" rowspan="1">27/46</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Brown fat</td>
<td align="left" colspan="1" rowspan="1">2</td>
<td align="left" colspan="1" rowspan="1">Quadriplegia</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">78.2</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Brown fat</td>
<td align="left" colspan="1" rowspan="1">3</td>
<td align="left" colspan="1" rowspan="1">Aging</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">73.9</td>
<td align="left" colspan="1" rowspan="1">27/30</td>
<td align="left" colspan="1" rowspan="1">36/48</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Brown fat</td>
<td align="left" colspan="1" rowspan="1">4</td>
<td align="left" colspan="1" rowspan="1">Duchenne Muscular Dystrophy (DMD)</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">86.9</td>
<td align="left" colspan="1" rowspan="1">7/8</td>
<td align="left" colspan="1" rowspan="1">14/16</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Brown fat</td>
<td align="left" colspan="1" rowspan="1">5</td>
<td align="left" colspan="1" rowspan="1">Myocardial infarction</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">84.1</td>
<td align="left" colspan="1" rowspan="1">5/5</td>
<td align="left" colspan="1" rowspan="1">25/26</td>
</tr>
</tbody>
</table><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">B. Query: Skeletal Muscle</td>
<td align="left" colspan="1" rowspan="1"/>
<td align="left" colspan="1" rowspan="1"/>
<td align="left" colspan="1" rowspan="1"/>
<td align="left" colspan="1" rowspan="1"/>
<td align="left" colspan="1" rowspan="1"/>
<td align="left" colspan="1" rowspan="1"/>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">Skeletal muscle</td>
<td align="left" colspan="1" rowspan="1">1</td>
<td align="left" colspan="1" rowspan="1">Quadriplegia</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">100</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Skeletal muscle</td>
<td align="left" colspan="1" rowspan="1">2</td>
<td align="left" colspan="1" rowspan="1">Duchenne Muscular Dystrophy (DMD)</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">99.9</td>
<td align="left" colspan="1" rowspan="1">7/8</td>
<td align="left" colspan="1" rowspan="1">14/16</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Skeletal muscle</td>
<td align="left" colspan="1" rowspan="1">3</td>
<td align="left" colspan="1" rowspan="1">Oral cancer</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">89.5</td>
<td align="left" colspan="1" rowspan="1">2/2</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Skeletal muscle</td>
<td align="left" colspan="1" rowspan="1">4</td>
<td align="left" colspan="1" rowspan="1">Nerve injury</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">78.1</td>
<td align="left" colspan="1" rowspan="1">4/4</td>
<td align="left" colspan="1" rowspan="1">10/14</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Skeletal muscle</td>
<td align="left" colspan="1" rowspan="1">5</td>
<td align="left" colspan="1" rowspan="1">Myocardial infarction</td>
<td align="left" colspan="1" rowspan="1">-</td>
<td align="left" colspan="1" rowspan="1">77.2</td>
<td align="left" colspan="1" rowspan="1">6/6</td>
<td align="left" colspan="1" rowspan="1">24/27</td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt104"><label/><p>Brown fat and muscle normal tissue signatures queried against all disease-related signatures in different studies and organisms. Top diseases with negatively correlated genes to brown fat and muscle are shown (for additional results see <xref ref-type="supplementary-material" rid="pone.0013066.s003">Table S3</xref>, Supporting Information section).</p></fn></table-wrap-foot></table-wrap>
<p>The strong negative correlation of the brown fat signature with aging (<xref ref-type="table" rid="pone-0013066-t004">Table 4</xref>) may be partly explained by the potential age-related suppression of pathways leading to brown fat cell production. This is supported by the fact that that brown tissue deposits are more abundant in fetuses and newborns, but are less prominent in adults <xref ref-type="bibr" rid="pone.0013066-Gesta1">[29]</xref>. The strong negative correlation with quadriplegia, Duchenne muscular dystrophy, and myocardial infarction is consistent with our earlier findings of molecular similarities between brown fat and muscle tissues. Also, normal tissue-specific gene expression is suppressed in the atrophied muscles associated with these disease phenotypes <xref ref-type="bibr" rid="pone.0013066-Lehnert1">[30]</xref>.</p>
</sec></sec><sec id="s2g">
<title>Use Case 3: Comparative Analysis across Chemical Perturbations Data</title>
<sec id="s2g1">
<title>Brown preadipocytes differentiation signature positively correlates with reversine</title>
<p>A comparative analysis of gene signatures derived from a large collection of compound treatment experiments can identify those compounds with similar biological properties, pinpoint treatments with toxic side-effects, and discover novel indications for existing compounds <xref ref-type="bibr" rid="pone.0013066-Lamb1">[11]</xref>. Furthermore, by exploring a large collection of compound signatures, investigators can identify chemical perturbations that can activate or deactivate cell type-specific differentiation programs and use them as additional tools in future experiments.</p>
<p>To explore compounds that may affect differentiation of brown preadipocytes, we first derived a differentiation signature of 2,000 probe sets by comparing mature brown adipocytes to brown preadipocytes (<xref ref-type="supplementary-material" rid="pone.0013066.s006">Table S6</xref>) <xref ref-type="bibr" rid="pone.0013066-Timmons1">[23]</xref>. We then queried this signature against all compound-related data. The strongest positive correlation discovered was with the signature of the small molecule reversine (<xref ref-type="table" rid="pone-0013066-t005">Table 5</xref>). Interestingly, Kim <italic>et al.</italic> demonstrated that reversine stimulates adipocyte differentiation in 3T3-L1 cells <xref ref-type="bibr" rid="pone.0013066-Kim1">[31]</xref>. There is also a strong positive correlation between the signatures of mature brown fat and reversine (<xref ref-type="supplementary-material" rid="pone.0013066.s007">Table S7</xref>) <xref ref-type="bibr" rid="pone.0013066-Lee1">[32]</xref>. This suggests that reversine may be a useful compound in future studies of brown fat and may act to stimulate brown preadipocytes differentiation into mature cells.</p>
<table-wrap id="pone-0013066-t005" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t005</object-id><label>Table 5</label><caption>
<title>Brown mature adipocytes signature correlation with compounds.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t005-5" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t005" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">Compound</td>
<td align="left" colspan="1" rowspan="1">Correlation Direction</td>
<td align="left" colspan="1" rowspan="1">Correlation Score</td>
<td align="left" colspan="1" rowspan="1"># Correlated/Total Studies</td>
<td align="left" colspan="1" rowspan="1"># Total Correlated Signatures</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">Reversine</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">100</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Dasatinib</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">79.8</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">9/9</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Matrigel</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">79.6</td>
<td align="left" colspan="1" rowspan="1">7/8</td>
<td align="left" colspan="1" rowspan="1">24/26</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Gentamicin</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">79.5</td>
<td align="left" colspan="1" rowspan="1">3/5</td>
<td align="left" colspan="1" rowspan="1">12/16</td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt105"><label/><p>Top five query results for gene expression signature comparing mature brown adipocytes to differentiating brown preadipocytes across all signatures tagged with “Compounds” category (for additional results see <xref ref-type="supplementary-material" rid="pone.0013066.s006">Table S6</xref>, Supporting Information section).</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s2h">
<title>Use Case 4: Comparative Analysis across Genetic Perturbations Data</title>
<sec id="s2h1">
<title>Analysis of brown versus white preadipocytes signatures</title>
<p>Genetic perturbation experiments represent animal or cell line models in which a gene was deleted, modified, or silenced using transcript-specific siRNAs. Identifying genes whose perturbation causes similar gene expression changes as found in the target condition might help reveal common, key mechanisms involved in the regulation of processes leading to normal and disease development.</p>
<p>To identify genetic perturbations that resulted in altered gene expression patterns similar to the brown preadipocyte signature, we again used the 2,302 probe set derived by comparing the gene expression profiles of brown and white preadipocytes (<xref ref-type="supplementary-material" rid="pone.0013066.s004">Table S4</xref>) <xref ref-type="bibr" rid="pone.0013066-Timmons1">[23]</xref>. A query against all genetic perturbation experiments in NextBio (1,235 datasets containing 6,170 gene signatures for a total of 135 perturbed gene products) revealed that perturbations of <italic>SNF5</italic> correlated most positively and those of <italic>MYC</italic> correlated most negatively (<xref ref-type="table" rid="pone-0013066-t006">Table 6</xref>). Positive correlation between brown vs. white preadipocytes signature and <italic>SNF5</italic> perturbation implies that ablation of <italic>SNF5</italic> function induces gene expression changes that are positively correlated with white preadipocytes. Current literature supports the notion that the <italic>SNF5</italic> gene positively regulates adipocyte differentiation during adipogenesis <xref ref-type="bibr" rid="pone.0013066-Caramel1">[33]</xref>. Our results suggest that <italic>SNF5</italic> expression may direct cell fate towards white preadipocyte differentiation. The negative correlation with <italic>MYC</italic> gene perturbations indicates that <italic>MYC</italic> regulated pathways may positively regulate brown adipocyte differentiation as compared to white adipocytes. A number of reports suggest that overexpression of <italic>MYC</italic> suppresses adipogenesis and that its deletion can stimulate accumulation of white fat in pancreas <xref ref-type="bibr" rid="pone.0013066-Freytag1">[34]</xref>, <xref ref-type="bibr" rid="pone.0013066-Bonal1">[35]</xref>. Overall, we find that our system allows a deeper look at the regulatory networks involved in regulating brown and white preadipocyte differentiation into mature cell types.</p>
<table-wrap id="pone-0013066-t006" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0013066.t006</object-id><label>Table 6</label><caption>
<title>Brown versus white preadipocytes signature correlation with genetic perturbations.</title>
</caption><!--===== Grouping alternate versions of objects =====--><alternatives><graphic id="pone-0013066-t006-6" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.t006" xlink:type="simple"/><table><colgroup span="1"><col align="left" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/><col align="center" span="1"/></colgroup>
<thead>
<tr>
<td align="left" colspan="1" rowspan="1">Perturbed Gene</td>
<td align="left" colspan="1" rowspan="1">Correlation Direction</td>
<td align="left" colspan="1" rowspan="1">Correlation Score</td>
<td align="left" colspan="1" rowspan="1"># Correlated/Total Studies</td>
<td align="left" colspan="1" rowspan="1"># Total Correlated Signatures</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left" colspan="1" rowspan="1">SMARCB1 (SNF5)</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">100</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">Tcrb</td>
<td align="left" colspan="1" rowspan="1">+</td>
<td align="left" colspan="1" rowspan="1">96</td>
<td align="left" colspan="1" rowspan="1">1/1</td>
<td align="left" colspan="1" rowspan="1">10/11</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">MYC</td>
<td align="left" colspan="1" rowspan="1">−</td>
<td align="left" colspan="1" rowspan="1">95</td>
<td align="left" colspan="1" rowspan="1">21/23</td>
<td align="left" colspan="1" rowspan="1">44/65</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">MYOD1</td>
<td align="left" colspan="1" rowspan="1">−</td>
<td align="left" colspan="1" rowspan="1">95</td>
<td align="left" colspan="1" rowspan="1">5/5</td>
<td align="left" colspan="1" rowspan="1">20/28</td>
</tr>
<tr>
<td align="left" colspan="1" rowspan="1">RHO</td>
<td align="left" colspan="1" rowspan="1">−</td>
<td align="left" colspan="1" rowspan="1">95</td>
<td align="left" colspan="1" rowspan="1">3/3</td>
<td align="left" colspan="1" rowspan="1">8/9</td>
</tr>
</tbody>
</table></alternatives><table-wrap-foot><fn id="nt106"><label/><p>Top five query results for gene expression signature comparing brown to white preadipocytes across all signatures tagged with “Genetic Perturbation” category (for additional results see <xref ref-type="supplementary-material" rid="pone.0013066.s008">Table S8</xref>, Supporting Information section).</p></fn></table-wrap-foot></table-wrap></sec></sec></sec><sec id="s3">
<title>Discussion</title>
<p>In this study we presented a novel strategy for mining global collections of large-scale biological data using a combination of ranked-based enrichment statistics and ontology-based meta-analysis. We have implemented our approach within the NextBio platform (<ext-link ext-link-type="uri" xlink:href="http://www.nextbio.com" xlink:type="simple">www.nextbio.com</ext-link>) and demonstrated how researchers can use their own gene sets of interest to perform queries in the context of the globally available public data. In a series of case studies, we showed how this strategy can be used to scan thousands of microarray experiments in the public domain against brown fat-related gene sets to glean insight into adipose tissue biology. These adipose-related gene sets were analyzed within the context of gene expression data from normal and disease tissue comparisons, chemical compound studies, and genetic perturbation experiments. Insights drawn from these case studies were consistent with previously published results and also provided a novel foundation for the formation of new hypotheses.</p>
<p>Two key factors driving the significance of our data-driven <italic>in silico</italic> analysis are the sheer volume of data that we independently correlated and ranked with brown fat-related gene sets (over 4,000 experiments comprising 25,000 signatures) and the replication of observed correlations across multiple independent datasets. Using brown fat-derived gene sets, we demonstrated the strategy of exploring tissue development, cell-type specific expression, and disease etiology. Furthermore, we demonstrated the discovery of compounds and genetic perturbations that could potentially influence gene expression programs involved in adipocyte differentiation. The identification of compounds and genetic perturbations also furthers our understanding of cell type-specific expression and helps in designing new experiments to study white and brown fat biology.</p>
<p>Our strategy also provides a method that addresses, at multiple levels, the data quality concerns that are often raised with respect to publicly available data. First, the data goes through rounds of preprocessing, quality control, and curation. Second, all analysis results are rank-ordered according to enrichment statistics. Finally, the meta-analysis framework ensures that the majority of findings are supported by multiple, independent datasets, which significantly increases the overall confidence of our results.</p>
<p>The brown fat case study represents a hypothesis-generation strategy that can be applied to a variety of biological questions. As the amount of large-scale public data continues to grow, such data-driven <italic>in silico</italic> analyses are becoming increasingly important and provide a complementary methodology to traditional hypothesis-driven research. Similar strategies can also be applied to study the function of genes, pathways, and other biological entities of interest. Additionally, within clinical research settings, it can be used to study common and distinct genomic signatures of different patient cohorts or to identify novel drug indications, among other applications.</p>
<p>The ontology-based meta-analysis strategy presented here enables a higher order view of biological connections within the combined corpus of public and user-generated data. As thousands of new datasets become available, such meta-level analyses can provide a practical way to mine vast quantities of diverse large-scale datasets. Microarray-based gene expression data is the obvious starting point, given the large amount of public data that has become available in the last several years. The next logical step for our strategy would be to extend the current framework to incorporate orthogonal data types generated by proteomics, SNP genotyping, and next-generation sequencing platforms. The combination of orthogonal data will ultimately provide a broader view of biological systems and enable comprehensive <italic>in silico</italic> investigations to take place.</p>
</sec><sec id="s4" sec-type="methods">
<title>Methods</title>
<sec id="s4a">
<title>Raw data pre-processing</title>
<p>A majority of studies currently processed within NextBio adhere to certain criteria for inclusion:</p>
<list list-type="bullet"><list-item>
<p>Comprehensive coverage of genes - The platform used should contain over 12,000 probes for human, mouse, or rat studies. For all other organisms, the array should contain at least half the number of probes as there are estimated genes in the genome.</p>
</list-item><list-item>
<p>Presence of a baseline or control group.</p>
</list-item><list-item>
<p>Access to raw or normalized expression values</p>
</list-item><list-item>
<p>Sample annotations provided</p>
</list-item></list>
<p>To ensure standardization in the processing pipeline, studies were processed from raw data whenever available and from pre-processed data otherwise. For example, for Affymetrix-based studies in which CEL files are available, RMA normalization was applied <xref ref-type="bibr" rid="pone.0013066-Parrish1">[36]</xref>. Otherwise, expression summary intensities, such as those processed generated by MAS5 (Affymetrix) or dChip were processed <xref ref-type="bibr" rid="pone.0013066-Li1">[37]</xref>. All datasets went through appropriate processing steps that depended on the data type, platform, and experimental design used to generate the data and include:</p>
<list list-type="bullet"><list-item>
<p>Background subtraction, if applicable</p>
</list-item><list-item>
<p>Expression summarization, e.g. using RMA when CEL data is available</p>
</list-item><list-item>
<p>Data transformation (log) and technical replicate averaging and negative value correction</p>
</list-item><list-item>
<p>Normalization – RMA, per-chip median or Lowess where applicable</p>
</list-item><list-item>
<p>Quality control assessment</p>
</list-item><list-item>
<p>Statistical (differential expression) analysis</p>
</list-item></list>
<p>The vast majority of processed data in the system falls into the category of case-control experimental design analyzed using Welch or standard t-tests, paired or unpaired, as appropriate. Quality assessment methods were employed to review sample-level and dataset-level integrity – these included curator review of pre- and post-normalization boxplots, missing value counts, and p-value histograms (after statistical testing) with FDR analysis to determine whether the number of significantly changing genes is greater than expected by chance.</p>
<p>A p-value significance cutoff of 0.05 (without any multiple testing correction) and a minimum absolute fold-change cutoff of 1.2 (typically the lowest sensitivity threshold of commercial microarray platforms) was used to obtain the final set of signatures of differentially-expressed genes. This double filtering procedure serves to address different aspects of variability in the data <xref ref-type="bibr" rid="pone.0013066-Kittleson1">[38]</xref>–<xref ref-type="bibr" rid="pone.0013066-Quinn1">[40]</xref>. To address the potential unreliability of the fold-change metric at low intensity levels <xref ref-type="bibr" rid="pone.0013066-Tusher1">[41]</xref>, genes with signals lower than a 20<sup>th</sup> percentile cutoff in both control and test groups were discarded from the signature.</p>
<p>Expression profiles can vary considerably from study to study and from platform to platform. Different platform technologies can yield different dynamic ranges, distributions of fold-changes, and p-values that reflect the technologies used. To allow inter-study comparability, a non-parametric approach was established so that ranks were assigned to each final gene signature based on the magnitude of fold change. Fold-change, as a ranking metric, had a better concordance across platforms than p-values from statistical tests <xref ref-type="bibr" rid="pone.0013066-Shi1">[42]</xref>. Ranks were then further normalized to eliminate any bias due to varying platform sizes.</p>
<p>In the absence of a “gold standard” for processing microarray data, these statistical threshold cutoffs serve to maintain a reasonable and consistent level of data quality across all studies analyzed within NextBio and are commonly adopted in the literature <xref ref-type="bibr" rid="pone.0013066-Wang1">[43]</xref>–<xref ref-type="bibr" rid="pone.0013066-Grigoryev1">[45]</xref>. The thresholds are intentionally permissive to ensure that signatures contain <italic>all</italic> potentially interesting elements. The potential for introducing noise, i.e., more false positives, is balanced by (a) enforcing the basic quality control metrics described above and (b) incorporating a normalized rank-based scheme that captures the relative importance of each gene in a signature. This key metric of the meta-analysis framework is described below. In summary, this strategy results in the following advantages:</p>
<list list-type="order"><list-item>
<p>The normalized ranking approach enables comparability across data from different studies, platforms, and analysis methods by removing dependence on absolute values of fold-change, minimizing some of the effects of normalization methods used, and accounting for platform effects.</p>
</list-item><list-item>
<p>During pair-wise comparison of signatures, the <italic>Running Fisher</italic> algorithm (described below) dynamically determines the best cutoffs corresponding to the maximal similarity score by scanning all of the potentially interesting data. Most of the time, low ranking genes do not contribute to the maximal score, thus reducing the dependence on the minor variations of the actual cutoffs used in generating biosets.</p>
</list-item><list-item>
<p>A meta-analysis identifies genes with consistent signals across several experiments. This rescues potentially interesting gene signatures that might otherwise have fallen below the margin of significance in an analysis based on a single study.</p>
</list-item></list>
</sec><sec id="s4b">
<title>Cross-platform comparisons</title>
<p>An index of microarray platforms was compiled to aid in the comparison of microarray data. The index provides a standardized mapping of commonly used public- and vendor-specific vendor gene identifiers to reference identifiers such as NCBI Entrez Gene, UniGene, Ensembl, RefSeq, or GenBank accession numbers.</p>
</sec><sec id="s4c">
<title>Cross-species comparisons</title>
<p>To enable seamless comparison across different species, orthologs were identified for each pair of organisms and were grouped into ortholog clusters. Ortholog information was derived from Mouse Genome Informatics (MGI) at Jackson Lab (<ext-link ext-link-type="uri" xlink:href="http://www.informatics.jax.org" xlink:type="simple">http://www.informatics.jax.org</ext-link>), HomoloGene at NCBI (<ext-link ext-link-type="uri" xlink:href="http://www.ncbi.nlm.nih.gov" xlink:type="simple">http://www.ncbi.nlm.nih.gov</ext-link>), and Ensembl (<ext-link ext-link-type="uri" xlink:href="http://www.ensembl.org" xlink:type="simple">http://www.ensembl.org</ext-link>). Ortholog clusters were generated as follows: 1) the manually curated pairwise ortholog data among human, mouse, and rat from MGI were retrieved and clustered to form initial ortholog clusters. 2) The homology group data among human, mouse, rat, fly, and worm were analyzed to remove those in conflict with MGI data. The filtered homology group data were then entered into the ortholog clusters. 3) The whole genome pairwise sequence similarity data from Ensembl were processed to identify reciprocal best hits as candidate orthologs for all pairwise organisms among human, mouse, rat, fly, worm, and yeast. The candidate orthologs were prioritized based on the percentage sequence identity and examined against the existing ortholog cluster. Qualified ortholog candidates were then entered into the ortholog cluster.</p>
</sec><sec id="s4d">
<title>Computing pairwise correlation scores between gene signatures</title>
<p>The directional relationship between the two signatures is captured by the sign of the correlation score. The up-regulated genes (b<sup>+</sup>) and the down-regulated genes (b<sup>−</sup>) are separated into directional subsets, and correlation scores are computed for each directional subset from one signature (b1<sup>+</sup>, b1<sup>−</sup>) against each subset from the other signature (b2<sup>+</sup>, b2<sup>−</sup>). A positive sign is given to a subset pair of the same direction (b1<sup>+</sup>b2<sup>+</sup>, b1<sup>−</sup>b2<sup>−</sup>), and a negative sign is given to a subset pair of opposite directions (b1<sup>+</sup>b2<sup>−</sup>, b1<sup>−</sup>b2<sup>+</sup>). The overall correlation score is the sum of directional subset scores and the sign of the sum determines whether the two signatures are positively or negatively correlated (<xref ref-type="fig" rid="pone-0013066-g003">Figure 3</xref>). The matching genes between two typical gene signatures are depicted in <xref ref-type="fig" rid="pone-0013066-g002">Figure 2</xref>. The directionality and ranks for each gene is shown.</p>
<p>The detailed steps given two gene signature sets (<italic>b1, b2</italic>) are as follows:</p>
<p>First, each gene signature set is rank-ordered according to fold change, p-value or a particular score. If appropriate metrics are not provided, then the gene signature set is unranked. The up-regulated genes and down regulated genes are noted with positive and negative signs to imply directionality, respectively. A directional subset is generated for each direction, such as <italic>b1<sup>+</sup></italic>, <italic>b1<sup>−</sup></italic>, <italic>b2<sup>+</sup></italic>, and <italic>b2</italic><sup>−</sup> (<xref ref-type="fig" rid="pone-0013066-g003">Figure 3</xref>). If no directional data are provided, then the gene signature set is not directional and only one subset is formed with the whole signature set, such as <italic>b1<sup>o</sup></italic>, or <italic>b2<sup>o</sup></italic>.</p>
<p>Second, all the subset pairs are identified: <italic>b1Di</italic>, <italic>b2Dj</italic>, where <italic>Di</italic> and <italic>Dj</italic> are the available directions (+, −, or o) in <italic>b1</italic> and <italic>b2</italic>, respectively. The Running Fisher algorithm is applied to each subset pair. The top ranking genes in the first subset <italic>b1Di</italic> are collected as a group <italic>G</italic>, and the second subset <italic>b2Dj</italic> is scanned top to bottom in the rank order to identify each rank with a gene matching a member in the group <italic>G</italic>. If the subset is unranked, all the genes in the subset are retrieved at the first scan.</p>
<p>At each matching rank <italic>K</italic>, the scanned portion of the second subset <italic>b2Dj</italic> consists of <italic>N</italic> genes, and the overlap between group <italic>G</italic> and <italic>N</italic> genes is <italic>M</italic>. A Fisher's exact test is performed at rank <italic>K</italic>, to evaluate the statistical significance of observing <italic>M</italic> overlaps between a set of size <italic>G</italic> and a set of size <italic>N</italic>, where the set of size <italic>G</italic> comes from platform <italic>P1</italic> and the set of size <italic>N</italic> comes from platform <italic>P2</italic>, given the sizes of <italic>P1</italic> and <italic>P2</italic> as well as the overlap between <italic>P1</italic> and <italic>P2</italic>.</p>
<p>At the end of the scan, the best p-value is retained, and a multiple hypothesis testing correction factor is applied. The multiple testing factor is the expected number of overlaps between the two subsets of the given sizes, given the two platforms <italic>P1</italic> and <italic>P2</italic>. The negative log of the multiple testing corrected best p-value is a score for the subset pair.</p>
<p>Next, the Running Fisher algorithm is performed in the reverse direction: the top ranking genes in the second subset <italic>b2Dj</italic> are collected as a group <italic>G</italic>, and the first subset <italic>b1Di</italic> is scanned in the rank order. The same procedure in this reverse direction produces another score for the same subset pair. The two scores are averaged to represent the magnitude of the similarity between the two subsets. A positive sign is given to the final subset pair-score if <italic>Di</italic> and <italic>Dj</italic> are the same. A negative sign is given if <italic>Di</italic> and <italic>Dj</italic> are opposite. The score is unsigned if any of <italic>Di</italic> and <italic>Dj</italic> is not directional.</p>
<p>Finally, the overall score is computed by summing up all directional subset pair scores (<xref ref-type="fig" rid="pone-0013066-g003">Figure 3</xref>). The sign of the sum determines whether the two signatures are positively or negatively correlated. If one of the two signature sets is directional and the other is not directional, the overall score is represented by the larger of the two subset pair scores, annotated with the contributing direction from the directional signature. If both signatures are not directional, then a single unsigned pair score is calculated between the two biosets.</p>
</sec><sec id="s4e">
<title>Ontology-driven Meta-Analysis</title>
<p>Given a gene signature representing the set of genes of interest from a given experiment, a meta-analysis of the tens of thousands of tagged gene signatures in the NextBio system can be used to determine tissues, diseases, and compounds associated strongly with the query set. Conceptually, the problem is that of ranking ontology terms (concepts) based on how strongly enriched signatures tagged with those concepts are with a set of genes of interest. For a set of genes, ranked or unranked, the gene set enrichment analysis described above is used to identify other strongly associated signatures. Based on the strength of the association, the aggregated scores for their associated semantic concepts were computed.</p>
<p>Meta-analysis scores were computed separately for tissue, disease, and compound categories. Given a query gene signature, a list of contributing signatures was obtained for each category based on two criteria – (1) They have enrichment scores with the query signature above a pre-determined threshold and (2) They pass an initial screening logic that ensures that they are tagged with the appropriate combination of concepts to allow them to contribute to that category.</p>
<p>Based on the list of contributing signatures, along with the associated concepts and enrichment scores, the equation below describes the various factors considered in determining a score for a concept.<disp-formula><graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.e001" xlink:type="simple"/></disp-formula></p>
<p>The normalized hit count for a concept is the sum of the ratio of associated score of each signature tagged with that concept to the overall best association score. The background count of a concept is the number of signatures in the NextBio system tagged with that concept. Inclusion of the background count reduces the bias toward popular concepts that have more associated gene signatures than others. Finally, the average weighted rank represents the average rank of a tagged signature relative to all other correlated signatures weighted by the associated normalized score.</p>
<p>Given the dynamic nature of the NextBio system where the distribution of data from various species, platforms, data types, and semantic categories changes on a continuous basis, it is not obvious at the outset what the relative contributions of each of these factors should be toward determining an optimal overall scoring function for determining top ranked concepts. These are determined empirically by optimizing the model described above using gold standard use cases and tuning parameters (a,b,c in the equation above) for each of the factors.</p>
<p>This meta-analysis scoring process results in a ranked list of ontological terms for each tissue, disease, and compound category. It should be noted that some concepts are part of a hierarchical ontological framework. When that was the case, enrichment scores for signatures tagged with specialized concepts are accordingly propagated to more general parent concepts in the hierarchy. After scores for all concepts are computed, children concepts with lower scores than parent concepts are clustered and presented to the user.</p>
</sec><sec id="s4f">
<title>Computing direction of signature-concept correlation</title>
<p>The overall direction of association (positive or negative) between a query signature and a concept refers to the <italic>net</italic> correlation of the query signature with a set of contributing signatures (see previous section) tagged with that concept. Recall that these contributing signatures may have positive or negative pairwise correlation scores with the query signature. A <italic>cumulative</italic> positive and negative score is obtained by aggregating the scores for the positively and negatively associated signatures separately. To minimize the effect of any spuriously strong signature association, the cumulative scores are each <italic>weighted</italic> by the ratio of the respective number of positively or negatively associated signatures to the total number of contributing signatures. Finally, the overall direction is called depending on which <italic>weighted cumulative score</italic> is greater, as long as a minimum difference threshold is met.</p>
</sec></sec><sec id="s5">
<title>Supporting Information</title>
<supplementary-material id="pone.0013066.s001" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s001" xlink:type="simple"><label>Table S1</label><caption>
<p>The list of public databases containing raw microarray data available to the public. Only major databases were included in this list.</p>
<p>(0.02 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s002" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s002" xlink:type="simple"><label>Table S2</label><caption>
<p>Meta-analysis results for the brown fat tissue signature query against all other public datasets on normal tissue analysis.</p>
<p>(0.02 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s003" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s003" xlink:type="simple"><label>Table S3</label><caption>
<p>Meta-analysis results for the white fat tissue signature query against all other public datasets on normal tissue analysis.</p>
<p>(0.38 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s004" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s004" xlink:type="simple"><label>Table S4</label><caption>
<p>Brown preadipocytes gene expression signature. The signature was identified by comparing cultured brown preadipocytes to white preadipocytes at day 4.</p>
<p>(0.01 MB DOC)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s005" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s005" xlink:type="simple"><label>Table S5</label><caption>
<p>Meta-analysis results for the brown preadipocyte signature query against all other public datasets on normal tissue analysis. Brown preadipocyte signature was determined by comparing gene expression of cultured brown preadipocytes versus white preadipocytes at 4 days.</p>
<p>(0.01 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s006" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s006" xlink:type="simple"><label>Table S6</label><caption>
<p>Query results for gene expression signature comparing mature brown adipocytes to differentiating brown preadipocytes across all signatures tagged with “Compounds” ontology category.</p>
<p>(0.01 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s007" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s007" xlink:type="simple"><label>Table S7</label><caption>
<p>Reversine gene expression signature. The signature was identified by comparing cultured C2C12 mouse myoblasts treated with reversine to non-treated control myoblasts.</p>
<p>(0.62 MB XLS)</p>
</caption></supplementary-material>
<supplementary-material id="pone.0013066.s008" mimetype="application/x-excel" position="float" xlink:href="info:doi/10.1371/journal.pone.0013066.s008" xlink:type="simple"><label>Table S8</label><caption>
<p>Meta-analysis results for the brown preadipocyte signature query against all other public datasets on genetic perturbations analysis. Brown preadipocyte signature was determined by comparing gene expression of cultured brown preadipocytes versus white preadipocytes at 4 days</p>
<p>(0.01 MB XLS)</p>
</caption></supplementary-material>
</sec></body>
<back>
<ack>
<p>The authors would like to thank members of the NextBio team, as well as Dr. Kathleen M. Scully and Dr. Michael G. Rosenfeld. We would like to thank Dr. Saeed Tavazoie and Dr. Joakim Lundeberg for useful comments on this manuscript.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0013066-GardinerGarden1"><label>1</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Gardiner-Garden</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Littlejohn</surname><given-names>TG</given-names></name>
</person-group>             <year>2001</year>             <article-title>A comparison of microarray databases.</article-title>             <source>Brief Bioinform</source>             <volume>2</volume>             <fpage>143</fpage>             <lpage>158</lpage>          </element-citation></ref>
<ref id="pone.0013066-Rhodes1"><label>2</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Rhodes</surname><given-names>DR</given-names></name>
<name name-style="western"><surname>Barrette</surname><given-names>TR</given-names></name>
<name name-style="western"><surname>Rubin</surname><given-names>MA</given-names></name>
<name name-style="western"><surname>Ghosh</surname><given-names>D</given-names></name>
<name name-style="western"><surname>Chinnaiyan</surname><given-names>AM</given-names></name>
</person-group>             <year>2002</year>             <article-title>Meta-analysis of microarrays: interstudy validation of gene expression profiles reveals pathway dysregulation in prostate cancer.</article-title>             <source>Cancer Res</source>             <volume>62</volume>             <fpage>4427</fpage>             <lpage>4433</lpage>          </element-citation></ref>
<ref id="pone.0013066-Ghosh1"><label>3</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Ghosh</surname><given-names>D</given-names></name>
<name name-style="western"><surname>Barette</surname><given-names>TR</given-names></name>
<name name-style="western"><surname>Rhodes</surname><given-names>D</given-names></name>
<name name-style="western"><surname>Chinnaiyan</surname><given-names>AM</given-names></name>
</person-group>             <year>2003</year>             <article-title>Statistical issues and methods for meta-analysis of microarray data: a case study in prostate cancer.</article-title>             <source>Funct Integr Genomics</source>             <volume>3</volume>             <fpage>180</fpage>             <lpage>188</lpage>          </element-citation></ref>
<ref id="pone.0013066-Jiang1"><label>4</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Jiang</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Deng</surname><given-names>Y</given-names></name>
<name name-style="western"><surname>Chen</surname><given-names>HS</given-names></name>
<name name-style="western"><surname>Tao</surname><given-names>L</given-names></name>
<name name-style="western"><surname>Sha</surname><given-names>Q</given-names></name>
<etal/></person-group>             <year>2004</year>             <article-title>Joint analysis of two microarray gene-expression data sets to select lung adenocarcinoma marker genes.</article-title>             <source>BMC Bioinformatics</source>             <volume>5</volume>             <fpage>81</fpage>          </element-citation></ref>
<ref id="pone.0013066-Griffith1"><label>5</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Griffith</surname><given-names>OL</given-names></name>
<name name-style="western"><surname>Melck</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Jones</surname><given-names>SJM</given-names></name>
<name name-style="western"><surname>Wiseman</surname><given-names>SM</given-names></name>
</person-group>             <year>2006</year>             <article-title>Meta-analysis and meta-review of thyroid cancer gene expression profiling studies identifies important diagnostic biomarkers.</article-title>             <source>J Clin Oncol</source>             <volume>24</volume>             <fpage>5043</fpage>             <lpage>5051</lpage>          </element-citation></ref>
<ref id="pone.0013066-Fishel1"><label>6</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Fishel</surname><given-names>I</given-names></name>
<name name-style="western"><surname>Kaufman</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Ruppin</surname><given-names>E</given-names></name>
</person-group>             <year>2007</year>             <article-title>Meta-analysis of gene expression data: a predictor-based approach.</article-title>             <source>Bioinformatics</source>             <volume>23</volume>             <fpage>1599</fpage>             <lpage>1606</lpage>          </element-citation></ref>
<ref id="pone.0013066-Chan1"><label>7</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Chan</surname><given-names>SK</given-names></name>
<name name-style="western"><surname>Griffith</surname><given-names>OL</given-names></name>
<name name-style="western"><surname>Tai</surname><given-names>IT</given-names></name>
<name name-style="western"><surname>Jones</surname><given-names>SJM</given-names></name>
</person-group>             <year>2008</year>             <article-title>Meta-analysis of colorectal cancer gene expression profiling studies identifies consistently reported candidate biomarkers.</article-title>             <source>Cancer Epidemiol Biomarkers Prev</source>             <volume>17</volume>             <fpage>543</fpage>             <lpage>552</lpage>          </element-citation></ref>
<ref id="pone.0013066-deMagalhes1"><label>8</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>de Magalhães</surname><given-names>JP</given-names></name>
<name name-style="western"><surname>Curado</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Church</surname><given-names>GM</given-names></name>
</person-group>             <year>2009</year>             <article-title>Meta-analysis of age-related gene expression profiles identifies common signatures of aging.</article-title>             <source>Bioinformatics</source>             <volume>25</volume>             <fpage>875</fpage>             <lpage>881</lpage>          </element-citation></ref>
<ref id="pone.0013066-Wirapati1"><label>9</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Wirapati</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Sotiriou</surname><given-names>C</given-names></name>
<name name-style="western"><surname>Kunkel</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Farmer</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Pradervand</surname><given-names>S</given-names></name>
<etal/></person-group>             <year>2008</year>             <article-title>Meta-analysis of gene expression profiles in breast cancer: toward a unified understanding of breast cancer subtyping and prognosis signatures.</article-title>             <source>Breast Cancer Res</source>             <volume>10</volume>             <fpage>R65</fpage>          </element-citation></ref>
<ref id="pone.0013066-Miller1"><label>10</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Miller</surname><given-names>BG</given-names></name>
<name name-style="western"><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name>
</person-group>             <year>2010</year>             <article-title>Integrative Meta-Analysis of Differential Gene Expression in Acute Myeloid Leukemia.</article-title>             <source>PLoS ONE</source>             <volume>5</volume>             <fpage>e9466</fpage>          </element-citation></ref>
<ref id="pone.0013066-Lamb1"><label>11</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Lamb</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Crawford</surname><given-names>ED</given-names></name>
<name name-style="western"><surname>Peck</surname><given-names>D</given-names></name>
<name name-style="western"><surname>Modell</surname><given-names>JW</given-names></name>
<name name-style="western"><surname>Blat</surname><given-names>IC</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>The Connectivity Map: using gene-expression signatures to connect small molecules, genes, and disease.</article-title>             <source>Science</source>             <volume>313</volume>             <fpage>1929</fpage>             <lpage>1935</lpage>          </element-citation></ref>
<ref id="pone.0013066-Edgar1"><label>12</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Edgar</surname><given-names>R</given-names></name>
<name name-style="western"><surname>Domrachev</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Lash</surname><given-names>AE</given-names></name>
</person-group>             <year>2002</year>             <article-title>Gene Expression Omnibus: NCBI gene expression and hybridization array data repository.</article-title>             <source>Nucleic Acids Res</source>             <volume>30</volume>             <fpage>207</fpage>             <lpage>210</lpage>          </element-citation></ref>
<ref id="pone.0013066-Brazma1"><label>13</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Brazma</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Parkinson</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Sarkans</surname><given-names>U</given-names></name>
<name name-style="western"><surname>Shojatalab</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Vilo</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Abeygunawardena</surname><given-names>N</given-names></name>
<name name-style="western"><surname>Holloway</surname><given-names>E</given-names></name>
<name name-style="western"><surname>Kapushesky</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Kemmeren</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Lara</surname><given-names>GG</given-names></name>
<etal/></person-group>             <year>2003</year>             <article-title>ArrayExpress–a public repository for microarray gene expression data at the EBI.</article-title>             <source>Nucleic Acids Res</source>             <volume>31</volume>             <fpage>68</fpage>             <lpage>71</lpage>          </element-citation></ref>
<ref id="pone.0013066-Sherlock1"><label>14</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Sherlock</surname><given-names>G</given-names></name>
<name name-style="western"><surname>Hernandez-Boussard</surname><given-names>T</given-names></name>
<name name-style="western"><surname>Kasarskis</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Binkley</surname><given-names>G</given-names></name>
<name name-style="western"><surname>Matese</surname><given-names>JC</given-names></name>
<etal/></person-group>             <year>2001</year>             <article-title>The Stanford Microarray Database.</article-title>             <source>Nucleic Acids Res</source>             <volume>29</volume>             <fpage>152</fpage>             <lpage>155</lpage>          </element-citation></ref>
<ref id="pone.0013066-Park1"><label>15</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Park</surname><given-names>YR</given-names></name>
<name name-style="western"><surname>Lee</surname><given-names>HW</given-names></name>
<name name-style="western"><surname>Kim</surname><given-names>JH</given-names></name>
</person-group>             <year>2005</year>             <article-title>Integrating microarray gene expression object model and clinical document architecture for cancer genomics research.</article-title>             <source>AMIA Annu Symp Proc</source>             <volume>2005</volume>             <fpage>1073</fpage>          </element-citation></ref>
<ref id="pone.0013066-Harbig1"><label>16</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Harbig</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Sprinkle</surname><given-names>R</given-names></name>
<name name-style="western"><surname>Enkemann</surname><given-names>SA</given-names></name>
</person-group>             <year>2005</year>             <article-title>A sequence-based identification of the genes detected by probesets on the Affymetrix U133 plus 2.0 array.</article-title>             <source>Nucleic Acids Res</source>             <volume>33</volume>             <fpage>e31</fpage>          </element-citation></ref>
<ref id="pone.0013066-Dai1"><label>17</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Dai</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Wang</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Boyd</surname><given-names>AD</given-names></name>
<name name-style="western"><surname>Kostov</surname><given-names>G</given-names></name>
<name name-style="western"><surname>Athey</surname><given-names>B</given-names></name>
<etal/></person-group>             <year>2005</year>             <article-title>Evolving gene/transcript definitions significantly alter the interpretation of GeneChip data.</article-title>             <source>Nucleic Acids Res</source>             <volume>33</volume>             <fpage>e175</fpage>          </element-citation></ref>
<ref id="pone.0013066-Yu1"><label>18</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Yu</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Wang</surname><given-names>F</given-names></name>
<name name-style="western"><surname>Tu</surname><given-names>K</given-names></name>
<name name-style="western"><surname>Xie</surname><given-names>L</given-names></name>
<name name-style="western"><surname>Li</surname><given-names>Y-Y</given-names></name>
<name name-style="western"><surname>Li</surname><given-names>Y-X</given-names></name>
</person-group>             <year>2007</year>             <article-title>Transcript-level annotation of Affymetrix probesets improves the interpretation of gene expression data.</article-title>             <source>BMC Bioinformatics</source>             <volume>8</volume>             <fpage>194</fpage>          </element-citation></ref>
<ref id="pone.0013066-Subramanian1"><label>19</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Subramanian</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Tamayo</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Mootha</surname><given-names>VK</given-names></name>
<name name-style="western"><surname>Mukherjee</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Ebert</surname><given-names>BL</given-names></name>
<etal/></person-group>             <year>2005</year>             <article-title>Gene set enrichment analysis: a knowledge-based approach for interpreting genome-wide expression profiles.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>102</volume>             <fpage>15545</fpage>             <lpage>15550</lpage>          </element-citation></ref>
<ref id="pone.0013066-Newton1"><label>20</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Newton</surname><given-names>MA</given-names></name>
<name name-style="western"><surname>Quintana</surname><given-names>FA</given-names></name>
<name name-style="western"><surname>den Boon</surname><given-names>JA</given-names></name>
<name name-style="western"><surname>Sengupta</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Ahlquist</surname><given-names>P</given-names></name>
</person-group>             <year>2007</year>             <article-title>Random-set methods identify distinct aspects of the enrichment signal in gene-set analysis.</article-title>             <source>Ann Appl Stat</source>             <volume>1</volume>             <fpage>85</fpage>             <lpage>106</lpage>          </element-citation></ref>
<ref id="pone.0013066-Su1"><label>21</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Su</surname><given-names>AI</given-names></name>
<name name-style="western"><surname>Wiltshire</surname><given-names>T</given-names></name>
<name name-style="western"><surname>Batalov</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Lapp</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Ching</surname><given-names>KA</given-names></name>
<etal/></person-group>             <year>2004</year>             <article-title>A gene atlas of the mouse and human protein-encoding transcriptomes.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>101</volume>             <fpage>6062</fpage>             <lpage>6067</lpage>          </element-citation></ref>
<ref id="pone.0013066-Seale1"><label>22</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Seale</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Bjork</surname><given-names>B</given-names></name>
<name name-style="western"><surname>Yang</surname><given-names>W</given-names></name>
<name name-style="western"><surname>Kajimura</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Chin</surname><given-names>S</given-names></name>
<etal/></person-group>             <year>2008</year>             <article-title>PRDM16 controls a brown fat/skeletal muscle switch.</article-title>             <source>Nature</source>             <volume>454</volume>             <fpage>961</fpage>             <lpage>967</lpage>          </element-citation></ref>
<ref id="pone.0013066-Timmons1"><label>23</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Timmons</surname><given-names>JA</given-names></name>
<name name-style="western"><surname>Wennmalm</surname><given-names>K</given-names></name>
<name name-style="western"><surname>Larsson</surname><given-names>O</given-names></name>
<name name-style="western"><surname>Walden</surname><given-names>TB</given-names></name>
<name name-style="western"><surname>Lassmann</surname><given-names>T</given-names></name>
<etal/></person-group>             <year>2007</year>             <article-title>Myogenic gene expression signature establishes that brown and white adipocytes originate from distinct cell lineages.</article-title>             <source>Proc Nat Acad Sci U S A</source>             <volume>104</volume>             <fpage>4401</fpage>             <lpage>4406</lpage>          </element-citation></ref>
<ref id="pone.0013066-Atit1"><label>24</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Atit</surname><given-names>R</given-names></name>
<name name-style="western"><surname>Sgaier</surname><given-names>SK</given-names></name>
<name name-style="western"><surname>Mohamed</surname><given-names>OA</given-names></name>
<name name-style="western"><surname>Taketo</surname><given-names>MM</given-names></name>
<name name-style="western"><surname>Dufort</surname><given-names>D</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>Beta-catenin activation is necessary and sufficient to specify the dorsal dermal fate in the mouse.</article-title>             <source>Dev Biol</source>             <volume>296</volume>             <fpage>164</fpage>             <lpage>176</lpage>          </element-citation></ref>
<ref id="pone.0013066-Bild1"><label>25</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Bild</surname><given-names>AH</given-names></name>
<name name-style="western"><surname>Yao</surname><given-names>G</given-names></name>
<name name-style="western"><surname>Chang</surname><given-names>JT</given-names></name>
<name name-style="western"><surname>Wang</surname><given-names>Q</given-names></name>
<name name-style="western"><surname>Potti</surname><given-names>A</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>Oncogenic pathway signatures in human cancers as a guide to targeted therapies.</article-title>             <source>Nature</source>             <volume>439</volume>             <fpage>353</fpage>             <lpage>357</lpage>          </element-citation></ref>
<ref id="pone.0013066-Noushmehr1"><label>26</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Noushmehr</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Weisenberger</surname><given-names>DJ</given-names></name>
<name name-style="western"><surname>Diefes</surname><given-names>K</given-names></name>
<name name-style="western"><surname>Phillips</surname><given-names>HS</given-names></name>
<name name-style="western"><surname>Pujara</surname><given-names>K</given-names></name>
<etal/></person-group>             <year>2010</year>             <article-title>Identification of a CpG Island Methylator Phenotype that Defines a Distinct Subgroup of Glioma.</article-title>             <source>Cancer Cell</source>             <volume>17</volume>             <fpage>510</fpage>             <lpage>522</lpage>          </element-citation></ref>
<ref id="pone.0013066-Almind1"><label>27</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Almind</surname><given-names>K</given-names></name>
<name name-style="western"><surname>Manieri</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Sivitz</surname><given-names>WI</given-names></name>
<name name-style="western"><surname>Cinti</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Kahn</surname><given-names>CR</given-names></name>
</person-group>             <year>2007</year>             <article-title>Ectopic brown adipose tissue in muscle provides a mechanism for differences in risk of metabolic syndrome in mice.</article-title>             <source>Proc Natl Acad Sci U S A</source>             <volume>104</volume>             <fpage>2366</fpage>             <lpage>2371</lpage>          </element-citation></ref>
<ref id="pone.0013066-Farmer1"><label>28</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Farmer</surname><given-names>SR</given-names></name>
</person-group>             <year>2008</year>             <article-title>Brown fat and skeletal muscle: unlikely cousins?</article-title>             <source>Cell</source>             <volume>134</volume>             <fpage>726</fpage>             <lpage>727</lpage>          </element-citation></ref>
<ref id="pone.0013066-Gesta1"><label>29</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Gesta</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Tseng</surname><given-names>YH</given-names></name>
<name name-style="western"><surname>Kahn</surname><given-names>CR</given-names></name>
</person-group>             <year>2007</year>             <article-title>Developmental origin of fat: tracking obesity to its source.</article-title>             <source>Cell</source>             <volume>131</volume>             <fpage>242</fpage>             <lpage>256</lpage>          </element-citation></ref>
<ref id="pone.0013066-Lehnert1"><label>30</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Lehnert</surname><given-names>SA</given-names></name>
<name name-style="western"><surname>Byrne</surname><given-names>KA</given-names></name>
<name name-style="western"><surname>Reverter</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Nattrass</surname><given-names>GS</given-names></name>
<name name-style="western"><surname>Greenwood</surname><given-names>PL</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>Gene expression profiling of bovine skeletal muscle in response to and during recovery from chronic and severe undernutrition.</article-title>             <source>Journal Anim Sci</source>             <volume>84</volume>             <fpage>3239</fpage>             <lpage>3250</lpage>          </element-citation></ref>
<ref id="pone.0013066-Kim1"><label>31</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Kim</surname><given-names>YK</given-names></name>
<name name-style="western"><surname>Choi</surname><given-names>HY</given-names></name>
<name name-style="western"><surname>Kim</surname><given-names>NH</given-names></name>
<name name-style="western"><surname>Lee</surname><given-names>W</given-names></name>
<name name-style="western"><surname>Seo</surname><given-names>DW</given-names></name>
<etal/></person-group>             <year>2007</year>             <article-title>Reversine stimulates adipocyte differentiation and downregulates Akt and p70(s6k) signaling pathways in 3T3-L1 cells.</article-title>             <source>Biochem Biophys Res Commun</source>             <volume>358</volume>             <fpage>553</fpage>             <lpage>558</lpage>          </element-citation></ref>
<ref id="pone.0013066-Lee1"><label>32</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Lee</surname><given-names>EK</given-names></name>
<name name-style="western"><surname>Bae</surname><given-names>GU</given-names></name>
<name name-style="western"><surname>You</surname><given-names>JS</given-names></name>
<name name-style="western"><surname>Lee</surname><given-names>JC</given-names></name>
<name name-style="western"><surname>Jeon</surname><given-names>YJ</given-names></name>
<etal/></person-group>             <year>2009</year>             <article-title>Reversine increases the plasticity of lineage-committed cells toward neuroectodermal lineage.</article-title>             <source>J Biol Chem</source>             <volume>284</volume>             <fpage>2891</fpage>             <lpage>2901</lpage>          </element-citation></ref>
<ref id="pone.0013066-Caramel1"><label>33</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Caramel</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Medjkane</surname><given-names>S</given-names></name>
<name name-style="western"><surname>Quignon</surname><given-names>F</given-names></name>
<name name-style="western"><surname>Delattre</surname><given-names>O</given-names></name>
</person-group>             <year>2008</year>             <article-title>The requirement for SNF5/INI1 in adipocyte differentiation highlights new features of malignant rhabdoid tumors.</article-title>             <source>Oncogene</source>             <volume>27</volume>             <fpage>2035</fpage>             <lpage>2044</lpage>          </element-citation></ref>
<ref id="pone.0013066-Freytag1"><label>34</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Freytag</surname><given-names>SO</given-names></name>
<name name-style="western"><surname>Geddes</surname><given-names>TJ</given-names></name>
</person-group>             <year>1992</year>             <article-title>Reciprocal regulation of adipogenesis by Myc and C/EBP alpha.</article-title>             <source>Science</source>             <volume>256</volume>             <fpage>379</fpage>             <lpage>382</lpage>          </element-citation></ref>
<ref id="pone.0013066-Bonal1"><label>35</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Bonal</surname><given-names>C</given-names></name>
<name name-style="western"><surname>Thorel</surname><given-names>F</given-names></name>
<name name-style="western"><surname>Ait-Lounis</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Reith</surname><given-names>W</given-names></name>
<name name-style="western"><surname>Trumpp</surname><given-names>A</given-names></name>
<name name-style="western"><surname>Herrera</surname><given-names>PL</given-names></name>
</person-group>             <year>2008</year>             <article-title>Pancreatic inactivation of c-Myc decreases acinar mass and transdifferentiates acinar cells into adipocytes in mice.</article-title>             <source>Gastroenterology</source>             <volume>136</volume>             <fpage>309</fpage>             <lpage>319</lpage>          </element-citation></ref>
<ref id="pone.0013066-Parrish1"><label>36</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Parrish</surname><given-names>RS</given-names></name>
<name name-style="western"><surname>Spencer</surname><given-names>HJ</given-names><suffix>3rd</suffix></name>
</person-group>             <year>2004</year>             <article-title>Effect of normalization on significance testing for oligonucleotide microarrays.</article-title>             <source>J Biopharm Stat</source>             <volume>14</volume>             <fpage>575</fpage>             <lpage>589</lpage>          </element-citation></ref>
<ref id="pone.0013066-Li1"><label>37</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Li</surname><given-names>C</given-names></name>
<name name-style="western"><surname>Wong</surname><given-names>WH</given-names></name>
</person-group>             <year>2001</year>             <article-title>Model-based analysis of oligonucleotide arrays: expression index computation and outlier detection.</article-title>             <source>Proc Natl Acad Sci USA</source>             <volume>98</volume>             <fpage>31</fpage>             <lpage>36</lpage>          </element-citation></ref>
<ref id="pone.0013066-Kittleson1"><label>38</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Kittleson</surname><given-names>MM</given-names></name>
<name name-style="western"><surname>Minhas</surname><given-names>KM</given-names></name>
<name name-style="western"><surname>Irizarry</surname><given-names>RA</given-names></name>
<name name-style="western"><surname>Ye</surname><given-names>SQ</given-names></name>
<name name-style="western"><surname>Edness</surname><given-names>G</given-names></name>
<etal/></person-group>             <year>2005</year>             <article-title>Gene expression in giant cell myocarditis: Altered expression of immune response genes.</article-title>             <source>Int J Cardiol</source>             <volume>102</volume>             <fpage>333</fpage>             <lpage>340</lpage>          </element-citation></ref>
<ref id="pone.0013066-Li2"><label>39</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Li</surname><given-names>Y</given-names></name>
<name name-style="western"><surname>Elashoff</surname><given-names>D</given-names></name>
<name name-style="western"><surname>Oh</surname><given-names>M</given-names></name>
<name name-style="western"><surname>Sinha</surname><given-names>U</given-names></name>
<name name-style="western"><surname>St John</surname><given-names>MAR</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>Serum circulating human mRNA profiling and its utility for oral cancer detection.</article-title>             <source>J Clin Oncol</source>             <volume>24</volume>             <fpage>1754</fpage>             <lpage>1760</lpage>          </element-citation></ref>
<ref id="pone.0013066-Quinn1"><label>40</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Quinn</surname><given-names>P</given-names></name>
<name name-style="western"><surname>Bowers</surname><given-names>RM</given-names></name>
<name name-style="western"><surname>Zhang</surname><given-names>X</given-names></name>
<name name-style="western"><surname>Wahlund</surname><given-names>TM</given-names></name>
<name name-style="western"><surname>Fanelli</surname><given-names>MA</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>cDNA microarrays as a tool for identification of biomineralization proteins in the coccolithophorid Emiliania hux-leyi (Haptophyta).</article-title>             <source>Appl Environ Microbiol</source>             <volume>72</volume>             <fpage>5512</fpage>             <lpage>5526</lpage>          </element-citation></ref>
<ref id="pone.0013066-Tusher1"><label>41</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Tusher</surname><given-names>VG</given-names></name>
<name name-style="western"><surname>Tibshirani</surname><given-names>R</given-names></name>
<name name-style="western"><surname>Chu</surname><given-names>G</given-names></name>
</person-group>             <year>2001</year>             <article-title>Significance analysis of microarrays applied to the ionizing radiation response.</article-title>             <source>Proc Nat Acad Sci U S A</source>             <volume>98</volume>             <fpage>5116</fpage>             <lpage>5121</lpage>          </element-citation></ref>
<ref id="pone.0013066-Shi1"><label>42</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Shi</surname><given-names>L</given-names></name>
<name name-style="western"><surname>Tong</surname><given-names>W</given-names></name>
<name name-style="western"><surname>Fang</surname><given-names>H</given-names></name>
<name name-style="western"><surname>Scherf</surname><given-names>U</given-names></name>
<name name-style="western"><surname>Han</surname><given-names>J</given-names></name>
<etal/></person-group>             <year>2005</year>             <article-title>Cross-platform comparability of microarray technology: intra-platform consistency and appropriate data analysis procedures are essential.</article-title>             <source>BMC Bioinformatics</source>             <volume>6</volume>             <issue>Suppl 2</issue>             <fpage>S12</fpage>          </element-citation></ref>
<ref id="pone.0013066-Wang1"><label>43</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Wang</surname><given-names>Y</given-names></name>
<name name-style="western"><surname>Barbacioru</surname><given-names>C</given-names></name>
<name name-style="western"><surname>Hyland</surname><given-names>F</given-names></name>
<name name-style="western"><surname>Xiao</surname><given-names>W</given-names></name>
<name name-style="western"><surname>Hunkapiller</surname><given-names>KL</given-names></name>
<etal/></person-group>             <year>2006</year>             <article-title>Large scale real-time PCR validation on gene expression measurements from two commercial long-oligonucleotide microarrays.</article-title>             <source>BMC Genomics</source>             <volume>7</volume>             <fpage>59</fpage>          </element-citation></ref>
<ref id="pone.0013066-Knowles1"><label>44</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Knowles</surname><given-names>LM</given-names></name>
<name name-style="western"><surname>Smith</surname><given-names>JW</given-names></name>
</person-group>             <year>2007</year>             <article-title>Genome-wide changes accompanying knockdown of fatty acid synthase in breast cancer.</article-title>             <source>BMC Genomics</source>             <volume>8</volume>             <fpage>168</fpage>          </element-citation></ref>
<ref id="pone.0013066-Grigoryev1"><label>45</label><element-citation publication-type="journal" xlink:type="simple">             <person-group person-group-type="author">
<name name-style="western"><surname>Grigoryev</surname><given-names>DM</given-names></name>
<name name-style="western"><surname>Ma</surname><given-names>SF</given-names></name>
<name name-style="western"><surname>Irizarry</surname><given-names>RA</given-names></name>
<name name-style="western"><surname>Ye</surname><given-names>SQ</given-names></name>
<name name-style="western"><surname>Quackenbush</surname><given-names>J</given-names></name>
<name name-style="western"><surname>Garcia</surname><given-names>JGN</given-names></name>
</person-group>             <year>2004</year>             <article-title>Orthologous gene-expression profiling in multi-species models: search for candidate genes.</article-title>             <source>Genome Biol</source>             <volume>5</volume>             <fpage>R34</fpage>          </element-citation></ref>
</ref-list>

</back>
</article>