<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article
  PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="3.0" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id><journal-title-group>
<journal-title>PLoS ONE</journal-title></journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, USA</publisher-loc></publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PONE-D-14-29933</article-id>
<article-id pub-id-type="doi">10.1371/journal.pone.0113053</article-id>
<article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Artificial intelligence</subject><subj-group><subject>Machine learning</subject></subj-group></subj-group></subj-group></subj-group><subj-group><subject>Computational biology</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Transcriptome analysis</subject><subj-group><subject>Genome expression analysis</subject></subj-group></subj-group><subj-group><subject>Genomic databases</subject></subj-group></subj-group><subj-group><subject>Genomics statistics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Computer and information sciences</subject></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Statistics (mathematics)</subject><subj-group><subject>Biostatistics</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v2"><subject>Research and analysis methods</subject><subj-group><subject>Database and informatics methods</subject><subj-group><subject>Bioinformatics</subject><subject>Biological databases</subject><subject>Information retrieval</subject></subj-group></subj-group><subj-group><subject>Mathematical and statistical techniques</subject><subj-group><subject>Multivariate data analysis</subject></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Toward Computational Cumulative Biology by Combining Models of Biological Datasets</article-title>
<alt-title alt-title-type="running-head">Computational Cumulative Biology</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Faisal</surname><given-names>Ali</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Peltonen</surname><given-names>Jaakko</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Georgii</surname><given-names>Elisabeth</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Rung</surname><given-names>Johan</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref></contrib>
<contrib contrib-type="author" xlink:type="simple"><name name-style="western"><surname>Kaski</surname><given-names>Samuel</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="corresp" rid="cor1"><sup>*</sup></xref></contrib>
</contrib-group>
<aff id="aff1"><label>1</label><addr-line>Helsinki Institute for Information Technology HIIT, Department of Information and Computer Science, Aalto University, Espoo, Finland</addr-line></aff>
<aff id="aff2"><label>2</label><addr-line>European Molecular Biology Laboratory, European Bioinformatics Institute (EMBL-EBI), Wellcome Trust Genome Campus, Hinxton, United Kingdom</addr-line></aff>
<aff id="aff3"><label>3</label><addr-line>Helsinki Institute for Information Technology HIIT, Department of Computer Science, University of Helsinki, Helsinki, Finland</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple"><name name-style="western"><surname>Qian</surname><given-names>Xiaoning</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/></contrib>
</contrib-group>
<aff id="edit1"><addr-line>University of South Florida, United States of America</addr-line></aff>
<author-notes>
<corresp id="cor1">* E-mail: <email xlink:type="simple">samuel.kaski@aalto.fi</email></corresp>
<fn fn-type="conflict"><p>The authors have declared that no competing interests exist.</p></fn>
<fn fn-type="con"><p>Conceived and designed the experiments: AF JP EG SK. Performed the experiments: AF EG. Analyzed the data: AF EG JR. Wrote the paper: AF JP EG JR SK.</p></fn>
</author-notes>
<pub-date pub-type="collection"><year>2014</year></pub-date>
<pub-date pub-type="epub"><day>26</day><month>11</month><year>2014</year></pub-date>
<volume>9</volume>
<issue>11</issue>
<elocation-id>e113053</elocation-id>
<history>
<date date-type="received"><day>4</day><month>7</month><year>2014</year></date>
<date date-type="accepted"><day>17</day><month>10</month><year>2014</year></date>
</history>
<permissions>
<copyright-year>2014</copyright-year>
<copyright-holder>Faisal et al</copyright-holder><license xlink:type="simple"><license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license></permissions>
<abstract>
<p>A main challenge of data-driven sciences is how to make maximal use of the progressively expanding databases of experimental datasets in order to keep research cumulative. We introduce the idea of a modeling-based dataset retrieval engine designed for relating a researcher's experimental dataset to earlier work in the field. The search is (i) data-driven to enable new findings, going beyond the state of the art of keyword searches in annotations, (ii) modeling-driven, to include both biological knowledge and insights learned from data, and (iii) scalable, as it is accomplished without building one unified grand model of all data. Assuming each dataset has been modeled beforehand, by the researchers or automatically by database managers, we apply a rapidly computable and optimizable combination model to decompose a new dataset into contributions from earlier relevant models. By using the data-driven decomposition, we identify a network of interrelated datasets from a large annotated human gene expression atlas. While tissue type and disease were major driving forces for determining relevant datasets, the found relationships were richer, and the model-based search was more accurate than the keyword search; moreover, it recovered biologically meaningful relationships that are not straightforwardly visible from annotations—for instance, between cells in different developmental stages such as thymocytes and T-cells. Data-driven links and citations matched to a large extent; the data-driven links even uncovered corrections to the publication data, as two of the most linked datasets were not highly cited and turned out to have wrong publication entries in the database.</p>
</abstract>
<funding-group><funding-statement>Academy of Finland (<ext-link ext-link-type="uri" xlink:href="http://www.aka.fi" xlink:type="simple">http://www.aka.fi</ext-link>), Finnish Centre of Excellence in Computational Inference Research COIN, 251170, to AF, JP, EG, SK. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement></funding-group><counts><page-count count="17"/></counts><custom-meta-group><custom-meta id="data-availability" xlink:type="simple"><meta-name>Data Availability</meta-name><meta-value>The authors confirm that, for approved reasons, some access restrictions apply to the data underlying the findings. The authors confirm that all gene expression data underlying the findings are fully available without restriction from ArrayExpress (E-MTAB-62, E-CBIL-30, E-GEOD-12648, E-GEOD-4667, E-GEOD-8441, E-GEOD-10760, E-GEOD-1295, E-GEOD-474, E-GEOD-9105, E-GEOD-11686, E-GEOD-1786, E-GEOD-6011, E-GEOD-9397, E-GEOD-11971, E-GEOD-3307, E-GEOD-7146, E-GEOD-9676). The citation data underlying the findings (citation graph, h-indexes and impact factors) are available from Thomson Reuters. Reason for restriction of public deposition of citation data: We are not allowed to make the citation data available because it is third party - Copyright Thomson Reuters, 2011. To re-create the citation graph, interested users need to contact Thomson Reuters, Emma Dennis of the research analytics team at “ts.researchservices@thomson.com” and request for license for “raw tagged data”. The H-indexes and impact factors are also available through the same contact. More details at: <ext-link ext-link-type="uri" xlink:href="http://thomsonreuters.com/terms-of-use/" xlink:type="simple">http://thomsonreuters.com/terms-of-use/</ext-link>.</meta-value></custom-meta></custom-meta-group></article-meta>
</front>
<body><sec id="s1">
<title>Introduction</title>
<p>Molecular biology, historically driven by the pursuit of experimentally characterizing each component of the living cell, has been transformed into a data-driven science <xref ref-type="bibr" rid="pone.0113053-Greene1">[1]</xref>–<xref ref-type="bibr" rid="pone.0113053-Gerber1">[6]</xref> with just as much importance given to the computational and statistical analysis as to experimental design and assay technology. This has brought to the fore new computational challenges, such as the processing of massive new sequencing data, and new statistical challenges arising from the problem of having relatively few (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e001" xlink:type="simple"/></inline-formula>) samples characterized for relatively many (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e002" xlink:type="simple"/></inline-formula>) variables—the “large <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e003" xlink:type="simple"/></inline-formula>, small <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e004" xlink:type="simple"/></inline-formula>” problem. High-throughput technologies often are developed to assay many parallel variables for a single sample in a run, rather than many parallel samples for a single variable, whereas the statistical power to infer properties of biological conditions increases with larger sample sizes. For cost reasons, most labs are restricted to generating datasets with the statistical power to detect only the strongest effects. In combination with the penalties of multiple hypothesis testing, the limitations of “large <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e005" xlink:type="simple"/></inline-formula>, small <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e006" xlink:type="simple"/></inline-formula>” datasets are obvious. It is, therefore, not surprising that much work has been devoted to address this problem.</p>
<p>Some of the most successful methods rely on increasing the effective number of samples by combining with data from other, similarly designed, experiments, in a large meta-analysis <xref ref-type="bibr" rid="pone.0113053-Tseng1">[7]</xref>. Unfortunately, this is not straightforward, either. Although public data repositories, such as the ones at NCBI in the United States and the EBI in Europe, serve the research community with ever-growing amounts of experimental data, they largely rely on annotation and meta-data provided by the submitter. Database curators and semantic tools such as ontologies provide some help in harmonizing and standardizing the annotation, but the user who wants to find datasets that are combinable with her own most often must resort to searches in free text or in controlled vocabularies, which would need significant downstream curation and data analysis before any meta-analysis can be done <xref ref-type="bibr" rid="pone.0113053-Rung1">[8]</xref>.</p>
<p>Ideally, we would like to let the data speak for themselves. Instead of searching for datasets that have been described similarly, which may not correspond to a statistical similarity in the datasets themselves, we would like to conduct that search in a data-driven way, using as the query the dataset itself or a statistical (rather than a semantic) description of it. This is implicitly done, for example, in multi-task learning, a method from the machine learning field <xref ref-type="bibr" rid="pone.0113053-Baxter1">[9]</xref>, <xref ref-type="bibr" rid="pone.0113053-Caruana1">[10]</xref>, where several related estimation tasks are pursued together, assuming shared properties across tasks. Multi-task learning is a form of global analysis, which builds a single unified model of the datasets. But as the number of datasets keeps increasing and the amount of quantitative biological knowledge keeps accumulating, the complexity of building an accurate unified model becomes increasingly prohibitive.</p>
<p>Addressing the “large <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e007" xlink:type="simple"/></inline-formula>, small <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e008" xlink:type="simple"/></inline-formula>” problem requires taking into account both the uncertainty in the data and the existing biological knowledge. We now consider the hypothesized scenario where future researchers increasingly develop hypotheses in terms of (probabilistic) models of their data. Although far from realistic today, a similar trend exists for sequence motif data, which are often published as Hidden Markov models, for instance in the Pfam database <xref ref-type="bibr" rid="pone.0113053-Finn1">[11]</xref>.</p>
<p>In this paper, we report on a feasibility study that uses the scenario in which many experiments have been modeled beforehand, potentially by the researcher generating the data or automatically by the database storing the model together with the data. We ask <italic>what could be done with these models towards cumulatively building knowledge from data in molecular biology</italic>? Speaking about models generally and assuming the many practical issues can be solved technically, we arrive at our answer: we propose creating <italic>a modeling-driven dataset retrieval engine</italic>, which a researcher can use for positioning her own measurement data into the context of the earlier biology. The engine will point out relationships between experiments in the form of the retrieval results, which is a naturally understandable interface. The retrieval will be based on data, instead of the state-of-the-art practice of using keywords and ontologies, which will make unexpected and previously unknown findings possible. The retrieval will use the models of the datasets, which, by our assumption above, incorporate the knowledge of the researchers producing the data about what is important in the data, but the retrieval will be designed to be more scalable than building one unified grand model of all data. This also implies that the way the models are utilized needs to be approximate. Compared to existing data-driven retrieval methods <xref ref-type="bibr" rid="pone.0113053-Caldas1">[3]</xref>, <xref ref-type="bibr" rid="pone.0113053-Schmid1">[5]</xref>, whole datasets, incorporating the experimental designs, will be matched, instead of individual observations. The remaining question is how to design the retrieval so that it both reveals the interesting and important relationships and is fast to compute.</p>
<p>The model we present is a first step towards this goal. We assume that a new dataset can be explained by a combination of the models for the earlier datasets and a novelty term. This is a mixture modeling or regression task, in which the weights can be computed rapidly; the resulting method scales well to large numbers of datasets, and the speed of the mixture modeling does not depend on the sizes of the earlier datasets. The largest weights in the mixture model point at the most relevant earlier datasets. The method is applicable to several types of measurement datasets, assuming that suitable models exist. Unlike traditional mixture modeling, we do not limit the form of the mixture components; thus, we bring in the knowledge built into the stored models of each dataset. We apply this approach to a large set of experiments from EBI's ArrayExpress gene expression database <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>, treating each experiment in turn as a new dataset, queried against all earlier datasets. Under our assumptions, the retrieval results can be interpreted as studies that the authors of the study generating the query set could have cited, and we show that the actual citations overlap with the retrieval results. The discovered links between datasets additionally enable forming a “hall of fame” of gene expression studies, containing the studies that would have been influential, assuming the retrieval system existed. The links in the “hall of fame” verify and complement the citation links: in our study, they revealed corrections to the citation data, as two frequently retrieved studies were not highly cited and turned out to have erroneous publication entries in the database. We provide an online resource for exploring and searching this “hall of fame”: <ext-link ext-link-type="uri" xlink:href="http://research.ics.aalto.fi/mi/setretrieval" xlink:type="simple">http://research.ics.aalto.fi/mi/setretrieval</ext-link>.</p>
<p>Earlier work on relating datasets has provided partial solutions along this line, with the major limitation of being restricted to pairwise dataset comparisons, in contrast to the proposed approach of decomposing a dataset into contributions from a set of earlier datasets. Russ and Futschik <xref ref-type="bibr" rid="pone.0113053-Russ1">[13]</xref> represented each dataset by pairwise correlations of genes, and used them to compute dataset similarities. This dataset representation is ill suited for typical functional genomics experiments, as a large number of samples is required to sensibly estimate gene correlation matrices. In addition, it makes the dataset comparison computationally expensive, as the representation is bulkier than the original dataset. In other works, specific case-control designs <xref ref-type="bibr" rid="pone.0113053-Suthram1">[14]</xref> or known biological processes <xref ref-type="bibr" rid="pone.0113053-Huttenhower1">[15]</xref> are assumed; we generalize by using decompositions over arbitrary models.</p>
<p>In summary, our work is the first approach that allows data-driven retrieval of relevant datasets by decomposing a query dataset into contributions from several earlier datasets, without requiring specific designs for the earlier datasets or their models. Unlike existing state-of-the-art retrieval, our approach is not limited to available dataset annotation. Unlike the Pfam database <xref ref-type="bibr" rid="pone.0113053-Finn1">[11]</xref>, we not only store models but use them in retrieval. Unlike existing data-driven approaches <xref ref-type="bibr" rid="pone.0113053-Caldas1">[3]</xref>, <xref ref-type="bibr" rid="pone.0113053-Schmid1">[5]</xref> that match individual observations, we match whole datasets incorporating their experimental designs. We fully decompose datasets instead of only computing pairwise similarities, as in <xref ref-type="bibr" rid="pone.0113053-Russ1">[13]</xref>, and we allow decomposition over arbitrary models available for the datasets instead of requiring restricted settings, such as specific case-control designs <xref ref-type="bibr" rid="pone.0113053-Suthram1">[14]</xref> or known biological processes <xref ref-type="bibr" rid="pone.0113053-Huttenhower1">[15]</xref>. Unlike a hypothetical approach where a unified model of all data is built, our approach is fast and scalable to large data.</p>
<sec id="s1a">
<title>Combination of Stored Models for Dataset Retrieval</title>
<p>Our goal is to infer data-driven relationships between a new “query” dataset <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e009" xlink:type="simple"/></inline-formula> and earlier datasets. The query is a dataset of <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e010" xlink:type="simple"/></inline-formula> samples <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e011" xlink:type="simple"/></inline-formula>; in the ArrayExpress study, the samples are gene expression profiles, with the element <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e012" xlink:type="simple"/></inline-formula> being expression of the gene set <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e013" xlink:type="simple"/></inline-formula> in the sample <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e014" xlink:type="simple"/></inline-formula> of the query <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e015" xlink:type="simple"/></inline-formula>, but the setup is general and applicable to other experimental data, as well. Assume further a dataset repository of <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e016" xlink:type="simple"/></inline-formula> earlier datasets, and assume that each dataset <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e017" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e018" xlink:type="simple"/></inline-formula>, has already been modeled with a model denoted by <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e019" xlink:type="simple"/></inline-formula>, later called a base model. The base models are assumed to be probabilistic generative models, <italic>i.e.</italic>, principled data descriptions capturing prior knowledge and data-driven discoveries under specific distributional assumptions. Base models for different datasets may come from different model families, as chosen by the researchers who authored each dataset. In this paper, we use two types of base models, which are discrete variants of principal component analysis (<xref ref-type="sec" rid="s2"><italic>Results</italic></xref>), but any probabilistic generative models can be applied.</p>
<p>As an illustrative setting, suppose that the dataset repository contains several datasets arising from base experiments, so that each base experiment studies one known important biological effect, the experiment has been designed so that the effect is present in the resulting dataset, and together the base experiments cover the set of known important biological effects. In the special example case of metagenomics with known constituent organisms, an obvious set of base experiments would be the set of genomes of those organisms <xref ref-type="bibr" rid="pone.0113053-Meinicke1">[16]</xref>. A new experiment could then be expressed as a combination of the base experiments, and potential novel effects. More generally, such as in a broad gene expression atlas, it would be hard, if not impossible, to settle on a clean, well-defined, and up-to-date base set of experiments to correspond to each known effect, so we chose to <italic>use the comprehensive collection of experiments in the current databases as the base experiments</italic>. The problem setting then changes from searching for a unique explanation of the new experiment to the down-to-earth and realistic task of data-driven retrieval of a set of relevant earlier experiments, relevant in the sense of having induced one or more of the known or as-of-yet unknown biological effects.</p>
<p>We combined the earlier datasets by a method that is probabilistic but simple and fast. We built a <italic>combination model</italic> for the query dataset as a mixture model of base distributions <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e020" xlink:type="simple"/></inline-formula>, which have been estimated beforehand. In our scenario, generative models <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e021" xlink:type="simple"/></inline-formula> are available in the repository along with datasets <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e022" xlink:type="simple"/></inline-formula>; note that the <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e023" xlink:type="simple"/></inline-formula> need not all have the same form. In the mixture model parameterized by <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e024" xlink:type="simple"/></inline-formula>, the likelihood of observing the query is <disp-formula id="pone.0113053.e025"><graphic position="anchor" xlink:href="info:doi/10.1371/journal.pone.0113053.e025" xlink:type="simple"/><label>(1)</label></disp-formula>where <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e026" xlink:type="simple"/></inline-formula> is the mixture proportion or <italic>weight</italic> of the <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e027" xlink:type="simple"/></inline-formula>th base distribution (model of dataset <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e028" xlink:type="simple"/></inline-formula>), and <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e029" xlink:type="simple"/></inline-formula> is the weight for the novelty term. The novelty is modeled by a background model <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e030" xlink:type="simple"/></inline-formula>, a broad nonspecific distribution covering overall gene-set activity across the whole dataset repository. All weights are non-negative and <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e031" xlink:type="simple"/></inline-formula>. In essence, this representation assumes that biological activity in the query dataset can be approximately explained as a combination of earlier datasets and a novelty term.</p>
<p>The remaining task is to infer the combination model <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e032" xlink:type="simple"/></inline-formula> for each query <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e033" xlink:type="simple"/></inline-formula> given the known models <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e034" xlink:type="simple"/></inline-formula> of datasets in the repository. We infer a maximum a posteriori (MAP) estimate of the weights <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e035" xlink:type="simple"/></inline-formula>. Alternatively, we could sample over the posterior, but MAP inference already yielded good results. We optimize the combination weights to maximize their (log) posterior probability <disp-formula id="pone.0113053.e036"><graphic position="anchor" xlink:href="info:doi/10.1371/journal.pone.0113053.e036" xlink:type="simple"/><label>(2)</label></disp-formula>where <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e037" xlink:type="simple"/></inline-formula> is a naturally non-sparse <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e038" xlink:type="simple"/></inline-formula> prior distribution for the weights with a regularization term <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e039" xlink:type="simple"/></inline-formula>. The cost function (2) is strictly concave (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>), and standard constrained convex optimization techniques can be used to find the optimized weights. Algorithmic details for the Frank-Wolfe algorithm and a proof of convergence are provided in <xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>. After computing the MAP estimate, we rank the datasets for retrieval according to decreasing combination weights.</p>
<p>This modeling-driven approach has several advantages: 1) the approximations become more accurate as more datasets are submitted to the repository, naturally increasing the number of base distributions; 2) it is fast, as only the models of the datasets are needed, not the large datasets themselves; 3) any model types can be included, as long as likelihoods of an observed sample can be computed; hence, all expert knowledge built into the models in the repository can be used; 4) relevant datasets are not assumed to be similar to the query in any na?ve sense, as they only need to explain a part of the query set; 5) the relevance scores of datasets have a natural quantitative meaning as weights in the probabilistic combination model.</p>
</sec><sec id="s1b">
<title>Scalability</title>
<p>As the size of repositories such as ArrayExpress doubles every two years or even more rapidly <xref ref-type="bibr" rid="pone.0113053-Parkinson1">[17]</xref>, fast computation with respect to the number <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e040" xlink:type="simple"/></inline-formula> of background datasets is crucial for future-proof search methods. The first method above already has a fast linear computation time in <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e041" xlink:type="simple"/></inline-formula> (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>), and an approximate variant can be run in sublinear time. For that, the model combination will be optimized only over the <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e042" xlink:type="simple"/></inline-formula> background datasets most similar to the query, which can be found in time <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e043" xlink:type="simple"/></inline-formula> where <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e044" xlink:type="simple"/></inline-formula> is an approximation parameter <xref ref-type="bibr" rid="pone.0113053-Gionis1">[18]</xref>, by suitable hashing functions.</p>
</sec></sec><sec id="s2">
<title>Results</title>
<sec id="s2a">
<title>Data-driven retrieval of experiments is more accurate than standard keyword search</title>
<p>We benchmarked the combination model against state-of-the-art dataset retrieval by keyword search, in the scenario in which a user queries with a new dataset against a database of earlier released datasets represented by models. The data were from a large human gene expression atlas <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>, containing 206 public datasets with <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e045" xlink:type="simple"/></inline-formula> samples that have been systematically annotated and consistently normalized. To make use of prior biological knowledge, we preprocessed the data by gene set enrichment analysis <xref ref-type="bibr" rid="pone.0113053-Subramanian1">[19]</xref>, representing each sample by an integer vector telling for each gene set the number of leading edge active genes <xref ref-type="bibr" rid="pone.0113053-Caldas2">[20]</xref> (<xref ref-type="sec" rid="s4"><italic>Methods</italic></xref>). As base models, we used two model types previously applied in gene expression analysis <xref ref-type="bibr" rid="pone.0113053-Caldas1">[3]</xref>, <xref ref-type="bibr" rid="pone.0113053-Gerber1">[6]</xref>, <xref ref-type="bibr" rid="pone.0113053-Caldas2">[20]</xref>, <xref ref-type="bibr" rid="pone.0113053-Engreitz1">[21]</xref>: a discrete principal component analysis method called Latent Dirichlet Allocation <xref ref-type="bibr" rid="pone.0113053-Pritchard1">[22]</xref>, <xref ref-type="bibr" rid="pone.0113053-Blei1">[23]</xref>, and a simpler variant called mixture of unigrams <xref ref-type="bibr" rid="pone.0113053-Nigam1">[24]</xref> (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>). Of the two types, for each dataset, we chose the model yielding the larger predictive likelihood (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>). For each query (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e046" xlink:type="simple"/></inline-formula>), the earlier datasets (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e047" xlink:type="simple"/></inline-formula>) were ranked in descending order of the combination proportion (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e048" xlink:type="simple"/></inline-formula>; estimated from Eq. (2)). That is, base models that explained a larger proportion of the gene set activity in the query were ranked higher. The approach yields good retrieval: the retrieval result was consistently better than with keyword searches applied to the titles and textual descriptions of the datasets (<xref ref-type="fig" rid="pone-0113053-g001">Fig. 1</xref>), which is a standard approach for dataset retrieval from repositories <xref ref-type="bibr" rid="pone.0113053-Zhu1">[25]</xref>.</p>
<fig id="pone-0113053-g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0113053.g001</object-id><label>Figure 1</label><caption>
<title>Data-driven retrieval outperforms the state of the art of keyword search on the human gene expression atlas <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>.</title>
<p>Blue: Traditional precision-recall curve where progressively more datasets are retrieved from left to right. All experiments sharing one or more of the 96 biological categories of the atlas were considered relevant. In keyword retrieval, either the category names (“Keyword: 96 classes”) or the disease annotations (“Keyword: disease”) were used as keywords. All datasets having at least 10 samples were used as query datasets, and the curves are averages over all queries.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0113053.g001" position="float" xlink:type="simple"/></fig>
<p>We checked that the result was not only due to laboratory effects by discarding, in a follow-up study, all retrieved results coming from the same laboratory. The mean average precision decreased slightly (from <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e049" xlink:type="simple"/></inline-formula> to <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e050" xlink:type="simple"/></inline-formula>; precision-recall curve in <xref ref-type="supplementary-material" rid="pone.0113053.s002">Fig. S2</xref>) but still supports the same conclusion.</p>
</sec><sec id="s2b">
<title>Network of computationally recommended dataset connections reveals biological relationships</title>
<p>When each dataset in turn is used as a query, the estimated combination weights form a “relevance network” between datasets (<xref ref-type="fig" rid="pone-0113053-g002">Fig. 2</xref>, left), where each dataset is linked to the relevant earlier datasets (for details, see <xref ref-type="sec" rid="s4"><italic>Methods</italic></xref> and an interactive searchable version at <ext-link ext-link-type="uri" xlink:href="http://research.ics.aalto.fi/mi/setretrieval" xlink:type="simple">http://research.ics.aalto.fi/mi/setretrieval</ext-link>). The network structure is dominated but not fully explained by the tissue type. Normal and neoplastic solid tissues (cluster 1) are clearly separate from cell lines (cluster 2) and from hematopoietic tissue (cluster 4); the same main clusters were observed in <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>. Note that the model has not seen the tissue types but has found them from the data. Upon closer inspection of the clusters, some finer structure is evident. The muscle and heart datasets (gray) form an interconnected subnetwork in the left edge of the image: nodes near the bottom of the image (downstream) are explained by earlier (upstream) nodes, which in turn are explained by nodes even further upstream. As another example, in cluster 4, myeloma and leukemia datasets are concentrated on the left side of the cluster, whereas the right side mostly contains normal or infected mononuclear cells.</p>
<fig id="pone-0113053-g002" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0113053.g002</object-id><label>Figure 2</label><caption>
<title>Relevance network of datasets in the human gene expression atlas; data-driven links from the model (left) and citation links (right).</title>
<p>Left: each dataset was used as a query to retrieve earlier datasets; a link from an earlier dataset to a later one means the earlier dataset is relevant as a partial model of activity in the later dataset. Link width is proportional to the normalized relevance weight (combination weight <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e051" xlink:type="simple"/></inline-formula>; only links with <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e052" xlink:type="simple"/></inline-formula> are shown, and datasets without links have been discarded). Right: links are direct (gray) and indirect (purple) citations. Node size is proportional to the estimated influence, <italic>i.e.</italic>, the total outgoing weight. Colors: tissue types (six meta tissue types <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>). The node layout was computed from the data-driven network (details in <xref ref-type="sec" rid="s4"><italic>Methods</italic></xref>).</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0113053.g002" position="float" xlink:type="simple"/></fig>
<p>There is a substantial number of links both across clusters and across tissue categories. Among the top 30 cross-category links, 25 involve heterogeneous datasets containing samples from diverse tissue origins. The strongest link connects GSE6365, a study on multile myeloma, with GSE2113, a larger study from the same lab, which largely includes the GSE6365 samples. The dataset E-MEXP-66 is a hub connected to all of the clusters and to nodes in its own cluster that have different tissue labels. It contains samples studying Kaposi sarcoma, and it also includes control samples from skin endothelial cells from blood vessels and the lymph system. Blood vessels and cells belonging to the lymph system are expected to be present in almost any solid tissue biopsy as well as in samples based on blood samples. The strongest link between two homogeneous datasets of different tissue types connects GSE3307, which compares skeletal muscle samples from healthy individuals with 12 groups of patients affected by various muscle diseases, to GSE5392, which measures the transcriptome profiles of the normal brain and a brain with bipolar disorder. Interestingly, the shortening of telomeres has been associated both with bipolar disorder <xref ref-type="bibr" rid="pone.0113053-Martinsson1">[26]</xref> and muscular disorder <xref ref-type="bibr" rid="pone.0113053-Mourkioti1">[27]</xref>. Treatment of bipolar disorder has been found to also slow down the onset of skeletal muscle disorder <xref ref-type="bibr" rid="pone.0113053-Kitazawa1">[28]</xref>.</p>
<p>Next, we investigated “outlier" datasets where the tissue type does not match the main tissue types of a cluster, implying that they might reveal commonalities between cellular conditions across tissues. Cluster 1 contained three outlier datasets: two hematopoietic datasets and one cell line dataset. The two hematopoietic outlier datasets are studies related to macrophages and are both strongly connected to GSE2004, which contains samples from the kidney, liver, and spleen, sites of long-lived macrophages. The first hematopoietic outlier, GSE2018, studies bronchoalveolar lavage cells from lung transplant receipts; the majority of these cells are macrophages. The dataset has strong links to solid tissue datasets, including GSE2004, and the diverse dataset E-MEXP-66. The second hematopoietic outlier, GSE2665, is also strongly connected to GSE2004 and measures the expression of the lymphatic organs (sentinel lymph node) that contain sinusoidal macrophages and sinusoidal endothelial cells. The third outlier, E-MEXP-101, studies a colon carcinoma cell line and has connections to other cancer datasets in cluster 1.</p>
</sec><sec id="s2c">
<title>Top dataset links overlap well with citation graph</title>
<p>We compared the model-driven network to the actual citation links (<xref ref-type="fig" rid="pone-0113053-g002">Fig. 2</xref>, right) to find out to what extent the citation practice in the research community matches the data-driven relationships. Of the top 200 data-driven edges, 50% overlapped with direct or indirect citation links (see <xref ref-type="sec" rid="s4"><italic>Methods</italic></xref>, <xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref> and <xref ref-type="supplementary-material" rid="pone.0113053.s003">Fig. S3</xref>). Most of the direct citations appear within the four tissue clusters (<xref ref-type="fig" rid="pone-0113053-g002">Fig. 2</xref>, right). The two cross-cluster citations are not due to the biological similarity of the datasets. The publication for GSE1869 cites the publication for GSE1159 regarding the method of differential expression detection. The GSE7007, a study on Ewing sarcoma samples, cites the study on human mesenchymal stem cells (E-MEXP-168), stating that the overall gene expression profiles differ between those samples.</p>
<p>We additionally compared the densely connected sets of experiments between the two networks. In the citation graph, the breast cancer datasets GSE2603, GSE3494, GSE2990, GSE4922, and GSE1456 form an interconnected clique in cluster 1, while the three leukocyte datasets GSE2328, GSE3284, and GSE5580 form an interconnected module in cluster 4. In the relevance network, the corresponding edges for both cliques are among the strongest links for those datasets, and some of them are among the top 20 strongest edges in the network (see <xref ref-type="supplementary-material" rid="pone.0113053.s005">Table S1</xref> for the list of top 20 edges). There are also densely connected modules in the relevance network that are not strongly connected in the citation graph; when we systematically sought cliques associated with each of the top 20 edges, the strongest edges constitute a clique among E-MEXP-750, GSE6740, and GSE473, all three studying CD4+ T helper cells, which are an essential part of the human immune system. Another interesting set is among three T-cell related datasets in cluster 3. Two of the datasets contain T lymphoblastic leukemia samples (E-MEXP-313 and E-MEXP-549), whereas E-MEXP-337 reports thymocyte profiles. Thymocytes are developing T lymphocytes that are matured in thymus, so this connection is biologically meaningful but not straightforward to find from dataset annotations. Other strongly connected cliques are discussed in <xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>.</p>
</sec><sec id="s2d">
<title>Analysis of network hubs discovers datasets deserving more citations</title>
<p>Datasets that have high weights in explaining other datasets have a large weighted outdegree in the data-driven relevance network, and they are expected to be useful for many other studies. We checked whether the publications corresponding to these <italic>central hubs</italic> are highly cited in the research community. There is a low but statistically significant correlation between the weighted outdegree of datasets and their citation counts (<xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>; Spearman <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e053" xlink:type="simple"/></inline-formula>, <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e054" xlink:type="simple"/></inline-formula>). Both quantities were normalized to avoid bias due to different release times of the datasets (<xref ref-type="sec" rid="s4"><italic>Methods</italic></xref>). We further examined whether the prestige of the publication venue (measured by impact factor) and the senior author (h-index of the last author) biased the citation counts, which could explain the low correlation between the outdegree and the citation count, and the answer was affirmative (<xref ref-type="sec" rid="s4"><italic>Methods</italic></xref>).</p>
<fig id="pone-0113053-g003" position="float"><object-id pub-id-type="doi">10.1371/journal.pone.0113053.g003</object-id><label>Figure 3</label><caption>
<title>Data-driven prediction of usefulness of datasets vs. their citation counts.</title>
<p>Manual checks comparing sets for which the two scores differed revealed inconsistent database records for two datasets; the blue arrows point to their corrected locations, which are more in line with the data-driven model. Regions A, B, and C: see text.</p>
</caption><graphic mimetype="image" xlink:href="info:doi/10.1371/journal.pone.0113053.g003" position="float" xlink:type="simple"/></fig>
<p>We inspected more closely the datasets where the recommended or the actual citation counts were high (<xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>): (A) datasets having low citation counts but high outdegrees, (B) datasets having both high citation counts and high outdegrees, and (C) datasets having high citation counts but low outdegrees. We manually checked the publication records of region A in Gene Expression Omnibus (GEO) <xref ref-type="bibr" rid="pone.0113053-Barrett1">[29]</xref> and ArrayExpress <xref ref-type="bibr" rid="pone.0113053-Parkinson1">[17]</xref>, to find out why the datasets had low citation counts despite their high outdegree (data-driven citation recommendations). Two of the eight datasets had an inconsistent publication record. The blue arrows in <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref> point from their original position to the corrected position confirmed by GEO and ArrayExpress. Thus, the data-driven network revealed the inconsistency, and the new positions, corresponding to higher citation counts, validate the model-based finding that these datasets are good explainers for other datasets. In region B, most of the papers have been published in high-impact journals and have a relatively high number of samples (average sample size of <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e055" xlink:type="simple"/></inline-formula>) compared to region A (average sample size of <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e056" xlink:type="simple"/></inline-formula>). One of the eight datasets in the collection is the well-known Connectivity Map experiment (GSE5258). Lastly, the set C mostly contains unique targeted studies; there are five studies in the set, which are about leukocytes of injured patients, Polycomb group (PcG) proteins, senescence, Alzheimer's disease, and the effect of cAMP agonist forskolin, a traditional Indian medicine. The studies have been published in high-impact forums, and a possible reason of their low outdegree is their specific cellular responses, which are not very common in the atlas.</p>
</sec></sec><sec id="s3">
<title>Discussion</title>
<p>Our main goal was to test the feasibility of the scenario where researchers let the data speak for themselves when relating new research to earlier studies. The conclusion is positive: even a relatively straightforward and scalable mixture modeling approach found both expected relationships such as tissue types, and relationships not easily found with keyword searches, including cells in different developmental stages or treatments resembling conditions in other cell types. While biologists could find such connections by bringing expert knowledge into keyword searches, the ultimate advantage of the data-driven approach is that it also yields connections beyond current knowledge, giving rise to new hypotheses and follow-up studies. For example, it seems surprising that the skeletal muscle dataset GSE6011 is linked also to kidney and brain datasets. Closer inspection yielded possible partial explanations. Some kidney areas are rich in blood vessels, lined by smooth muscle. Studies have shown common gene signatures between skeletal muscle and brain. Abnormal expression of the protein dystrophin leads to Duchenne muscular dystrophy, exhibited by a majority of samples in GSE6011; the brain is another major expression site for dystrophin <xref ref-type="bibr" rid="pone.0113053-Culligan1">[30]</xref>. Interestingly, the top three potentially novel datasets, where only less than 50% of the expression pattern is modelled by earlier datasets (i.e., <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e057" xlink:type="simple"/></inline-formula>), are GSE2603 (a central breast cancer set), the Connectivity Map data (GSE5258), and the Burkitt's Lymphoma set (GSE4475, a cancer fundamentally distinct from other types of lymphoma). The first two are also recovered by the citation data (as they have relatively high citation counts and appear in region B in <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>), unlike the third (which is part of region A in <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>).</p>
<p>Our case study focused on a global analysis of the relevance network obtained for a representative dataset collection, allowing for comparisons with the citation graph. The data-driven relationships corresponded to actual citations when available but were richer and were able to spot out errors in citation links. Another intended use of the retrieval method is to support researchers in finding relevant data on a particular topic of interest. We performed a study with additional skeletal muscle datasets (<xref ref-type="supplementary-material" rid="pone.0113053.s006">Table S2</xref>) to obtain insights into relationships among skeletal muscle datasets (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>) as well as between skeletal muscle and other datasets (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref> and <xref ref-type="supplementary-material" rid="pone.0113053.s007">Table S3</xref>), and we showed that the retrieval method lessens the need for laborious manual searches (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref> and <xref ref-type="supplementary-material" rid="pone.0113053.s004">Fig. S4</xref>).</p>
<p>In this work, we made simplifying assumptions: we only employed two model families, included biological knowledge only as pre-chosen gene sets, and assumed all new experiments to be mixtures of earlier ones, instead of finding common effects in them and combining them either as mixtures or sums. We expect the results to improve considerably with more advanced future alternatives, with the research challenge being to maintain scalability. Generalizability of the search across measurement batches, laboratories, and measurement platforms is a challenge. Our feasibility study showed that for carefully preprocessed datasets (of the microarray atlas <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref>), data-driven retrieval is useful even across laboratories. Our method is generally applicable to any single platform, and it takes into account the expert knowledge built into models of datasets for that platform; abstraction-based data representations, such as the gene set enrichment representation we used, have the potential to facilitate cross-platform analyses. As data integration approaches develop further <xref ref-type="bibr" rid="pone.0113053-Tripathi1">[31]</xref>, <xref ref-type="bibr" rid="pone.0113053-Virtanen1">[32]</xref>, it may be possible to do searches even across different omics types; here, integration of meta data (pioneered in a specific semi-supervised framework <xref ref-type="bibr" rid="pone.0113053-Wise1">[33]</xref>), several ontologies (MGED ontology, experimental factor ontology, and ontology of biomedical investigations <xref ref-type="bibr" rid="pone.0113053-Zheng1">[34]</xref>) and text mining results <xref ref-type="bibr" rid="pone.0113053-Jensen1">[35]</xref>, <xref ref-type="bibr" rid="pone.0113053-Rzhetsky1">[36]</xref> are obviously useful first steps.</p>
</sec><sec id="s4" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="s4a">
<title>Gene expression data</title>
<p>We used the human gene expression atlas <xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref> available at ArrayExpress under accession number E-MTAB-62. The data were preprocessed by gene set enrichment analysis (GSEA) using the canonical pathway collection (C2-CP) from the Molecular Signatures Database <xref ref-type="bibr" rid="pone.0113053-Subramanian1">[19]</xref>. Each sample was represented by its top enriched gene sets <xref ref-type="bibr" rid="pone.0113053-Caldas2">[20]</xref> (<xref ref-type="supplementary-material" rid="pone.0113053.s008">Text S1</xref>).</p>
</sec><sec id="s4b">
<title>Node layout and normalized relevance weight</title>
<p>The weight matrix contains a weight vector for each query dataset, encoding the amount of variation in that query explained by each earlier dataset. As query datasets from early years have only a few even earlier sets available, there is a bias towards the edges being stronger for the datasets from early years. To remove the bias we normalized, for the visualizations, the edge strengths of each query data set by the number of earlier datasets. To visualize the relationship network over time in <xref ref-type="fig" rid="pone-0113053-g002">Fig. 2</xref>, we needed a layout algorithm that positions the datasets on the horizontal axis highlighting structure and avoiding tangling. We used a <italic>cluster-emphasizing</italic> Sammon's mapping; Sammon's mapping <xref ref-type="bibr" rid="pone.0113053-Sammon1">[37]</xref> is a nonlinear projection method or multidimensional scaling algorithm that aims at preserving the interpoint distances (here <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e058" xlink:type="simple"/></inline-formula>). By clustering the network (with unsupervised Markov clustering <xref ref-type="bibr" rid="pone.0113053-vanDongen1">[38]</xref>) and increasing between-cluster distances by adding a constant (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e059" xlink:type="simple"/></inline-formula>) to them, the mapping was made to emphasize clusters and hence untangle the layout.</p>
</sec><sec id="s4c">
<title>Citation graph</title>
<p>Direct citations between dataset-linked publications were extracted from the Web of Science (26 Jul 2012) and PubMed (17 Oct 2012). We additionally considered two types of indirect edges. Firstly, we introduced links between datasets whose publications share common references. This covers, for instance, related datasets whose publications appeared close in time, making direct citation unlikely. A natural measure of edge strength is given by the number of shared references. Secondly, we connect datasets whose articles are cited together, because co-citation is a sign that the community perceives the articles as related. Here, the edge strength was taken to be the number of articles co-citing the two dataset publications; these edges dominate the indirect links in the citation graph. For this analysis, we used citation data, available for <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e060" xlink:type="simple"/></inline-formula> datasets and provided by Thomson Reuters as of 13 September 2012.</p>
</sec><sec id="s4d">
<title>Normalization of citation counts and weighted outdegrees</title>
<p>As early datasets have many more papers that can cite them and many more later datasets that they can help model, both the citation counts and estimated weighted outdegrees are expected to be upwards biased for them. For <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>, we normalized the quantities; for each dataset, we normalized the outdegree by the number of newer datasets and the citation count by the time difference between publishing the data and the newest dataset in the atlas. To make sure the normalization did not introduce side effects, we additionally checked that the same conclusions were reached without the citation count normalization (<xref ref-type="supplementary-material" rid="pone.0113053.s001">Fig. S1</xref>; plotted as stratified subfigures for each 1-year time window). The citation counts were extracted from PubMed on 16 May 2012.</p>
</sec><sec id="s4e">
<title>Citation counts are strongly influenced by external esteem of the publication forum and the senior author</title>
<p>We stratified the data sets according to the numbers of data-driven citation recommendations, and studied whether the impact factor of the forum or the h-index of the last author were predictive of the actual citation count in each stratum. The strata were the top and bottom quartiles, and for each, we compared the top and bottom quartiles of the actual citation counts (resulting in comparing the four corners of <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref>). For low outdegree (low recommended citation count), the h-index was lower for less cited datasets (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e061" xlink:type="simple"/></inline-formula>; mean value <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e062" xlink:type="simple"/></inline-formula> vs <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e063" xlink:type="simple"/></inline-formula>), and the impact factor was lower (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e064" xlink:type="simple"/></inline-formula>; mean value <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e065" xlink:type="simple"/></inline-formula> vs <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e066" xlink:type="simple"/></inline-formula>). Similarly, for the high recommended citation count, the impact factor for the little-cited datasets was lower (<inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e067" xlink:type="simple"/></inline-formula>; mean value <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e068" xlink:type="simple"/></inline-formula> vs <inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e069" xlink:type="simple"/></inline-formula>), while the difference in h-index was not significant. All t statistics and p-values were computed by one-sided independent sample Welch's t-tests. The h-indices and impact factors were collected from Thomson Reuters Web of Knowledge and Journal Citation Reports 2011, respectively, on 23rd July 2012.</p>
</sec></sec><sec id="s5">
<title>Supporting Information</title>
<supplementary-material id="pone.0113053.s001" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pone.0113053.s001" position="float" xlink:type="simple"><label>Figure S1</label><caption>
<p><bold>Stratified data-driven prediction of usefulness of datasets vs. their citation counts.</bold> Black solid lines mark the boundary for potentially interesting datasets; the boundaries are set to hold the same percentiles of data as in <xref ref-type="fig" rid="pone-0113053-g003">Fig. 3</xref> in the main paper. <italic>ImpFac</italic> stands for Impact Factor of the publication venue.</p>
<p>(TIFF)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s002" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pone.0113053.s002" position="float" xlink:type="simple"><label>Figure S2</label><caption>
<p><bold>Removal of laboratory effects changes the retrieval performance only slightly, as measured by the precision-recall curves.</bold> <italic>Original</italic>: Replicated from <xref ref-type="fig" rid="pone-0113053-g001">Fig. 1</xref> of the main paper; <italic>Lab. effects removed</italic>: all retrieval results from the same laboratory as the query data have been discarded.</p>
<p>(TIFF)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s003" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pone.0113053.s003" position="float" xlink:type="simple"><label>Figure S3</label><caption>
<p><bold>Overlap of data-driven recommendations with the actual citation graph: Precision </bold><inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e070" xlink:type="simple"/></inline-formula><bold> for top edges that explain more than </bold><inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e071" xlink:type="simple"/></inline-formula><bold> variation.</bold> The gold standard is the extended citation graph, which is built as the union of edges from 1) the original directed graph, 2) between any two articles that are cited together by some other article, and 3) between any two articles that have at least one common reference.</p>
<p>(TIFF)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s004" mimetype="image/tiff" xlink:href="info:doi/10.1371/journal.pone.0113053.s004" position="float" xlink:type="simple"><label>Figure S4</label><caption>
<p><bold>Retrieval performance evaluation of the data-driven model against keyword search in the skeletal muscle case study.</bold> The precision-recall curves are averaged across the 16 skeletal muscle datasets having at least 10 samples.</p>
<p>(TIFF)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s005" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pone.0113053.s005" position="float" xlink:type="simple"><label>Table S1</label><caption>
<p><bold>Top </bold><inline-formula><inline-graphic xlink:href="info:doi/10.1371/journal.pone.0113053.e072" xlink:type="simple"/></inline-formula><bold> strongest edges in the relevance network.</bold></p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s006" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pone.0113053.s006" position="float" xlink:type="simple"><label>Table S2</label><caption>
<p><bold>ArrayExpress accession numbers of 16 skeletal muscle datasets used in the retrieval case study in addition to the human gene expression atlas </bold><xref ref-type="bibr" rid="pone.0113053-Lukk1">[12]</xref><bold>.</bold> All datasets were measured with the human genome platform HG-U133A, the same used in the atlas.</p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s007" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xlink:href="info:doi/10.1371/journal.pone.0113053.s007" position="float" xlink:type="simple"><label>Table S3</label><caption>
<p><bold>Skeletal muscle queries with at least one retrieved non-skeletal muscle dataset, sorted according to decreasing precision.</bold></p>
<p>(XLSX)</p>
</caption></supplementary-material><supplementary-material id="pone.0113053.s008" mimetype="application/pdf" xlink:href="info:doi/10.1371/journal.pone.0113053.s008" position="float" xlink:type="simple"><label>Text S1</label><caption>
<p><bold>More details on methods and results.</bold></p>
<p>(PDF)</p>
</caption></supplementary-material></sec></body>
<back>
<ack>
<p>We thank Matti Nelimarkka and Tuukka Ruotsalo for helping with citation data. Certain data included herein are derived from the following indices: Science Citation Index Expanded, Social Science Citation Index and Arts &amp; Humanities Citation Index, prepared by Thomson Reuters, Philadelphia, Pennsylvania, USA, Copyright Thomson Reuters, 2011.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pone.0113053-Greene1"><label>1</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Greene</surname><given-names>CS</given-names></name>, <name name-style="western"><surname>Troyanskaya</surname><given-names>OG</given-names></name> (<year>2011</year>) <article-title>PILGRM: An interactive data-driven discovery platform for expert biologists</article-title>. <source>Nucleic Acids Res</source> <volume>39</volume>:<fpage>W368</fpage>–<lpage>374</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Tanay1"><label>2</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Tanay</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Steinfeld</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Kupiec</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Shamir</surname><given-names>R</given-names></name> (<year>2005</year>) <article-title>Integrative analysis of genome-wide experiments in the context of a large high-throughput data compendium</article-title>. <source>Mol Syst Biol</source> <volume>1</volume>:<fpage>e1</fpage>–<lpage>10</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Caldas1"><label>3</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Caldas</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Gehlenborg</surname><given-names>N</given-names></name>, <name name-style="western"><surname>Kettunen</surname><given-names>E</given-names></name>, <name name-style="western"><surname>Faisal</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Rönty</surname><given-names>M</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>Data-driven information retrieval in heterogeneous collections of transcriptomics data links <italic>SIM2s</italic> to malignant pleural mesothelioma</article-title>. <source>Bioinformatics</source> <volume>28</volume>:<fpage>i246</fpage>–<lpage>i253</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Adler1"><label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Adler</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Kolde</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Kull</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Tkachenko</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Peterson</surname><given-names>H</given-names></name>, <etal>et al</etal>. (<year>2009</year>) <article-title>Mining for coexpression across hundreds of datasets using novel rank aggregation and visualization methods</article-title>. <source>Genome Biol</source> <volume>10</volume>:<fpage>R139</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Schmid1"><label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schmid</surname><given-names>PR</given-names></name>, <name name-style="western"><surname>Palmer</surname><given-names>NP</given-names></name>, <name name-style="western"><surname>Kohane</surname><given-names>IS</given-names></name>, <name name-style="western"><surname>Berger</surname><given-names>B</given-names></name> (<year>2012</year>) <article-title>Making sense out of massive data by going beyond differential expression</article-title>. <source>Proc Natl Acad Sci U S A</source> <volume>109</volume>:<fpage>5594</fpage>–<lpage>5599</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Gerber1"><label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Gerber</surname><given-names>GK</given-names></name>, <name name-style="western"><surname>Dowell</surname><given-names>RD</given-names></name>, <name name-style="western"><surname>Jaakkola</surname><given-names>TS</given-names></name>, <name name-style="western"><surname>Gifford</surname><given-names>DK</given-names></name> (<year>2007</year>) <article-title>Automated discovery of functional generality of human gene expression programs</article-title>. <source>PLoS Comput Biol</source> <volume>3</volume>:<fpage>e148</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Tseng1"><label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Tseng</surname><given-names>GC</given-names></name>, <name name-style="western"><surname>Ghosh</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Feingold</surname><given-names>E</given-names></name> (<year>2012</year>) <article-title>Comprehensive literature review and statistical considerations for microarray meta-analysis</article-title>. <source>Nucleic Acids Res</source> <volume>40</volume>:<fpage>3785</fpage>–<lpage>3799</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Rung1"><label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rung</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Brazma</surname><given-names>A</given-names></name> (<year>2012</year>) <article-title>Reuse of public genome-wide gene expression data</article-title>. <source>Nature Rev Genet</source> <volume>14</volume>:<fpage>89</fpage>–<lpage>99</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Baxter1"><label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Baxter</surname><given-names>J</given-names></name> (<year>1997</year>) <article-title>A Bayesian/information theoretic model of learning to learn via multiple task sampling</article-title>. <source>Machine Learning</source> <volume>28</volume>:<fpage>7</fpage>–<lpage>39</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Caruana1"><label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Caruana</surname><given-names>R</given-names></name> (<year>1997</year>) <article-title>Multitask learning</article-title>. <source>Machine Learning</source> <volume>28</volume>:<fpage>41</fpage>–<lpage>75</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Finn1"><label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Finn</surname><given-names>RD</given-names></name>, <name name-style="western"><surname>Bateman</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Clements</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Coggill</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Eberhardt</surname><given-names>RY</given-names></name>, <etal>et al</etal>. (<year>2012</year>) <article-title>The Pfam protein families database</article-title>. <source>Nucleic Acids Research</source> <volume>40</volume>:<fpage>D290</fpage>–<lpage>D301</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Lukk1"><label>12</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Lukk</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Kapushesky</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Nikkila</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Parkinson</surname><given-names>H</given-names></name>, <name name-style="western"><surname>Goncalves</surname><given-names>A</given-names></name>, <etal>et al</etal>. (<year>2010</year>) <article-title>A global map of human gene expression</article-title>. <source>Nat Biotechnol</source> <volume>28</volume>:<fpage>322</fpage>–<lpage>324</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Russ1"><label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Russ</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Futschik</surname><given-names>ME</given-names></name> (<year>2010</year>) <article-title>Comparison and consolidation of microarray data sets of human tissue expression</article-title>. <source>BMC Genomics</source> <volume>11</volume>:<fpage>305</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Suthram1"><label>14</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Suthram</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Dudley</surname><given-names>JT</given-names></name>, <name name-style="western"><surname>Chiang</surname><given-names>AP</given-names></name>, <name name-style="western"><surname>Chen</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Hastie</surname><given-names>TJ</given-names></name>, <name name-style="western"><surname>Butte</surname><given-names>AJ</given-names></name> (<year>2010</year>) <article-title>Network-based elucidation of human disease similarities reveals common functional modules enriched for pluripotent drug targets</article-title>. <source>PLoS Comput Biol</source> <volume>6</volume>:<fpage>e1000662</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Huttenhower1"><label>15</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Huttenhower</surname><given-names>C</given-names></name>, <name name-style="western"><surname>Troyanskaya</surname><given-names>OG</given-names></name> (<year>2008</year>) <article-title>Assessing the functional structure of genomic data</article-title>. <source>Bioinformatics</source> <volume>24</volume>:<fpage>i330</fpage>–<lpage>338</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Meinicke1"><label>16</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Meinicke</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Asshauer</surname><given-names>KP</given-names></name>, <name name-style="western"><surname>Lingner</surname><given-names>T</given-names></name> (<year>2011</year>) <article-title>Mixture models for analysis of the taxonomic composition of metagenomes</article-title>. <source>Bioinformatics</source> <volume>27</volume>:<fpage>1618</fpage>–<lpage>1624</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Parkinson1"><label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Parkinson</surname><given-names>H</given-names></name>, <name name-style="western"><surname>Kapushesky</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Kolesnikov</surname><given-names>N</given-names></name>, <name name-style="western"><surname>Rustici</surname><given-names>G</given-names></name>, <name name-style="western"><surname>Shojatalab</surname><given-names>M</given-names></name>, <etal>et al</etal>. (<year>2009</year>) <article-title>ArrayExpress update—from an archive of functional genomics experiments to the atlas of gene expression</article-title>. <source>Nucleic Acids Res</source> <volume>37</volume>:<fpage>D868</fpage>–<lpage>72</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Gionis1"><label>18</label>
<mixed-citation publication-type="other" xlink:type="simple">Gionis A, Indyk P, Motwani R (1999) Similarity search in high dimensions via hashing. In: Proc 25th VLDB Conf. San Francisco, CA: Morgan Kaufmann, pp. 518–529.</mixed-citation>
</ref>
<ref id="pone.0113053-Subramanian1"><label>19</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Subramanian</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Tamayo</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Mootha</surname><given-names>VK</given-names></name>, <name name-style="western"><surname>Mukherjee</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Ebert</surname><given-names>BL</given-names></name>, <etal>et al</etal>. (<year>2005</year>) <article-title>Gene set enrichment analysis: A knowledge-based approach for interpreting genome-wide expression profiles</article-title>. <source>Proc Natl Acad Sci U S A</source> <volume>102</volume>:<fpage>15545</fpage>–<lpage>15550</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Caldas2"><label>20</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Caldas</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Gehlenborg</surname><given-names>N</given-names></name>, <name name-style="western"><surname>Faisal</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Brazma</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Kaski</surname><given-names>S</given-names></name> (<year>2009</year>) <article-title>Probabilistic retrieval and visualization of biologically relevant microarray experiments</article-title>. <source>Bioinformatics</source> <volume>25</volume>:<fpage>i145</fpage>–<lpage>i153</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Engreitz1"><label>21</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Engreitz</surname><given-names>JM</given-names></name>, <name name-style="western"><surname>Morgan</surname><given-names>AA</given-names></name>, <name name-style="western"><surname>Dudley</surname><given-names>JT</given-names></name>, <name name-style="western"><surname>Chen</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Thathoo</surname><given-names>R</given-names></name>, <etal>et al</etal>. (<year>2010</year>) <article-title>Content-based microarray search using differential expression profiles</article-title>. <source>BMC Bioinformatics</source> <volume>11</volume>:<fpage>603</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Pritchard1"><label>22</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Pritchard</surname><given-names>JK</given-names></name>, <name name-style="western"><surname>Stephens</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Donnelly</surname><given-names>P</given-names></name> (<year>2000</year>) <article-title>Inference of population structure using multilocus genotype data</article-title>. <source>Genetics</source> <volume>155</volume>:<fpage>945</fpage>–<lpage>959</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Blei1"><label>23</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Blei</surname><given-names>DM</given-names></name>, <name name-style="western"><surname>Ng</surname><given-names>AY</given-names></name>, <name name-style="western"><surname>Jordan</surname><given-names>MI</given-names></name>, <name name-style="western"><surname>Lafferty</surname><given-names>J</given-names></name> (<year>2003</year>) <article-title>Latent Dirichlet allocation</article-title>. <source>J Mach Learn Res</source> <volume>3</volume>:<fpage>993</fpage>–<lpage>1022</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Nigam1"><label>24</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nigam</surname><given-names>K</given-names></name>, <name name-style="western"><surname>McCallum</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Thrun</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Mitchell</surname><given-names>T</given-names></name> (<year>2000</year>) <article-title>Text classification from labeled and unlabeled documents using EM</article-title>. <source>Machine Learning</source> <volume>39</volume>:<fpage>103</fpage>–<lpage>134</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Zhu1"><label>25</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names></name>, <name name-style="western"><surname>Davis</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Stephens</surname><given-names>R</given-names></name>, <name name-style="western"><surname>Meltzer</surname><given-names>PS</given-names></name>, <name name-style="western"><surname>Chen</surname><given-names>Y</given-names></name> (<year>2008</year>) <article-title>GEOmetadb: powerful alternative search engine for the Gene Expression Omnibus</article-title>. <source>Bioinformatics</source> <volume>24</volume>:<fpage>2798</fpage>–<lpage>2800</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Martinsson1"><label>26</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Martinsson</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Wei</surname><given-names>Y</given-names></name>, <name name-style="western"><surname>Xu</surname><given-names>D</given-names></name>, <name name-style="western"><surname>Melas</surname><given-names>PA</given-names></name>, <name name-style="western"><surname>Mathé</surname><given-names>AA</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Long-term lithium treatment in bipolar disorder is associated with longer leukocyte telomeres</article-title>. <source>Transl Psychiatry</source> <volume>3</volume>:<fpage>e261</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Mourkioti1"><label>27</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Mourkioti</surname><given-names>F</given-names></name>, <name name-style="western"><surname>Kustan</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Kraft</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Dav</surname><given-names>JW</given-names></name>, <name name-style="western"><surname>Zhao</surname><given-names>MM</given-names></name>, <etal>et al</etal>. (<year>2013</year>) <article-title>Role of telomere dysfunction in cardiac failure in Duchenne muscular dystrophy</article-title>. <source>Nature Cell Bio</source> <volume>15</volume>:<fpage>895</fpage>–<lpage>904</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Kitazawa1"><label>28</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kitazawa</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Trinh</surname><given-names>DN</given-names></name>, <name name-style="western"><surname>LaFerla</surname><given-names>FM</given-names></name> (<year>2008</year>) <article-title>Inflammation induces tau pathology in inclusion body myositis model via glycogen synthase kinase-3 beta</article-title>. <source>Ann Neurol</source> <volume>64</volume>:<fpage>15</fpage>–<lpage>24</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Barrett1"><label>29</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Barrett</surname><given-names>T</given-names></name>, <name name-style="western"><surname>Troup</surname><given-names>DB</given-names></name>, <name name-style="western"><surname>Wilhite</surname><given-names>SE</given-names></name>, <name name-style="western"><surname>Ledoux</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Evangelista</surname><given-names>C</given-names></name>, <etal>et al</etal>. (<year>2011</year>) <article-title>NCBI GEO: archive for functional genomics data sets-10 years on</article-title>. <source>Nucleic Acids Res</source> <volume>39</volume>:<fpage>D1005</fpage>–<lpage>D1010</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Culligan1"><label>30</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Culligan</surname><given-names>K</given-names></name>, <name name-style="western"><surname>Glover</surname><given-names>L</given-names></name>, <name name-style="western"><surname>Dowling</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Ohlendieck</surname><given-names>K</given-names></name> (<year>2001</year>) <article-title>Brain dystrophin-glycoprotein complex: Persistent expression of beta-dystroglycan, impaired oligomerization of Dp71 and up-regulation of utrophins in animal models of muscular dystrophy</article-title>. <source>BMC Cell Biol</source> <volume>2</volume>:<fpage>2</fpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Tripathi1"><label>31</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Tripathi</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Klami</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Orešič</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Kaski</surname><given-names>S</given-names></name> (<year>2011</year>) <article-title>Matching samples of multiple views</article-title>. <source>Data Min Knowl Discov</source> <volume>23</volume>:<fpage>300</fpage>–<lpage>321</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Virtanen1"><label>32</label>
<mixed-citation publication-type="other" xlink:type="simple">Virtanen S, Klami A, Khan SA, Kaski S (2012) Bayesian group factor analysis. In: Lawrence N, Girolami M, editors. International Conference on Artificial Intelligence and Statistics. Vol. 22 of <italic>JMLR W&amp;CP</italic>, pp. 1269–1277.</mixed-citation>
</ref>
<ref id="pone.0113053-Wise1"><label>33</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wise</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Oltvai</surname><given-names>Z</given-names></name>, <name name-style="western"><surname>Bar-Joseph</surname><given-names>Z</given-names></name> (<year>2012</year>) <article-title>Matching experiments across species using expression values and textual information</article-title>. <source>Bioinformatics</source> <volume>28</volume>:<fpage>i258</fpage>–<lpage>i264</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Zheng1"><label>34</label>
<mixed-citation publication-type="other" xlink:type="simple">Zheng J, Stoyanovich J, Manduchi E, Liu J, Stoeckert CJ (2011) Annotcompute: annotation-based exploration and meta-analysis of genomics experiments. Database: Oxford. doi:10.1093/database/bar045</mixed-citation>
</ref>
<ref id="pone.0113053-Jensen1"><label>35</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Jensen</surname><given-names>LJ</given-names></name>, <name name-style="western"><surname>Saric</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Bork</surname><given-names>P</given-names></name> (<year>2006</year>) <article-title>Literature mining for the biologist: from information retrieval to biological discovery</article-title>. <source>Nat Rev Genet</source> <volume>7</volume>:<fpage>119</fpage>–<lpage>129</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Rzhetsky1"><label>36</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rzhetsky</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Seringhaus</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Gerstein</surname><given-names>M</given-names></name> (<year>2008</year>) <article-title>Seeking a new biology through text mining</article-title>. <source>Cell</source> <volume>134</volume>:<fpage>9</fpage>–<lpage>13</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-Sammon1"><label>37</label>
<mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Sammon</surname><given-names>JW</given-names></name> (<year>1969</year>) <article-title>A nonlinear mapping for data structure analysis</article-title>. <source>IEEE Trans Comput</source> <volume>18</volume>:<fpage>401</fpage>–<lpage>409</lpage>.</mixed-citation>
</ref>
<ref id="pone.0113053-vanDongen1"><label>38</label>
<mixed-citation publication-type="other" xlink:type="simple">van Dongen S (2000) Graph Clustering by Flow Simulation. Ph.D. thesis, University of Utrecht.</mixed-citation>
</ref>
</ref-list></back>
</article>