<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="letter" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id>
<journal-title-group>
<journal-title>PLOS Computational Biology</journal-title>
</journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1012403</article-id>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-24-00510</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Formal Comment</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Organisms</subject><subj-group><subject>Eukaryota</subject><subj-group><subject>Animals</subject><subj-group><subject>Vertebrates</subject><subj-group><subject>Amniotes</subject><subj-group><subject>Mammals</subject><subj-group><subject>Elephants</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Zoology</subject><subj-group><subject>Animals</subject><subj-group><subject>Vertebrates</subject><subj-group><subject>Amniotes</subject><subj-group><subject>Mammals</subject><subj-group><subject>Elephants</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Data management</subject><subj-group><subject>Data visualization</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Science policy</subject><subj-group><subject>Science and technology workforce</subject><subj-group><subject>Careers in research</subject><subj-group><subject>Scientists</subject><subj-group><subject>Biologists</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>People and places</subject><subj-group><subject>Population groupings</subject><subj-group><subject>Professions</subject><subj-group><subject>Scientists</subject><subj-group><subject>Biologists</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject><subj-group><subject>Structural genomics</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Engineering and technology</subject><subj-group><subject>Industrial engineering</subject><subj-group><subject>Quality control</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Earth sciences</subject><subj-group><subject>Geomorphology</subject><subj-group><subject>Topography</subject><subj-group><subject>Landforms</subject><subj-group><subject>Islands</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Computational biology</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Genome annotation</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Genome annotation</subject></subj-group></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>The art of seeing the elephant in the room: 2D embeddings of single-cell data do make sense</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0946-412X</contrib-id>
<name name-style="western">
<surname>Lause</surname>
<given-names>Jan</given-names>
</name>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Berens</surname>
<given-names>Philipp</given-names>
</name>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5639-7209</contrib-id>
<name name-style="western">
<surname>Kobak</surname>
<given-names>Dmitry</given-names>
</name>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
</contrib-group>
<aff id="aff001"><label>1</label> <addr-line>Hertie Institute for AI in Brain Health, University of Tübingen, Tübingen, Germany</addr-line></aff>
<aff id="aff002"><label>2</label> <addr-line>Tübingen AI Center, University of Tübingen, Tübingen, Germany</addr-line></aff>
<aff id="aff003"><label>3</label> <addr-line>IWR, Heidelberg University, Heidelberg, Germany</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Papin</surname>
<given-names>Jason A.</given-names>
</name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1"><addr-line>University of Virginia, UNITED STATES OF AMERICA</addr-line></aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">dmitry.kobak@uni-tuebingen.de</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>2</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<month>10</month>
<year>2024</year>
</pub-date>
<volume>20</volume>
<issue>10</issue>
<elocation-id>e1012403</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>3</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>9</day>
<month>8</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-year>2024</copyright-year>
<copyright-holder>Lause et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited. All used datasets are publicly available (see Table A in <xref ref-type="supplementary-material" rid="pcbi.1012403.s001">S1 Text</xref>). Our code in Python is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/berenslab/elephant-in-the-room" xlink:type="simple">https://github.com/berenslab/elephant-in-the-room</ext-link>.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pcbi.1012403"/>
<related-article ext-link-type="uri" id="related001" related-article-type="companion" xlink:href="info:doi/10.1371/journal.pcbi.1011288" xlink:type="simple">
<article-title>The specious art of single-cell genomics</article-title>
</related-article>
<abstract>
<p>A recent paper claimed that <italic>t</italic>-SNE and UMAP embeddings of single-cell datasets are “specious” and fail to capture true biological structure. The authors argued that such embeddings are as arbitrary and as misleading as forcing the data into an elephant shape. Here we show that this conclusion was based on inadequate and limited metrics of embedding quality. More appropriate metrics quantifying neighborhood and class preservation reveal the elephant in the room: while <italic>t</italic>-SNE and UMAP embeddings of single-cell data do not preserve high-dimensional distances, they can nevertheless provide biologically relevant information.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100001659</institution-id>
<institution>Deutsche Forschungsgemeinschaft</institution>
</institution-wrap>
</funding-source>
<award-id>EXC 2064, 390727645</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Berens</surname>
<given-names>Philipp</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="award002">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100001659</institution-id>
<institution>Deutsche Forschungsgemeinschaft</institution>
</institution-wrap>
</funding-source>
<award-id>EXC 2181, 390900948</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5639-7209</contrib-id>
<name name-style="western">
<surname>Kobak</surname>
<given-names>Dmitry</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="award003">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100003493</institution-id>
<institution>Gemeinnützige Hertie-Stiftung</institution>
</institution-wrap>
</funding-source>
<principal-award-recipient>
<name name-style="western">
<surname>Berens</surname>
<given-names>Philipp</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="award004">
<funding-source>
<institution>European Union</institution>
</funding-source>
<award-id>ERC 101039115</award-id>
<principal-award-recipient>
<name name-style="western">
<surname>Berens</surname>
<given-names>Philipp</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="award005">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100001659</institution-id>
<institution>Deutsche Forschungsgemeinschaft</institution>
</institution-wrap>
</funding-source>
<award-id>KO6282/2-1</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5639-7209</contrib-id>
<name name-style="western">
<surname>Kobak</surname>
<given-names>Dmitry</given-names>
</name>
</principal-award-recipient>
</award-group>
<funding-statement>The work was funded by the Gemeinnützige Hertie-Stiftung (PB), the European Union (ERC 101039115 "NextMechMod", PB) and the Deutsche Forschungsgemeinschaft (KO6282/2-1, DK; Excellence cluster "Machine Learning --- New Perspectives for Science", EXC 2064, 390727645, PB and DK; Excellence cluster "STRUCTURES", EXC 2181, 390900948, DK). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Views and opinions expressed are however those of the authors only and do not necessarily reflect those of the European Union or the European Research Council Executive Agency. Neither the European Union nor the granting authority can be held responsible for them.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<page-count count="5"/>
</counts>
</article-meta>
</front>
<body>
<p>In single-cell genomics, researchers often visualize data with 2D embedding methods such as <italic>t</italic>-SNE [<xref ref-type="bibr" rid="pcbi.1012403.ref001">1</xref>,<xref ref-type="bibr" rid="pcbi.1012403.ref002">2</xref>] and UMAP [<xref ref-type="bibr" rid="pcbi.1012403.ref003">3</xref>,<xref ref-type="bibr" rid="pcbi.1012403.ref004">4</xref>]. Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] criticize this practice: They claim that the resulting 2D embeddings fail to faithfully represent the original high-dimensional space, and that instead of meaningful structure these embeddings exhibit “arbitrary” and “specious” shapes. While we agree that 2D embeddings necessarily distort high-dimensional distances between data points [<xref ref-type="bibr" rid="pcbi.1012403.ref006">6</xref>,<xref ref-type="bibr" rid="pcbi.1012403.ref007">7</xref>], we believe that UMAP and <italic>t</italic>-SNE embeddings can nevertheless provide useful information. Here, we demonstrate that UMAP and <italic>t</italic>-SNE preserve cell neighborhoods and cell types, and that the conclusions of Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] are based on inadequate metrics of embedding quality.</p>
<p>To illustrate their point that <italic>t</italic>-SNE and UMAP embeddings are arbitrary, Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] designed Picasso, an autoencoder method that transforms data into an arbitrary predefined 2D shape, e.g., that of an elephant. The authors then compared four kinds of embeddings: the purposefully arbitrary elephant embedding, 2D PCA, <italic>t</italic>-SNE, and UMAP (<xref ref-type="fig" rid="pcbi.1012403.g001">Fig 1</xref>). For this, they used two metrics of embedding quality, both requiring class annotations: <italic>inter-class correlation</italic> measuring how well high-dimensional distances between class centroids are preserved in the 2D embedding and <italic>intra-class correlation</italic> measuring how well class variances are preserved. They found that across three scRNA-seq datasets, 2D PCA performed the best on those metrics, while the elephant embedding scored similar to or better than UMAP and <italic>t</italic>-SNE. We reproduced and confirmed these results (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2A–2B</xref>).</p>
<fig id="pcbi.1012403.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012403.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Evaluated embeddings.</title>
<p>Large panels: <monospace specific-use="no-wrap">Ex Utero</monospace> dataset. Small panels: <monospace specific-use="no-wrap">MERFISH</monospace> and <monospace specific-use="no-wrap">Smart-seq</monospace> datasets. Colors correspond to cell types and are taken from Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>]. See <xref ref-type="supplementary-material" rid="pcbi.1012403.s001">S1 Text</xref> for details.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012403.g001" xlink:type="simple"/>
</fig>
<fig id="pcbi.1012403.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pcbi.1012403.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Embedding quality metrics.</title>
<p>Panels correspond to metrics, colors correspond to embedding methods, marker shapes correspond to datasets. Averages over five runs, error bars go from the minimum to the maximum across runs. Dotted horizontal lines show the values of the metrics in the high-dimensional gene space. <bold>a–b:</bold> The two metrics from Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>], reproducing the results from their Fig 7<bold>C–</bold>7<bold>D:</bold> <italic>k</italic>NN accuracy and <italic>k</italic>NN recall (<italic>k</italic> = 10). <bold>e:</bold> Silhouette coefficient. <bold>f:</bold> Maximum adjusted mutual information between classes and 2D clusters obtained with HDBSCAN using a range of hyperparameter values.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012403.g002" xlink:type="simple"/>
</fig>
<p>According to the authors, this means that <italic>t</italic>-SNE and UMAP are as arbitrary and as misleading as the Picasso elephant. Most online discussions and debates about their paper, including posts by the authors themselves, have prominently featured this argument and the powerful elephant metaphor to argue that “it’s time to stop making <italic>t</italic>-SNE &amp; UMAP plots” [<xref ref-type="bibr" rid="pcbi.1012403.ref008">8</xref>]. In this Comment, we focus exclusively on this argument and do not discuss the rest of the Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] paper.</p>
<p>We believe that this argument is faulty because the metrics used by Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] are insufficient and only quantify a single aspect: both metrics focus on preservation of <italic>distances</italic>, where 2D PCA was unsurprisingly the best. But there is more to embeddings than distance preservation. It is visually apparent in the resulting embeddings that <italic>t</italic>-SNE and UMAP separate cell types, while 2D PCA and Picasso elephant lead to strongly overlapping types (<xref ref-type="fig" rid="pcbi.1012403.g001">Fig 1</xref>), but neither of the two metrics quantified that. Biologists are often interested in cell clusters, and so preservation of cell neighborhoods and visual separation of meaningful cell groups are important properties of 2D embeddings.</p>
<p>To quantify these aspects neglected by Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>], we used four additional metrics, commonly employed in benchmark studies [<xref ref-type="bibr" rid="pcbi.1012403.ref009">9</xref>–<xref ref-type="bibr" rid="pcbi.1012403.ref011">11</xref>]: <italic>k</italic>-nearest-neighbor (<italic>k</italic>NN) accuracy, <italic>k</italic>NN recall [<xref ref-type="bibr" rid="pcbi.1012403.ref012">12</xref>], the silhouette coefficient [<xref ref-type="bibr" rid="pcbi.1012403.ref013">13</xref>], and the adjusted mutual information (AMI) between clusters and class labels [<xref ref-type="bibr" rid="pcbi.1012403.ref014">14</xref>].</p>
<p>The <italic>k</italic>NN accuracy quantifies how often the 2D neighbors are from the same class, while the <italic>k</italic>NN recall quantifies how often the 2D neighbors are the same as the high-dimensional neighbors. In both metrics, UMAP and <italic>t</italic>-SNE consistently and strongly outperformed PCA and Picasso elephant embeddings (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2C–2D</xref>, &gt;90% vs. &lt;62% accuracy, &gt;15% vs. &lt;5% recall for all datasets). Even though the <italic>k</italic>NN recall was below 40% for all methods (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2D</xref>), <italic>k</italic>NN accuracy was always above 90% for both UMAP and <italic>t</italic>-SNE (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2C</xref>). This means that even though UMAP and <italic>t</italic>-SNE are not able to preserve high-dimensional nearest neighbors exactly, the low-dimensional neighbors tend to be from a close vicinity in the high-dimensional space, have the same cell type, and hence allow reliable <italic>k</italic>NN classification. In contrast, 2D PCA and the Picasso elephant fail at that.</p>
<p>The silhouette coefficient and the AMI both evaluate to what extent cell types appear as isolated islands in 2D. Specifically, the silhouette coefficient measures how compact and separated the given classes are in 2D, while the AMI evaluates how well clustering in 2D recovers the classes. In both metrics, <italic>t</italic>-SNE and UMAP strongly outperformed 2D PCA and Picasso elephant embeddings (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2E–2F</xref>, &gt;0.3 difference in silhouette score, &gt;0.25 difference in AMI), in agreement with the visual impression (<xref ref-type="fig" rid="pcbi.1012403.g001">Fig 1</xref>).</p>
<p>The <italic>k</italic>NN accuracy and the silhouette coefficient can also be computed directly in the high-dimensional gene space. We found that <italic>t</italic>-SNE and UMAP showed similar or higher <italic>k</italic>NN accuracy and much higher silhouette coefficient than the original high-dimensional space (<xref ref-type="fig" rid="pcbi.1012403.g002">Fig 2C and 2E</xref>). This suggests that high-dimensional distances suffer from the curse of dimensionality, and that it may in fact be undesirable to preserve them in 2D visualisations. Indeed, single-cell biologists rarely use multidimensional scaling (MDS), an embedding method explicitly designed to preserve distances, because MDS often fails to represent the cluster structure in the data. This further underscores why using only distance-preservation metrics, as Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] did, is misguided.</p>
<p>All presented metrics except <italic>k</italic>NN recall rely on class labels, and our analysis, following Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>], used labels derived in original publications via clustering. Therefore, these labels do not necessarily correspond to biological ground truth, and could potentially lead to biased comparisons. To address this concern, we used negative binomial sampling based on the Ex Utero dataset to simulate a dataset with known ground truth classes. Analyzing this simulated dataset gave the same conclusions: 2D PCA scored the best in the distance-based correlation metrics of Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>], but only <italic>t</italic>-SNE and UMAP could separate the true classes, while Picasso and 2D PCA failed at that (Fig A in <xref ref-type="supplementary-material" rid="pcbi.1012403.s001">S1 Text</xref>).</p>
<p>Taken together, our results point to the elephant in the room: Even though they are not designed to preserve pairwise distances, <italic>t</italic>-SNE and UMAP embeddings are not arbitrary and do preserve meaningful structure of single-cell data, especially local neighborhoods and cluster structure. Claiming that Picasso and <italic>t</italic>-SNE/UMAP are “quantitatively similar in terms of fidelity to the data in ambient dimension” [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] is wrong. They are not.</p>
<p>That said, we do agree with Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>] that 2D visualisations distort distances and should not be blindly trusted. Moreover, as Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>], we do not recommend to use 2D embeddings for quantitative downstream analysis. However, paraphrasing George Box [<xref ref-type="bibr" rid="pcbi.1012403.ref015">15</xref>], we can say that <italic>all 2D embeddings of high-dimensional data are wrong</italic>, <italic>but some are useful</italic>. Indeed, one can use 2D embeddings to form hypotheses about the data structure, ranging from data quality control and sanity-checking of any algorithmic output, to more general hypotheses about cluster separability, relationships between adjacent clusters, or presence of outlying clusters. Of course, any generated insight should then be validated in the high-dimensional data by other means. Here, our conclusion differs strongly from that of Chari and Pachter [<xref ref-type="bibr" rid="pcbi.1012403.ref005">5</xref>]: while they claim that UMAP and <italic>t</italic>-SNE are “counter-productive for exploratory […] analyses”, we endorse them for that very purpose.</p>
<sec id="sec001" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pcbi.1012403.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1012403.s001" xlink:type="simple">
<label>S1 Text</label>
<caption>
<title>Supplementary Methods and Supplementary Figures.</title>
<p>(PDF)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ack>
<p>We thank Erik van Nimwegen, Sebastian Damrich, and Pavlin Poličar for discussions.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pcbi.1012403.ref001"><label>1</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Van der Maaten</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Hinton</surname> <given-names>G</given-names></name>. <article-title>Visualizing data using t-SNE</article-title>. <source>Journal of Machine Learning Research</source>. <year>2008</year>;<volume>9</volume>(<issue>11</issue>).</mixed-citation></ref>
<ref id="pcbi.1012403.ref002"><label>2</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kobak</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Berens</surname> <given-names>P</given-names></name>. <article-title>The art of using t-SNE for single-cell transcriptomics.</article-title> <source>Nat Commun.</source> <year>2019</year>;<volume>10</volume>(<issue>1</issue>):<fpage>5416</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41467-019-13056-x" xlink:type="simple">10.1038/s41467-019-13056-x</ext-link></comment> <object-id pub-id-type="pmid">31780648</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref003"><label>3</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>McInnes</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Healy</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Melville</surname> <given-names>J</given-names></name>. <article-title>UMAP: Uniform manifold approximation and projection for dimension reduction.</article-title> <source>arXiv:180203426.</source> <year>2018</year>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Becht</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>McInnes</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Healy</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Dutertre</surname> <given-names>C-A</given-names></name>, <name name-style="western"><surname>Kwok</surname> <given-names>IWH</given-names></name>, <name name-style="western"><surname>Ng</surname> <given-names>LG</given-names></name>, <etal>et al</etal>. <article-title>Dimensionality reduction for visualizing single-cell data using UMAP</article-title>. <source>Nat Biotechnol</source>. <year>2019</year>;<volume>37</volume>(<issue>1</issue>):<fpage>38</fpage>–<lpage>44</lpage>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Chari</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Pachter</surname> <given-names>L</given-names></name>. <article-title>The specious art of single-cell genomics</article-title>. <source>PLoS Comput Biol</source>. <year>2023</year>;<volume>19</volume>(<issue>8</issue>):<fpage>e1011288</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1011288" xlink:type="simple">10.1371/journal.pcbi.1011288</ext-link></comment> <object-id pub-id-type="pmid">37590228</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nonato</surname> <given-names>LG</given-names></name>, <name name-style="western"><surname>Aupetit</surname> <given-names>M</given-names></name>. <article-title>Multidimensional projection for visual analytics: Linking techniques with distortions, tasks, and layout enrichment</article-title>. <source>IEEE Trans Vis Comput Graph</source>. <year>2018</year>;<volume>25</volume>(<issue>8</issue>):<fpage>2650</fpage>–<lpage>2673</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/TVCG.2018.2846735" xlink:type="simple">10.1109/TVCG.2018.2846735</ext-link></comment> <object-id pub-id-type="pmid">29994258</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref007"><label>7</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wang</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Sontag</surname> <given-names>ED</given-names></name>, <name name-style="western"><surname>Lauffenburger</surname> <given-names>DA</given-names></name>. <article-title>What cannot be seen correctly in 2D visualizations of single-cell ‘omics data</article-title>? <source>Cell Systems</source>. <year>2023</year>;<volume>14</volume>(<issue>9</issue>):<fpage>723</fpage>–<lpage>731</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.cels.2023.07.002" xlink:type="simple">10.1016/j.cels.2023.07.002</ext-link></comment> <object-id pub-id-type="pmid">37734322</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref008"><label>8</label><mixed-citation publication-type="other" xlink:type="simple">Pachter L, 2021. URL <ext-link ext-link-type="uri" xlink:href="https://web.archive.org/web/20240729115631/https://archive.is/2024.07.29-115414/https://x.com/lpachter/status/1431325969411821572" xlink:type="simple">https://web.archive.org/web/20240729115631/https://archive.is/2024.07.29-115414/https://x.com/lpachter/status/1431325969411821572</ext-link>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref009"><label>9</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Espadoto</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Martins</surname> <given-names>RM</given-names></name>, <name name-style="western"><surname>Kerren</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Hirata</surname> <given-names>NST</given-names></name>, <name name-style="western"><surname>Telea</surname> <given-names>AC</given-names></name>. <article-title>Toward a quantitative survey of dimension reduction techniques</article-title>. <source>IEEE Trans Vis Comput Graph</source>. <year>2021</year>;<volume>27</volume>(<issue>3</issue>):<fpage>2153</fpage>–<lpage>2173</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/TVCG.2019.2944182" xlink:type="simple">10.1109/TVCG.2019.2944182</ext-link></comment> <object-id pub-id-type="pmid">31567092</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref010"><label>10</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Huang</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Rudin</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Browne</surname> <given-names>EP</given-names></name>. <article-title>Towards a comprehensive evaluation of dimension reduction methods for transcriptomic data visualization</article-title>. <source>Communications Biology</source>. <year>2022</year>;<volume>5</volume>(<issue>1</issue>):<fpage>719</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s42003-022-03628-x" xlink:type="simple">10.1038/s42003-022-03628-x</ext-link></comment> <object-id pub-id-type="pmid">35853932</object-id></mixed-citation></ref>
<ref id="pcbi.1012403.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wang</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Yang</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Fangjiang</surname> <given-names>W</given-names></name>, <name name-style="western"><surname>Song</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>X</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>T</given-names></name>. <article-title>Comparative analysis of dimension reduction methods for cytometry by time-of-flight data.</article-title> <source>Nat Commun</source>. <year>1836</year>;<volume>14</volume>(<issue>1</issue>):<fpage>2023b</fpage>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Lee</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Verleysen</surname> <given-names>M</given-names></name>. <article-title>Quality assessment of dimensionality reduction: Rank-based criteria.</article-title> <source>Neurocomputing.</source> <year>2009</year>;<volume>72</volume>(<issue>7–9</issue>):<fpage>1431</fpage>–<lpage>1443</lpage>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref013"><label>13</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Rousseeuw</surname> <given-names>PJ</given-names></name>. <article-title>Silhouettes: a graphical aid to the interpretation and validation of cluster analysis</article-title>. <source>Journal of Computational and Applied Mathematics</source>. <year>1987</year>;<volume>20</volume>:<fpage>53</fpage>–<lpage>65</lpage>.</mixed-citation></ref>
<ref id="pcbi.1012403.ref014"><label>14</label><mixed-citation publication-type="other" xlink:type="simple">Vinh NX, Epps J, Bailey J. Information theoretic measures for clusterings comparison: is a correction for chance necessary? In <italic>Proceedings of the 26th Annual International Conference on Machine Learning</italic>, pages 1073–1080, 2009.</mixed-citation></ref>
<ref id="pcbi.1012403.ref015"><label>15</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Box</surname> <given-names>GEP</given-names></name>. <article-title>Robustness in the strategy of scientific model building</article-title>. In <source><italic>Robustness in statistics</italic></source>, pages <fpage>201</fpage>–<lpage>236</lpage>. Elsevier, <year>1979</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>