<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1d3 20150301//EN" "http://jats.nlm.nih.gov/publishing/1.1d3/JATS-journalpublishing1.dtd">
<article article-type="research-article" dtd-version="1.1d3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS ONE</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosone</journal-id>
<journal-title-group>
<journal-title>PLOS ONE</journal-title>
</journal-title-group>
<issn pub-type="epub">1932-6203</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">PONE-D-22-20332</article-id>
<article-id pub-id-type="doi">10.1371/journal.pone.0275472</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Research Article</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Biochemistry</subject><subj-group><subject>Nucleic acids</subject><subj-group><subject>RNA</subject><subj-group><subject>Non-coding RNA</subject><subj-group><subject>Natural antisense transcripts</subject><subj-group><subject>MicroRNAs</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Gene expression</subject><subj-group><subject>Gene regulation</subject><subj-group><subject>MicroRNAs</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Gene expression</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Mathematical and statistical techniques</subject><subj-group><subject>Statistical methods</subject><subj-group><subject>Multivariate analysis</subject><subj-group><subject>Principal component analysis</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Statistics</subject><subj-group><subject>Statistical methods</subject><subj-group><subject>Multivariate analysis</subject><subj-group><subject>Principal component analysis</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Physical sciences</subject><subj-group><subject>Mathematics</subject><subj-group><subject>Algebra</subject><subj-group><subject>Linear algebra</subject><subj-group><subject>Singular value decomposition</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Diagnostic medicine</subject><subj-group><subject>Virus testing</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Nephrology</subject><subj-group><subject>Renal cancer</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Organisms</subject><subj-group><subject>Viruses</subject><subj-group><subject>RNA viruses</subject><subj-group><subject>Coronaviruses</subject><subj-group><subject>SARS coronavirus</subject><subj-group><subject>SARS CoV 2</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Microbiology</subject><subj-group><subject>Medical microbiology</subject><subj-group><subject>Microbial pathogens</subject><subj-group><subject>Viral pathogens</subject><subj-group><subject>Coronaviruses</subject><subj-group><subject>SARS coronavirus</subject><subj-group><subject>SARS CoV 2</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Pathology and laboratory medicine</subject><subj-group><subject>Pathogens</subject><subj-group><subject>Microbial pathogens</subject><subj-group><subject>Viral pathogens</subject><subj-group><subject>Coronaviruses</subject><subj-group><subject>SARS coronavirus</subject><subj-group><subject>SARS CoV 2</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Organisms</subject><subj-group><subject>Viruses</subject><subj-group><subject>Viral pathogens</subject><subj-group><subject>Coronaviruses</subject><subj-group><subject>SARS coronavirus</subject><subj-group><subject>SARS CoV 2</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>Projection in genomic analysis: A theoretical basis to rationalize tensor decomposition and principal component analysis as feature selection tools</article-title>
<alt-title alt-title-type="running-head">Projection in genomic analysis</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0867-8986</contrib-id>
<name name-style="western">
<surname>Taguchi</surname> <given-names>Y-h.</given-names></name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Turki</surname> <given-names>Turki</given-names></name>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001">
<label>1</label>
<addr-line>Department of Physics, Chuo University, Bunkyo-ku, Tokyo, Japan</addr-line>
</aff>
<aff id="aff002">
<label>2</label>
<addr-line>Department of Computer Science, King Abdulaziz University, Jeddah, Saudi Arabia</addr-line>
</aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Chen</surname> <given-names>Chi-Hua</given-names></name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/>
</contrib>
</contrib-group>
<aff id="edit1">
<addr-line>Fuzhou University, CHINA</addr-line>
</aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">tag@granular.com</email></corresp>
</author-notes>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<pub-date pub-type="epub">
<day>29</day>
<month>9</month>
<year>2022</year>
</pub-date>
<volume>17</volume>
<issue>9</issue>
<elocation-id>e0275472</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>7</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>9</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-year>2022</copyright-year>
<copyright-holder>Taguchi, Turki</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pone.0275472"/>
<abstract>
<p>Identifying differentially expressed genes is difficult because of the small number of available samples compared with the large number of genes. Conventional gene selection methods employing statistical tests have the critical problem of heavy dependence of <italic>P</italic>-values on sample size. Although the recently proposed principal component analysis (PCA) and tensor decomposition (TD)-based unsupervised feature extraction (FE) has often outperformed these statistical test-based methods, the reason why they worked so well is unclear. In this study, we aim to understand this reason in the context of projection pursuit (PP) that was proposed a long time ago to solve the problem of dimensions; we can relate the space spanned by singular value vectors with that spanned by the optimal cluster centroids obtained from K-means. Thus, the success of PCA- and TD-based unsupervised FE can be understood by this equivalence. In addition to this, empirical threshold adjusted <italic>P</italic>-values of 0.01 assuming the null hypothesis that singular value vectors attributed to genes obey the Gaussian distribution empirically corresponds to threshold-adjusted <italic>P</italic>-values of 0.1 when the null distribution is generated by gene order shuffling. For this purpose, we newly applied PP to the three data sets to which PCA and TD based unsupervised FE were previously applied; these data sets treated two topics, biomarker identification for kidney cancers (the first two) and the drug discovery for COVID-19 (the thrid one). Then we found the coincidence between PP and PCA or TD based unsupervised FE is pretty well. Shuffling procedures described above are also successfully applied to these three data sets. These findings thus rationalize the success of PCA- and TD-based unsupervised FE for the first time.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100001691</institution-id>
<institution>Japan Society for the Promotion of Science</institution>
</institution-wrap>
</funding-source>
<award-id>KAKENHI [grant numbers 19H05270, 20H04848, and 20K12067]</award-id>
<principal-award-recipient>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0867-8986</contrib-id>
<name name-style="western">
<surname>Taguchi</surname> <given-names>Y-h.</given-names></name>
</principal-award-recipient>
</award-group>
<funding-statement>Japan Society for the Promotion of Science <ext-link ext-link-type="uri" xlink:href="http://dx.doi.org/10.13039/501100001691" xlink:type="simple">http://dx.doi.org/10.13039/501100001691</ext-link> KAKENHI [grant numbers 19H05270, 20H04848, and 20K12067] Professor Y-h. Taguchi The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="8"/>
<table-count count="9"/>
<page-count count="20"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All the data sets and source code are available in GitHub repositry <ext-link ext-link-type="uri" xlink:href="https://github.com/tagtag/peoj" xlink:type="simple">https://github.com/tagtag/peoj</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>In genomic sciences, selecting a limited number of differentially expressed genes (DEGs) among as many as several tens of thousands of genes is a critical problem. Unfortunately, this is a very difficult task as the number of genes, <italic>N</italic>, is usually much larger than the number of available samples, <italic>M</italic>. However, as this is not a mathematically solved problem, it has most frequently been tackled empirically using statistical test-based feature selection strategies [<xref ref-type="bibr" rid="pone.0275472.ref001">1</xref>, <xref ref-type="bibr" rid="pone.0275472.ref002">2</xref>]. Despite huge efforts along this direction, these statistical test-based feature selection strategies cannot be said to work well.</p>
<p>Selection of biologically informative genes including DEGs is essentially performed as follows (For simplicity, <inline-formula id="pone.0275472.e001"><alternatives><graphic id="pone.0275472.e001g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e001" xlink:type="simple"/><mml:math display="inline" id="M1"><mml:mrow><mml:msub><mml:mo>∑</mml:mo> <mml:mi>i</mml:mi></mml:msub> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:mn>0</mml:mn> <mml:mo>,</mml:mo> <mml:msub><mml:mo>∑</mml:mo> <mml:mi>i</mml:mi></mml:msub> <mml:msubsup><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mn>2</mml:mn></mml:msubsup> <mml:mo>=</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> and <italic>M</italic> samples are composed of multiple classes having an equal number of samples). Suppose that we have properties <inline-formula id="pone.0275472.e002"><alternatives><graphic id="pone.0275472.e002g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e002" xlink:type="simple"/><mml:math display="inline" id="M2"><mml:mrow><mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>M</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> attributed to <italic>M</italic> samples. We would like to relate a matrix form of some omics data, e.g., gene expression profiles, <inline-formula id="pone.0275472.e003"><alternatives><graphic id="pone.0275472.e003g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e003" xlink:type="simple"/><mml:math display="inline" id="M3"><mml:mrow><mml:mi>X</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> to <bold><italic>y</italic></bold>. The overall purpose is to derive <inline-formula id="pone.0275472.e004"><alternatives><graphic id="pone.0275472.e004g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e004" xlink:type="simple"/><mml:math display="inline" id="M4"><mml:mrow><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> whose absolute values represent the importance of the <italic>i</italic>th gene. The first and the most popular strategy outside genomic sciences is a regression strategy that requires minimization of
<disp-formula id="pone.0275472.e005"><alternatives><graphic id="pone.0275472.e005g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e005" xlink:type="simple"/><mml:math display="block" id="M5"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>-</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mi>X</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(1)</label></disp-formula>
resulting in
<disp-formula id="pone.0275472.e006"><alternatives><graphic id="pone.0275472.e006g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e006" xlink:type="simple"/><mml:math display="block" id="M6"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>=</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mo>)</mml:mo></mml:mrow> <mml:mrow><mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msup> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(2)</label></disp-formula></p>
<p>The regression approach, <xref ref-type="disp-formula" rid="pone.0275472.e006">Eq (2)</xref>, is less popular in genomic sciences than in other scientific fields, possibly because of <italic>N</italic> ≫ <italic>M</italic>, which always results in exactly (<bold><italic>y</italic></bold> − <bold><italic>b</italic></bold><italic>X</italic>)<sup>2</sup> = 0 with an infinitely large number of <bold><italic>b</italic></bold>. Thus, it is useless to select a limited number of important features among the total <italic>N</italic> features. Although adding the regulation term of <italic>L</italic><sub>2</sub> norm to <xref ref-type="disp-formula" rid="pone.0275472.e005">Eq (1)</xref> as
<disp-formula id="pone.0275472.e007"><alternatives><graphic id="pone.0275472.e007g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e007" xlink:type="simple"/><mml:math display="block" id="M7"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>-</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mi>X</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>+</mml:mo> <mml:mo>λ</mml:mo> <mml:msup><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(3)</label></disp-formula>
with the positive constant λ &gt; 0 enables selection of a unique <bold><italic>b</italic></bold> by minimizing <xref ref-type="disp-formula" rid="pone.0275472.e007">Eq (3)</xref> as
<disp-formula id="pone.0275472.e008"><alternatives><graphic id="pone.0275472.e008g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e008" xlink:type="simple"/><mml:math display="block" id="M8"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>=</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mo>+</mml:mo> <mml:mo>λ</mml:mo> <mml:mi>I</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mrow><mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msup> <mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(4)</label></disp-formula>
because it does not satisfy (<bold><italic>y</italic></bold> − <bold><italic>b</italic></bold><italic>X</italic>)<sup>2</sup> = 0 anymore, it is not an ideal solution. Although the solution using the Moore-Penrose Pseudoinverse [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>]
<disp-formula id="pone.0275472.e009"><alternatives><graphic id="pone.0275472.e009g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e009" xlink:type="simple"/><mml:math display="block" id="M9"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>=</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mo>†</mml:mo></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(5)</label></disp-formula>
might be better as it satisfies (<bold><italic>y</italic></bold> − <bold><italic>b</italic></bold><italic>X</italic>)<sup>2</sup> = 0 under the condition of <inline-formula id="pone.0275472.e010"><alternatives><graphic id="pone.0275472.e010g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e010" xlink:type="simple"/><mml:math display="inline" id="M10"><mml:mrow><mml:munder><mml:mtext>min</mml:mtext> <mml:mi mathvariant="bold-italic">b</mml:mi></mml:munder> <mml:mspace width="2pt"/><mml:msup><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, it is unclear whether <inline-formula id="pone.0275472.e011"><alternatives><graphic id="pone.0275472.e011g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e011" xlink:type="simple"/><mml:math display="inline" id="M11"><mml:mrow><mml:munder><mml:mtext>min</mml:mtext> <mml:mi mathvariant="bold-italic">b</mml:mi></mml:munder> <mml:mspace width="2pt"/><mml:msup><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> is a good constraint from the biological viewpoint. Adding the regulation term of <italic>L</italic><sub>1</sub> norm [<xref ref-type="bibr" rid="pone.0275472.ref004">4</xref>] to <xref ref-type="disp-formula" rid="pone.0275472.e005">Eq (1)</xref> <disp-formula id="pone.0275472.e012"><alternatives><graphic id="pone.0275472.e012g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e012" xlink:type="simple"/><mml:math display="block" id="M12"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>-</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mi>X</mml:mi> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>+</mml:mo> <mml:mo>λ</mml:mo> <mml:mrow><mml:mo>|</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(6)</label></disp-formula>
can yield at most <italic>M</italic> variables, which is not effective when <italic>N</italic> ≫ <italic>M</italic>, because variables larger than <italic>M</italic> might be biologically informative and should not be neglected. Moreover, addition of <italic>L</italic><sub>1</sub> norm is known to be a poor strategy when <italic>X</italic> is not composed of independent vectors, which are very common in genomic science.</p>
<p>The second strategy is a projection strategy
<disp-formula id="pone.0275472.e013"><alternatives><graphic id="pone.0275472.e013g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e013" xlink:type="simple"/><mml:math display="block" id="M13"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mo>=</mml:mo> <mml:mi mathvariant="bold-italic">y</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(7)</label></disp-formula>
that is equivalent to the maximization of
<disp-formula id="pone.0275472.e014"><alternatives><graphic id="pone.0275472.e014g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e014" xlink:type="simple"/><mml:math display="block" id="M14"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>·</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mi>X</mml:mi> <mml:mo>-</mml:mo> <mml:mfrac><mml:mn>1</mml:mn> <mml:mn>2</mml:mn></mml:mfrac> <mml:msup><mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(8)</label></disp-formula>
and is employed in PCA- and TD-based unsupervised FE (see below). Through the concept of projection pursuit [<xref ref-type="bibr" rid="pone.0275472.ref005">5</xref>] (PP), it is understood that seeking the projection vector <bold><italic>b</italic></bold> maximizes interestingness
<disp-formula id="pone.0275472.e015"><alternatives><graphic id="pone.0275472.e015g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e015" xlink:type="simple"/><mml:math display="block" id="M15"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi>H</mml:mi> <mml:mo>(</mml:mo> <mml:mi mathvariant="bold-italic">b</mml:mi> <mml:mi>X</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(9)</label></disp-formula>
which is <xref ref-type="disp-formula" rid="pone.0275472.e014">Eq (8)</xref> in this study. As <italic>H</italic>(<bold><italic>b</italic></bold><italic>X</italic>) is a function of <bold><italic>b</italic></bold>, it is also denoted as <italic>I</italic>(<bold><italic>b</italic></bold>), which is called projection index. <italic>I</italic>(<bold><italic>b</italic></bold>) can be any other function, but its selection should be decided such that the biologically most meaningful results are obtained. Upon obtaining <bold><italic>b</italic></bold> that maximizes <italic>I</italic>(<bold><italic>b</italic></bold>), we can select <italic>i</italic> having a larger absolute <italic>b</italic><sub><italic>i</italic></sub> as mentioned above. In the framework of PP, in a high dimensional system, almost all <bold><italic>b</italic></bold> have finite projections [<xref ref-type="bibr" rid="pone.0275472.ref006">6</xref>]. Thus, the only the point is if it is accidental or biologically meaningful.</p>
<p>In genomic science, projection strategy, <xref ref-type="disp-formula" rid="pone.0275472.e013">Eq (7)</xref>, is also unpopular. Although the reason for the unpopularity of the projection strategy, <xref ref-type="disp-formula" rid="pone.0275472.e013">Eq (7)</xref>, is unclear, this may be explained by the ignorance of the contribution perpendicular to <bold><italic>y</italic></bold>, <inline-formula id="pone.0275472.e016"><alternatives><graphic id="pone.0275472.e016g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e016" xlink:type="simple"/><mml:math display="inline" id="M16"><mml:mrow><mml:mrow><mml:mo>|</mml:mo></mml:mrow> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>·</mml:mo> <mml:mover accent="true"><mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mo>)</mml:mo></mml:mrow> <mml:mover accent="true"><mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>^</mml:mo></mml:mover> <mml:mo>-</mml:mo> <mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:math></alternatives></inline-formula>, where <inline-formula id="pone.0275472.e017"><alternatives><graphic id="pone.0275472.e017g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e017" xlink:type="simple"/><mml:math display="inline" id="M17"><mml:mover accent="true"><mml:mi mathvariant="bold-italic">y</mml:mi> <mml:mo>^</mml:mo></mml:mover></mml:math></alternatives></inline-formula> is a unit vector parallel to <bold><italic>y</italic></bold> and is defined as <bold><italic>y</italic></bold>/|<bold><italic>y</italic></bold>|. Nevertheless, in contrast to the regression strategy requiring the computation of (<italic>XX</italic><sup><italic>T</italic></sup>)<sup>−1</sup>, <xref ref-type="disp-formula" rid="pone.0275472.e013">Eq (7)</xref> can be always computable even if <italic>N</italic> ≫ <italic>M</italic>, which is a great advantage of the projection strategy when compared with the regression strategy.</p>
<p>Instead of these two strategies, feature selection based on statistical tests [<xref ref-type="bibr" rid="pone.0275472.ref001">1</xref>, <xref ref-type="bibr" rid="pone.0275472.ref002">2</xref>] is more popular in genomic sciences as mentioned above. They try to identify genes whose expression is significantly distinct between classes. Despite its popularity, feature selection based on statistical tests has critical problems; in particular, significance is heavily dependent on sample size, <italic>M</italic>. Even in the case of a small distinction, more significant results are obtained when more samples are considered; this is not applicable biologically because determination of whether gene expression between classes differs significantly should not be a function of sample size. To compensate this heavy sample dependence of significance, other criteria such as fold change between classes are often employed. Thus, feature selection based on statistical tests is at best, the best among the worst approaches. If better strategies can be employed, there will be no reason to employ strategies based on statistical tests.</p>
<p>Despite the unpopularity of projection strategy, it was sometimes evaluated as more effective [<xref ref-type="bibr" rid="pone.0275472.ref007">7</xref>, <xref ref-type="bibr" rid="pone.0275472.ref008">8</xref>] than the standard feature selection strategy based on statistical tests. Thus, it can be a candidate strategy that can be replaced with feature selection based on statistical tests. In this paper, we try to understand why PCA-based unsupervised FE and TD-based unsupervised FE [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>] are effective in feature selection based on projection strategy, since PCA-like as well as TD-like methods were successfully applied in other fields, too [<xref ref-type="bibr" rid="pone.0275472.ref009">9</xref>–<xref ref-type="bibr" rid="pone.0275472.ref011">11</xref>]. We consider the cases biomarker identification of kidney cancer [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>] as well as SARS-CoV-2 infection problem [<xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>]; in these studies, despite unsuccessful results obtained by conventional feature selection based on statistical tests, TD-based unsupervised FE identified biologically reasonable genes (for more details about how PCA- and TD-based unsupervised FE are superior to statistical test-based feature selection tools in these specific examples, see these previous studies [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>, <xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>]).</p>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Materials and methods</title>
<p>Sample R cods is available in <ext-link ext-link-type="uri" xlink:href="https://github.com/tagtag/peoj" xlink:type="simple">https://github.com/tagtag/peoj</ext-link>.</p>
<sec id="sec003">
<title>Expression profiles</title>
<p>mRNA, miRNA, and gene expression profiles in the first, second, and third data sets can be downloaded from TCGA as well as GEO. Their availability is described in detail in previous studies [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>, <xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>].</p>
</sec>
<sec id="sec004">
<title>Excluding low expressed miRNAs, mRNAs, and genes</title>
<p>To draw Figs <xref ref-type="fig" rid="pone.0275472.g001">1(B)</xref>, <xref ref-type="fig" rid="pone.0275472.g002">2(B)</xref> and <xref ref-type="fig" rid="pone.0275472.g003">3(B)</xref>, low expressed miRNAs, mRNAs, and genes were screened out. For this, we rank them using ∑<sub><italic>j</italic></sub> |<italic>x</italic><sub><italic>ij</italic></sub>|, ∑<sub><italic>j</italic></sub> |<italic>x</italic><sub><italic>ik</italic></sub>|, ∑<sub><italic>jkm</italic></sub> |<italic>x</italic><sub><italic>ijkm</italic></sub>| and only selected the top ranked ones.</p>
<fig id="pone.0275472.g001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g001</object-id>
<label>Fig 1</label>
<caption>
<title>Histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling when miRNAs in the first data set were considered.</title>
<p>(A) All miRNAs (B) Top 500 most expressive miRNAs.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g001" xlink:type="simple"/>
</fig>
<fig id="pone.0275472.g002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g002</object-id>
<label>Fig 2</label>
<caption>
<title>Histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling when the mRNAs in the first data set were considered.</title>
<p>(A) All mRNAs (B) Top 3000 most expressive mRNAs.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g002" xlink:type="simple"/>
</fig>
<fig id="pone.0275472.g003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g003</object-id>
<label>Fig 3</label>
<caption>
<title>Histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling when genes in the third data set were considered.</title>
<p>(A) All genes (B) Top 2780 most expressive genes.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g003" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec005">
<title>QQplot</title>
<p>QQplot [<xref ref-type="bibr" rid="pone.0275472.ref014">14</xref>] was used to visualize the coincidence between two distributions that do not always have same number of elements. The <monospace>qqplot</monospace> function implemented in R [<xref ref-type="bibr" rid="pone.0275472.ref015">15</xref>] was employed to draw QQplots (Figs <xref ref-type="fig" rid="pone.0275472.g004">4</xref> and <xref ref-type="fig" rid="pone.0275472.g005">5</xref>) in this study.</p>
<fig id="pone.0275472.g004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g004</object-id>
<label>Fig 4</label>
<caption>
<title>QQplot between <italic>P</italic>-values computed by TD-based unsupervised FE and projection (A) mRNA in the first data set (B) miRNA in the first data set (C) mRNA in the second data set (D) miRNA in the second data set.</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g004" xlink:type="simple"/>
</fig>
<fig id="pone.0275472.g005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g005</object-id>
<label>Fig 5</label>
<caption>
<title>QQplot of <italic>P</italic>-values between TD-based unsupervised FE and PP (the third data set).</title>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g005" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec006">
<title>Null distribution</title>
<p>The null distributions used for computing <italic>P</italic>-values in Figs <xref ref-type="fig" rid="pone.0275472.g001">1</xref>–<xref ref-type="fig" rid="pone.0275472.g003">3</xref> and <xref ref-type="fig" rid="pone.0275472.g006">6</xref> were generated by gene order shuffling as follows. First, the order of <italic>i</italic> was shuffled within each <italic>x</italic><sub><italic>ij</italic></sub> or within each <italic>x</italic><sub><italic>ijkm</italic></sub> and that of <italic>k</italic> was shuffled within each <italic>x</italic><sub><italic>kj</italic></sub>. Thus, the order of mRNAs, miRNAs, and genes was shuffled such that they differed between samples. Then SVD or TD was applied to <italic>x</italic><sub><italic>ijk</italic></sub> or <italic>x</italic><sub><italic>ijkm</italic></sub> and <italic>u</italic><sub>2<italic>i</italic></sub> and <italic>u</italic><sub>2<italic>k</italic></sub> from SVD and <italic>u</italic><sub>5<italic>i</italic></sub> from TD were generated one hundred times. The null distributions were composed of the generated singular value vectors and <italic>P</italic>-values were computed.</p>
<fig id="pone.0275472.g006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g006</object-id>
<label>Fig 6</label>
<caption>
<title>Histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling when the second data set were considered.</title>
<p>(A) All miRNAs (B) All mRNAs.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g006" xlink:type="simple"/>
</fig>
</sec>
</sec>
<sec id="sec007" sec-type="results">
<title>Results</title>
<p>
<xref ref-type="fig" rid="pone.0275472.g007">Fig 7</xref> shows the work flow of this study. In PP, the projection direction is predefined by <bold><italic>y</italic></bold> in a supervised manner while if we do not want to set projection directions in advance we can use those determined by PCA or TD, which we call unsupervised FE. There are some advantages of PCA and TD, which are not shared with PP. For example, projection directions not related to the label <bold><italic>y</italic></bold> may have additional information. In that case, PCA and TD can capture what PP cannot. PCA and TD can be applicable even if pre-defined <bold><italic>y</italic></bold> is not provided. Thus, PCA and TD have more potential to be applied to wide range of data sets that PP.</p>
<fig id="pone.0275472.g007" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g007</object-id>
<label>Fig 7</label>
<caption>
<title>Discussion of work flow used in this study.</title>
<p>Tensor decomposition (HOSVD) was applied to tenors and using obtained singular value vectors assumed to obey Gaussian distribution, <italic>P</italic>-values are attributed to genes. The genes associated with adjusted <italic>P</italic>-values less than 0.01 are selected. <italic>P</italic>-values are also computed by shuffling and the genes associated with adjusted <italic>P</italic>-values less than 0.1 are well coincident with the genes selected by HOSVD. The correspondence between singular value vectors and K-means applied to unfolded matrices is also discussed.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g007" xlink:type="simple"/>
</fig>
<sec id="sec008">
<title>PCA-based unsupervised FE</title>
<p>Before starting to rationalize PCA- and TD-based unsupervised FE, we briefly summarize how they work. The purpose of PCA- and TD-based unsupervised FE is to select biologically sound features (typically genes) based on the given omics data such as gene expression profiles, in an unsupervised manner. In this subsection, we introduce PCA-based unsupervised FE; TD-based unsupervised FE is an advanced version of PCA-based unsupervised FE and will be introduced in the next subsection.</p>
<p>Suppose that we have gene expression data in a matrix form, <inline-formula id="pone.0275472.e018"><alternatives><graphic id="pone.0275472.e018g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e018" xlink:type="simple"/><mml:math display="inline" id="M18"><mml:mrow><mml:mi>X</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> for <italic>N</italic> genes measured across <italic>M</italic> samples. First, we need to standardize <italic>X</italic> as ∑<sub><italic>i</italic></sub><italic>x</italic><sub><italic>ij</italic></sub> = 0 and <inline-formula id="pone.0275472.e019"><alternatives><graphic id="pone.0275472.e019g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e019" xlink:type="simple"/><mml:math display="inline" id="M19"><mml:mrow><mml:msub><mml:mo>∑</mml:mo> <mml:mi>i</mml:mi></mml:msub> <mml:msubsup><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mn>2</mml:mn></mml:msubsup> <mml:mo>=</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:math></alternatives></inline-formula> as we will attribute principal component (PC) scores to genes whereas PC loading will be attributed to samples. The <italic>ℓ</italic>th PC score attributed to the <italic>i</italic>th gene, <italic>u</italic><sub><italic>ℓ</italic><italic>i</italic></sub>, can be obtained as the <italic>i</italic>th component of the <italic>ℓ</italic>th eigenvector, <inline-formula id="pone.0275472.e020"><alternatives><graphic id="pone.0275472.e020g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e020" xlink:type="simple"/><mml:math display="inline" id="M20"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, of a gram matrix <inline-formula id="pone.0275472.e021"><alternatives><graphic id="pone.0275472.e021g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e021" xlink:type="simple"/><mml:math display="inline" id="M21"><mml:mrow><mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, where <italic>X</italic><sup><italic>T</italic></sup> is a transpose matrix of <italic>X</italic>, as
<disp-formula id="pone.0275472.e022"><alternatives><graphic id="pone.0275472.e022g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e022" xlink:type="simple"/><mml:math display="block" id="M22"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(10)</label></disp-formula>
where λ<sub><italic>ℓ</italic></sub> is the <italic>ℓ</italic>th eigenvalue. Further, the <italic>ℓ</italic>th PC score attributed to the <italic>j</italic>th sample, <italic>v</italic><sub><italic>ℓj</italic></sub>, can be obtained as the <italic>j</italic>th component of the vector <inline-formula id="pone.0275472.e023"><alternatives><graphic id="pone.0275472.e023g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e023" xlink:type="simple"/><mml:math display="inline" id="M23"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>M</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> defined as
<disp-formula id="pone.0275472.e024"><alternatives><graphic id="pone.0275472.e024g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e024" xlink:type="simple"/><mml:math display="block" id="M24"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(11)</label></disp-formula>
Notably, <bold><italic>v</italic></bold><sub><italic>ℓ</italic></sub> is also an eigenvector of the covariance matrix, <inline-formula id="pone.0275472.e025"><alternatives><graphic id="pone.0275472.e025g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e025" xlink:type="simple"/><mml:math display="inline" id="M25"><mml:mrow><mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mi>X</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>M</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> because
<disp-formula id="pone.0275472.e026"><alternatives><graphic id="pone.0275472.e026g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e026" xlink:type="simple"/><mml:math display="block" id="M26"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mi>X</mml:mi> <mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msup><mml:mi>X</mml:mi> <mml:mi>t</mml:mi></mml:msup> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(12)</label></disp-formula></p>
<p>PCA-based unsupervised FE works as follows. First, we need to identify the <bold><italic>v</italic></bold><sub><italic>ℓ</italic></sub> of interest. The <bold><italic>v</italic></bold><sub><italic>ℓ</italic></sub> of interest depends on the problem. It might be the one coincident with the samples cluster, or the one with monotonic dependence on some external parameter such as time. After identifying the <bold><italic>v</italic></bold><sub><italic>ℓ</italic></sub> of interest, we try to attribute <italic>P</italic>-values to genes assuming that the components of the corresponding <bold><italic>u</italic></bold><sub><italic>ℓ</italic></sub> follow a normal distribution
<disp-formula id="pone.0275472.e027"><alternatives><graphic id="pone.0275472.e027g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e027" xlink:type="simple"/><mml:math display="block" id="M27"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mrow><mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(13)</label></disp-formula>
where <italic>P</italic><sub><italic>χ</italic><sup>2</sup></sub>[&gt; <italic>x</italic>] is the cumulative <italic>χ</italic><sup>2</sup> distribution that the argument is larger than <italic>x</italic> and <italic>σ</italic><sub><italic>ℓ</italic></sub> is the standard deviation. Computed <italic>P</italic>-values are adjusted based on the BH criterion [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>] and features associated with adjusted <italic>P</italic>-values less than a specified threshold value can be selected. The reason for the proper working of such a simple procedure is explained later.</p>
<p>Finally, we would like to emphasize the equivalence between singular value decomposition (SVD) and PCA. Suppose we have the SVD of <italic>X</italic> as
<disp-formula id="pone.0275472.e028"><alternatives><graphic id="pone.0275472.e028g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e028" xlink:type="simple"/><mml:math display="block" id="M28"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mtext>min</mml:mtext> <mml:mo>(</mml:mo> <mml:mi>N</mml:mi> <mml:mo>,</mml:mo> <mml:mi>M</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:munderover> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(14)</label></disp-formula>
It is straight forward to show
<disp-formula id="pone.0275472.e029"><alternatives><graphic id="pone.0275472.e029g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e029" xlink:type="simple"/><mml:math display="block" id="M29"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:mrow><mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mrow> <mml:mo>=</mml:mo> <mml:mrow><mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(15)</label></disp-formula> <disp-formula id="pone.0275472.e030"><alternatives><graphic id="pone.0275472.e030g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e030" xlink:type="simple"/><mml:math display="block" id="M30"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:mrow><mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mi>X</mml:mi> <mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(16)</label></disp-formula>
where <bold><italic>u</italic></bold><sub><italic>ℓ</italic></sub> = (<italic>u</italic><sub><italic>ℓ</italic>1</sub>, <italic>u</italic><sub><italic>ℓ</italic>2</sub>, ⋯, <italic>u</italic><sub><italic>ℓ</italic><italic>N</italic></sub>)<sup><italic>T</italic></sup> and <bold><italic>v</italic></bold><sub><italic>ℓ</italic></sub> = (<italic>v</italic><sub><italic>ℓ</italic>1</sub>, <italic>v</italic><sub><italic>ℓ</italic>2</sub>, ⋯, <italic>v</italic><sub><italic>ℓ</italic><italic>M</italic></sub>)<sup><italic>T</italic></sup>. Thus, SVD and PCA are mathematically equivalent problems.</p>
</sec>
<sec id="sec009">
<title>TD-based unsupervised FE</title>
<p>TD-based unsupervised FE works quite similar to PCA-based unsupervised FE. Instead of PCA, we apply TD to <inline-formula id="pone.0275472.e031"><alternatives><graphic id="pone.0275472.e031g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e031" xlink:type="simple"/><mml:math display="inline" id="M31"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi> <mml:mo>×</mml:mo> <mml:mi>K</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, that is, for example, the expression of the <italic>i</italic>th gene measured in the <italic>k</italic>th tissue of the <italic>m</italic>th person (even though we consider a three-mode tensor here, extension to the higher mode tensor is straightforward). To obtain TD, we specify the higher-order singular decomposition [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>] (HOSVD) as
<disp-formula id="pone.0275472.e032"><alternatives><graphic id="pone.0275472.e032g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e032" xlink:type="simple"/><mml:math display="block" id="M32"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub></mml:munder> <mml:mi>G</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(17)</label></disp-formula>
where <inline-formula id="pone.0275472.e033"><alternatives><graphic id="pone.0275472.e033g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e033" xlink:type="simple"/><mml:math display="inline" id="M33"><mml:mrow><mml:mi>G</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>M</mml:mi> <mml:mo>×</mml:mo> <mml:mi>K</mml:mi> <mml:mo>×</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> is a core tensor, and <inline-formula id="pone.0275472.e034"><alternatives><graphic id="pone.0275472.e034g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e034" xlink:type="simple"/><mml:math display="inline" id="M34"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>M</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, <inline-formula id="pone.0275472.e035"><alternatives><graphic id="pone.0275472.e035g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e035" xlink:type="simple"/><mml:math display="inline" id="M35"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>K</mml:mi> <mml:mo>×</mml:mo> <mml:mi>K</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, <inline-formula id="pone.0275472.e036"><alternatives><graphic id="pone.0275472.e036g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e036" xlink:type="simple"/><mml:math display="inline" id="M36"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> are singular value matrices. After identifying the <inline-formula id="pone.0275472.e037"><alternatives><graphic id="pone.0275472.e037g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e037" xlink:type="simple"/><mml:math display="inline" id="M37"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> and <inline-formula id="pone.0275472.e038"><alternatives><graphic id="pone.0275472.e038g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e038" xlink:type="simple"/><mml:math display="inline" id="M38"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> of interest, for instance, the distinction between healthy controls and patients as well as tissue specific expression, we seek <italic>ℓ</italic><sub>3</sub> associated with <italic>G</italic>(<italic>ℓ</italic><sub>1</sub><italic>ℓ</italic><sub>2</sub><italic>ℓ</italic><sub>3</sub>) having the largest absolute value given as <italic>ℓ</italic><sub>1</sub>, <italic>ℓ</italic><sub>2</sub>. Then using the identified <italic>ℓ</italic><sub>3</sub>, we attribute <italic>P</italic>-values to the <italic>i</italic>th feature as in the case of PCA-based unsupervised FE,
<disp-formula id="pone.0275472.e039"><alternatives><graphic id="pone.0275472.e039g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e039" xlink:type="simple"/><mml:math display="block" id="M39"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(18)</label></disp-formula>
where <inline-formula id="pone.0275472.e040"><alternatives><graphic id="pone.0275472.e040g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e040" xlink:type="simple"/><mml:math display="inline" id="M40"><mml:msub><mml:mi>σ</mml:mi> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub></mml:msub></mml:math></alternatives></inline-formula> is the standard deviation. Computed <italic>P</italic>-values are adjusted based on the BH criterion and features associated with adjusted <italic>P</italic>-values less than a specified threshold value can be selected. The reason for the proper working of such a simple procedure is explained later.</p>
</sec>
<sec id="sec010">
<title>Rationalization of PCA- and TD-based unsupervised FE</title>
<p>To explain why PCA- and TD-based unsupervised FE work rather well, we consider two recent works [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>, <xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>], in which the superiority of PCA- and/or TD-based unsupervised FE over conventional statistical methods was shown; in these studies, conventional statistical test-based methods failed to select a reasonable number of genes whereas TD-based unsupervised FE successfully selected a biologically reasonable restricted number of genes.</p>
<p>In the first study [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>], two independent sets of data including the mRNA and miRNA expression of kidney cancer and normal kidney were analyzed in an integrated manner using PCA as well as TD-based unsupervised FE.</p>
<sec id="sec011">
<title>The first data set</title>
<p>The first data set comprised <italic>M</italic> = 324 samples including 253 kidney tumors and 71 normal kidney tissues. The expression of <italic>N</italic> mRNAs and <italic>K</italic> miRNAs was formatted as matrices as <inline-formula id="pone.0275472.e041"><alternatives><graphic id="pone.0275472.e041g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e041" xlink:type="simple"/><mml:math display="inline" id="M41"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> and <inline-formula id="pone.0275472.e042"><alternatives><graphic id="pone.0275472.e042g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e042" xlink:type="simple"/><mml:math display="inline" id="M42"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>k</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>K</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, respectively. The three mode-tensor <inline-formula id="pone.0275472.e043"><alternatives><graphic id="pone.0275472.e043g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e043" xlink:type="simple"/><mml:math display="inline" id="M43"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi> <mml:mo>×</mml:mo> <mml:mi>K</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> was generated as
<disp-formula id="pone.0275472.e044"><alternatives><graphic id="pone.0275472.e044g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e044" xlink:type="simple"/><mml:math display="block" id="M44"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>k</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(19)</label></disp-formula>
As the data were too large to be loaded into the memory available in a standard stand-alone server, it was impossible to obtain TD
<disp-formula id="pone.0275472.e045"><alternatives><graphic id="pone.0275472.e045g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e045" xlink:type="simple"/><mml:math display="block" id="M45"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub></mml:munder> <mml:mi>G</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(20)</label></disp-formula>
Instead, we generated
<disp-formula id="pone.0275472.e046"><alternatives><graphic id="pone.0275472.e046g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e046" xlink:type="simple"/><mml:math display="block" id="M46"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:mi>j</mml:mi></mml:munder> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(21)</label></disp-formula>
and SVD was applied to <italic>x</italic><sub><italic>ik</italic></sub> as
<disp-formula id="pone.0275472.e047"><alternatives><graphic id="pone.0275472.e047g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e047" xlink:type="simple"/><mml:math display="block" id="M47"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mo>=</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub></mml:mrow> <mml:mrow><mml:mtext>min</mml:mtext> <mml:mo>(</mml:mo> <mml:mi>N</mml:mi> <mml:mo>,</mml:mo> <mml:mi>K</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:munderover> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(22)</label></disp-formula>
to obtain <inline-formula id="pone.0275472.e048"><alternatives><graphic id="pone.0275472.e048g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e048" xlink:type="simple"/><mml:math display="inline" id="M48"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> and <inline-formula id="pone.0275472.e049"><alternatives><graphic id="pone.0275472.e049g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e049" xlink:type="simple"/><mml:math display="inline" id="M49"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> approximately. Missing singular value vectors attributed to mRNA and miRNA samples were approximately recovered using the equations
<disp-formula id="pone.0275472.e050"><alternatives><graphic id="pone.0275472.e050g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e050" xlink:type="simple"/><mml:math display="block" id="M50"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>mRNA</mml:mtext></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>i</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>N</mml:mi></mml:munderover> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(23)</label></disp-formula> <disp-formula id="pone.0275472.e051"><alternatives><graphic id="pone.0275472.e051g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e051" xlink:type="simple"/><mml:math display="block" id="M51"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>miRNA</mml:mtext></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>k</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>K</mml:mi></mml:munderover> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>k</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(24)</label></disp-formula>
respectively. Although we do not intend to insist that these approximations are precise enough, we decided to employ them as since they turned out to work well empirically. After investigating the obtained <inline-formula id="pone.0275472.e052"><alternatives><graphic id="pone.0275472.e052g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e052" xlink:type="simple"/><mml:math display="inline" id="M52"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>mRNA</mml:mtext></mml:msubsup></mml:math></alternatives></inline-formula> and <inline-formula id="pone.0275472.e053"><alternatives><graphic id="pone.0275472.e053g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e053" xlink:type="simple"/><mml:math display="inline" id="M53"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>miRNA</mml:mtext></mml:msubsup></mml:math></alternatives></inline-formula>, we realized that <italic>ℓ</italic><sub>1</sub> = <italic>ℓ</italic><sub>3</sub> = 2 are coincident with the distinction between tumors and normal tissues; therefore, we attributed <italic>P</italic>-values to mRNA and miRNA using <italic>u</italic><sub>2<italic>i</italic></sub> and <italic>u</italic><sub>2<italic>k</italic></sub>, respectively with the equations
<disp-formula id="pone.0275472.e054"><alternatives><graphic id="pone.0275472.e054g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e054" xlink:type="simple"/><mml:math display="block" id="M54"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mn>2</mml:mn> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mn>2</mml:mn></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(25)</label></disp-formula> <disp-formula id="pone.0275472.e055"><alternatives><graphic id="pone.0275472.e055g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e055" xlink:type="simple"/><mml:math display="block" id="M55"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>P</mml:mi> <mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mn>2</mml:mn> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:msubsup><mml:mi>σ</mml:mi> <mml:mn>2</mml:mn> <mml:mo>′</mml:mo></mml:msubsup></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(26)</label></disp-formula>
These <italic>P</italic>-values were corrected by the BH criterion and we selected 72 mRNAs and 11 miRNAs associated with adjusted <italic>P</italic>-values less than 0.01, respectively.</p>
</sec>
<sec id="sec012">
<title>The second data set</title>
<p>The second data set comprised <italic>M</italic> = 34 samples including 17 kidney tumors and 17 normal kidney tissues. The same procedures applied to the first data set were also applied to the second data set and we selected 209 mRNAs and 3 miRNAs associated with adjusted <italic>P</italic>-values less than 0.01, respectively. Although various biological evaluations were performed for mRNAs and miRNAs selected using the first data set, the most remarkable achievement was that all three miRNAs selected using the second data set were included in the 11 miRNAs selected using the first data set, and there were as many as 11 common mRNAs selected between the first and second data sets. If we consider that there are as many as several hundred miRNAs and a few tens of thousand mRNAs available, these overlaps are a great achievement as these two data sets are completely independent of each other.</p>
</sec>
<sec id="sec013">
<title>Comparisons with PP</title>
<p>To understand why such simple procedures can work well in the framework of PP, we replaced the singular value vectors attributed to samples with projections. For this, we applied PP as mentioned above.
<disp-formula id="pone.0275472.e056"><alternatives><graphic id="pone.0275472.e056g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e056" xlink:type="simple"/><mml:math display="block" id="M56"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>y</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:mo>{</mml:mo> <mml:mtable><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mo>-</mml:mo> <mml:mfrac><mml:mi>M</mml:mi> <mml:msub><mml:mi>M</mml:mi> <mml:mi>N</mml:mi></mml:msub></mml:mfrac> <mml:mo>,</mml:mo></mml:mrow></mml:mtd> <mml:mtd><mml:mrow><mml:mi>j</mml:mi> <mml:mo>≤</mml:mo> <mml:msub><mml:mi>N</mml:mi> <mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr> <mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mfrac><mml:mi>M</mml:mi> <mml:msub><mml:mi>M</mml:mi> <mml:mi>T</mml:mi></mml:msub></mml:mfrac> <mml:mo>,</mml:mo></mml:mrow></mml:mtd> <mml:mtd><mml:mrow><mml:mi>j</mml:mi> <mml:mo>&gt;</mml:mo> <mml:msub><mml:mi>N</mml:mi> <mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable> <mml:mo/></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(27)</label></disp-formula>
where <italic>M</italic><sub><italic>N</italic></sub>, <italic>M</italic><sub><italic>T</italic></sub> are the numbers of normal tissues and cancer samples, respectively, and <italic>M</italic><sub><italic>N</italic></sub> + <italic>M</italic><sub><italic>T</italic></sub> = <italic>M</italic>. Then we applied PP as
<disp-formula id="pone.0275472.e057"><alternatives><graphic id="pone.0275472.e057g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e057" xlink:type="simple"/><mml:math display="block" id="M57"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>b</mml:mi> <mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>j</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>M</mml:mi></mml:munderover> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>y</mml:mi> <mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(28)</label></disp-formula> <disp-formula id="pone.0275472.e058"><alternatives><graphic id="pone.0275472.e058g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e058" xlink:type="simple"/><mml:math display="block" id="M58"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>b</mml:mi> <mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>j</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>M</mml:mi></mml:munderover> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>k</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>y</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(29)</label></disp-formula>
Since <italic>b</italic><sub><italic>i</italic></sub>s and <italic>b</italic><sub><italic>k</italic></sub>s are expected to play the roles of <italic>u</italic><sub>2<italic>i</italic></sub> and <italic>u</italic><sub>2<italic>k</italic></sub> in Eqs <xref ref-type="disp-formula" rid="pone.0275472.e054">(25)</xref> and <xref ref-type="disp-formula" rid="pone.0275472.e055">(26)</xref>, respectively, we used the absolute values of <italic>b</italic><sub><italic>i</italic></sub> and <italic>b</italic><sub><italic>k</italic></sub> to select mRNAs and miRNAs that are presumably coincident with the distinction between tumors and normal tissues. <italic>P</italic>-values are attributed to mRNA and miRNA as
<disp-formula id="pone.0275472.e059"><alternatives><graphic id="pone.0275472.e059g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e059" xlink:type="simple"/><mml:math display="block" id="M59"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>b</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mi>b</mml:mi></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(30)</label></disp-formula> <disp-formula id="pone.0275472.e060"><alternatives><graphic id="pone.0275472.e060g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e060" xlink:type="simple"/><mml:math display="block" id="M60"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>P</mml:mi> <mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>b</mml:mi> <mml:mi>k</mml:mi></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mi>b</mml:mi></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(31)</label></disp-formula>
These <italic>P</italic>-values are corrected by the BH criterion and we selected 78 mRNAs and 13 miRNAs for the first data set and 194 mRNAs and 3 miRNAs for the second data set, associated with adjusted <italic>P</italic>-values less than 0.01, respectively.</p>
<p>We try to estimate the coincidence of genes between TD and PP; Tables <xref ref-type="table" rid="pone.0275472.t001">1</xref>–<xref ref-type="table" rid="pone.0275472.t004">4</xref> list the comparisons of genes between TD-based unsupervised FE and PP, Eqs <xref ref-type="disp-formula" rid="pone.0275472.e059">(30)</xref> or <xref ref-type="disp-formula" rid="pone.0275472.e060">(31)</xref> and demonstrate a high coincidence with each other. <xref ref-type="fig" rid="pone.0275472.g004">Fig 4</xref> show the comparisons of <italic>P</italic><sub><italic>i</italic></sub> and <italic>P</italic><sub><italic>k</italic></sub> between TD-based unsupervised FE and PP, Eqs <xref ref-type="disp-formula" rid="pone.0275472.e059">(30)</xref> or <xref ref-type="disp-formula" rid="pone.0275472.e060">(31)</xref>. It is obvious that smaller <italic>P</italic>-values used for gene selection as well as the overall distributions of <italic>P</italic>-values are coincident between TD-based unsupervised FE and PP, Eqs <xref ref-type="disp-formula" rid="pone.0275472.e059">(30)</xref> or <xref ref-type="disp-formula" rid="pone.0275472.e060">(31)</xref>.</p>
<table-wrap id="pone.0275472.t001" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t001</object-id>
<label>Table 1</label>
<caption>
<title>Confusion matrix of selected mRNAs between TD-based unsupervised FE and PP in the first data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 1.90 × 10<sup>−149</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t001g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t001" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">PP</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">19447</td>
<td align="center">17</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">11</td>
<td align="center">61</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<table-wrap id="pone.0275472.t002" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t002</object-id>
<label>Table 2</label>
<caption>
<title>Confusion matrix of selected miRNAs between TD-based unsupervised FE and PP in the first data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 2.76 × 10<sup>−23</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t002g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t002" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">PP</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.01</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.01</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.01</td>
<td align="center">812</td>
<td align="center">2</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.01</td>
<td align="center">0</td>
<td align="center">11</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<table-wrap id="pone.0275472.t003" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t003</object-id>
<label>Table 3</label>
<caption>
<title>Confusion matrix of selected mRNAs between TD-based unsupervised FE and PP in the second data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 0.0 within numerical accuracy (i.e., smaller than the possible smallest number given numerical accuracy).</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t003g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t003" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">PP</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">33781</td>
<td align="center">8</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">23</td>
<td align="center">186</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<table-wrap id="pone.0275472.t004" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t004</object-id>
<label>Table 4</label>
<caption>
<title>Confusion matrix of selected miRNAs between TD based unsupervised FE and PP in the second data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 1.87 × 10<sup>−7</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t004g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t004" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">PP</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.01</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.01</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.01</td>
<td align="center">316</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.01</td>
<td align="center">0</td>
<td align="center">3</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
</sec>
<sec id="sec014">
<title>Equivalence between K-means and PCA</title>
<p>To understand these excellent and unexpected coincidences between TD-based unsupervised FE and PP, we first considered the relationship between PCA and PP and later related it with TD. PCA was known to be equivalent to K-means [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>]; the space spanned by centroids of optimal sample clusters can be reproduced by the PC score attributed to the features. Suppose that we have <inline-formula id="pone.0275472.e061"><alternatives><graphic id="pone.0275472.e061g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e061" xlink:type="simple"/><mml:math display="inline" id="M61"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> which is the value of the <italic>i</italic>th feature of the <italic>j</italic>th sample. <italic>M</italic> samples are supposed to be clustered into <italic>S</italic> clusters. The centroid of <italic>s</italic>th cluster, <inline-formula id="pone.0275472.e062"><alternatives><graphic id="pone.0275472.e062g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e062" xlink:type="simple"/><mml:math display="inline" id="M62"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> is defined as
<disp-formula id="pone.0275472.e063"><alternatives><graphic id="pone.0275472.e063g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e063" xlink:type="simple"/><mml:math display="block" id="M63"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:mfrac><mml:mn>1</mml:mn> <mml:msub><mml:mi>n</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:mfrac> <mml:munder><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>j</mml:mi> <mml:mo>∈</mml:mo> <mml:msub><mml:mi>C</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:munder> <mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(32)</label></disp-formula>
where <inline-formula id="pone.0275472.e064"><alternatives><graphic id="pone.0275472.e064g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e064" xlink:type="simple"/><mml:math display="inline" id="M64"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mn>1</mml:mn> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>,</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mn>2</mml:mn> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>,</mml:mo> <mml:mo>⋯</mml:mo> <mml:mo>,</mml:mo> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mi>T</mml:mi></mml:msup> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, <italic>C</italic><sub><italic>s</italic></sub> is a set of <italic>j</italic>s that belong to the <italic>s</italic>th cluster, <italic>n</italic><sub><italic>s</italic></sub> is the size of the <italic>s</italic>th cluster. Here we define the projection of any vector <inline-formula id="pone.0275472.e065"><alternatives><graphic id="pone.0275472.e065g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e065" xlink:type="simple"/><mml:math display="inline" id="M65"><mml:mrow><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> onto the centroid subspace as
<disp-formula id="pone.0275472.e066"><alternatives><graphic id="pone.0275472.e066g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e066" xlink:type="simple"/><mml:math display="block" id="M66"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>S</mml:mi> <mml:mi>b</mml:mi></mml:msub> <mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:munderover> <mml:msub><mml:mi>n</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>(</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mi>s</mml:mi> <mml:mi>T</mml:mi></mml:msubsup> <mml:mo>·</mml:mo> <mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(33)</label></disp-formula>
where
<disp-formula id="pone.0275472.e067"><alternatives><graphic id="pone.0275472.e067g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e067" xlink:type="simple"/><mml:math display="block" id="M67"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>S</mml:mi> <mml:mi>b</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:munderover> <mml:msub><mml:mi>n</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mn>2</mml:mn> <mml:mi>T</mml:mi></mml:msubsup> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(34)</label></disp-formula>
where ⊗ is the Kronecker product. <italic>S</italic><sub><italic>b</italic></sub> is also known to be represented as
<disp-formula id="pone.0275472.e068"><alternatives><graphic id="pone.0275472.e068g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e068" xlink:type="simple"/><mml:math display="block" id="M68"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>S</mml:mi> <mml:mi>b</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:munderover> <mml:mi>X</mml:mi> <mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi> <mml:mi>T</mml:mi></mml:msubsup> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:mo>=</mml:mo> <mml:mi>X</mml:mi> <mml:mo>(</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:munderover> <mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi> <mml:mi>T</mml:mi></mml:msubsup> <mml:mo>)</mml:mo> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(35)</label></disp-formula>
where <inline-formula id="pone.0275472.e069"><alternatives><graphic id="pone.0275472.e069g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e069" xlink:type="simple"/><mml:math display="inline" id="M69"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>M</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> is
<disp-formula id="pone.0275472.e070"><alternatives><graphic id="pone.0275472.e070g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e070" xlink:type="simple"/><mml:math display="block" id="M70"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mrow><mml:mi mathvariant="bold-italic">j</mml:mi><mml:mi mathvariant="bold-italic">s</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:mo>{</mml:mo> <mml:mtable><mml:mtr><mml:mtd><mml:mfrac><mml:mn>1</mml:mn> <mml:msqrt><mml:msub><mml:mi>n</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:msqrt></mml:mfrac></mml:mtd> <mml:mtd><mml:mrow><mml:mi>j</mml:mi> <mml:mo>∈</mml:mo> <mml:msub><mml:mi>C</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr> <mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd> <mml:mtd><mml:mrow><mml:mi>j</mml:mi> <mml:mo>∉</mml:mo> <mml:msub><mml:mi>C</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable> <mml:mo/> <mml:mrow> <mml:mo>,</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(36)</label></disp-formula>
which take non-zero values only when the <italic>j</italic>th sample belongs to the <italic>s</italic>th cluster. K-means is an algorithm to find clusters that minimize
<disp-formula id="pone.0275472.e071"><alternatives><graphic id="pone.0275472.e071g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e071" xlink:type="simple"/><mml:math display="block" id="M71"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>J</mml:mi> <mml:mi>S</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:munderover> <mml:munder><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>j</mml:mi> <mml:mo>∈</mml:mo> <mml:msub><mml:mi>C</mml:mi> <mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:munder> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi mathvariant="bold-italic">x</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:mo>-</mml:mo> <mml:msub><mml:mi mathvariant="bold-italic">m</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(37)</label></disp-formula>
Minimization of <italic>J</italic><sub><italic>k</italic></sub> is known to be equivalent to the maximization of TrS<sub><italic>b</italic></sub>, which means the trace of matrix <italic>S</italic><sub><italic>b</italic></sub>. It is known that
<disp-formula id="pone.0275472.e072"><alternatives><graphic id="pone.0275472.e072g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e072" xlink:type="simple"/><mml:math display="block" id="M72"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:munder><mml:mtext>min</mml:mtext> <mml:mrow><mml:mo>{</mml:mo> <mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>}</mml:mo></mml:mrow></mml:munder> <mml:mspace width="2pt"/><mml:msub><mml:mi>S</mml:mi> <mml:mi>b</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mi>S</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:munderover> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi> <mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(38)</label></disp-formula>
where <inline-formula id="pone.0275472.e073"><alternatives><graphic id="pone.0275472.e073g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e073" xlink:type="simple"/><mml:math display="inline" id="M73"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mi>N</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> is the vector whose components are <italic>ℓ</italic>th PC scores attributed to the features and eigenvector of the gram matrix as
<disp-formula id="pone.0275472.e074"><alternatives><graphic id="pone.0275472.e074g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e074" xlink:type="simple"/><mml:math display="block" id="M74"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:mi>X</mml:mi> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(39)</label></disp-formula>
If we compare <xref ref-type="disp-formula" rid="pone.0275472.e068">Eq (35)</xref> with <xref ref-type="disp-formula" rid="pone.0275472.e072">Eq (38)</xref>, we can notice that <inline-formula id="pone.0275472.e075"><alternatives><graphic id="pone.0275472.e075g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e075" xlink:type="simple"/><mml:math display="inline" id="M75"><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>s</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>S</mml:mi></mml:msubsup> <mml:mi>X</mml:mi> <mml:msub><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">h</mml:mi> <mml:mi>s</mml:mi> <mml:mi>T</mml:mi></mml:msubsup> <mml:msup><mml:mi>X</mml:mi> <mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></alternatives></inline-formula> corresponds to <inline-formula id="pone.0275472.e076"><alternatives><graphic id="pone.0275472.e076g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e076" xlink:type="simple"/><mml:math display="inline" id="M76"><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mrow><mml:mi>S</mml:mi> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn></mml:mrow></mml:msubsup> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi></mml:msub> <mml:mo>⊗</mml:mo> <mml:msubsup><mml:mi mathvariant="bold-italic">u</mml:mi> <mml:mi>ℓ</mml:mi> <mml:mi>T</mml:mi></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>, and PCA can give us an optimal centroid subspace, <italic>S</italic><sub><italic>b</italic></sub>, even without realizing the clusters by K-means, i.e., in a fully unsupervised manner.</p>
<p>At first, when the clusters are the solution of K-means, the centroid subspace can be represented by the PC score which can also be expressed by <italic>X</italic> <bold><italic>h</italic></bold><sub><italic>s</italic></sub>. <bold><italic>h</italic></bold><sub><italic>s</italic></sub> is clearly coincident with <italic>y</italic><sub><italic>j</italic></sub> defined in <xref ref-type="disp-formula" rid="pone.0275472.e056">Eq (27)</xref>. This means that PP employing <bold><italic>u</italic></bold><sub><italic>ℓ</italic></sub> as <bold><italic>b</italic></bold> should result in projection onto the centroid subspace when <italic>y</italic><sub><italic>j</italic></sub> is coincident with the clusters. Here we define <italic>y</italic><sub><italic>j</italic></sub>, <xref ref-type="disp-formula" rid="pone.0275472.e056">Eq (27)</xref>, such that it can represent distinction between tumors and normal tissues, which should be detected by K-means. This explains why TD-based unsupervised FE works well and why PP can be replaced with TD-based unsupervised FE. To our knowledge, this is the first rationalization on why TD- and PCA-based unsupervised FE work well.</p>
<p>One might wonder whether the above explanation is applicable to PCA while TD was applied to the first and second data sets. This gap can be explained as follows. Tensor <italic>x</italic><sub><italic>ijk</italic></sub>, was generated as the product of <italic>x</italic><sub><italic>ij</italic></sub> and <italic>x</italic><sub><italic>kj</italic></sub>. Suppose these two are decomposed as
<disp-formula id="pone.0275472.e077"><alternatives><graphic id="pone.0275472.e077g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e077" xlink:type="simple"/><mml:math display="block" id="M77"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(40)</label></disp-formula> <disp-formula id="pone.0275472.e078"><alternatives><graphic id="pone.0275472.e078g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e078" xlink:type="simple"/><mml:math display="block" id="M78"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>k</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:msubsup><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msubsup> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:msubsup><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mo>′</mml:mo></mml:msubsup> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(41)</label></disp-formula>
If <inline-formula id="pone.0275472.e079"><alternatives><graphic id="pone.0275472.e079g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e079" xlink:type="simple"/><mml:math display="inline" id="M79"><mml:mrow><mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msubsup><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mo>′</mml:mo></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula> then
<disp-formula id="pone.0275472.e080"><alternatives><graphic id="pone.0275472.e080g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e080" xlink:type="simple"/><mml:math display="block" id="M80"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mi>j</mml:mi></mml:munder> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>j</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:mi>j</mml:mi></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:munder><mml:mo>∑</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup></mml:munder> <mml:msubsup><mml:mo>λ</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mo>′</mml:mo></mml:msubsup> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(42)</label></disp-formula> <disp-formula id="pone.0275472.e081"><alternatives><graphic id="pone.0275472.e081g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e081" xlink:type="simple"/><mml:math display="block" id="M81"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mo>=</mml:mo></mml:mtd> <mml:mtd columnalign="left"><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup></mml:munder> <mml:msubsup><mml:mo>λ</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mo>′</mml:mo></mml:msubsup> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:munder><mml:mo>∑</mml:mo> <mml:mi>j</mml:mi></mml:munder> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(43)</label></disp-formula> <disp-formula id="pone.0275472.e082"><alternatives><graphic id="pone.0275472.e082g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e082" xlink:type="simple"/><mml:math display="block" id="M82"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd/><mml:mtd><mml:mo>=</mml:mo></mml:mtd> <mml:mtd columnalign="left"><mml:mrow><mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:munder><mml:mo>∑</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup></mml:munder> <mml:msubsup><mml:mo>λ</mml:mo> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup> <mml:mo>′</mml:mo></mml:msubsup> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>δ</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:msup><mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:mi>ℓ</mml:mi></mml:munder> <mml:msub><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi></mml:msub> <mml:msubsup><mml:mo>λ</mml:mo> <mml:mi>ℓ</mml:mi> <mml:mo>′</mml:mo></mml:msubsup> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(44)</label></disp-formula>
This means that if <inline-formula id="pone.0275472.e083"><alternatives><graphic id="pone.0275472.e083g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e083" xlink:type="simple"/><mml:math display="inline" id="M83"><mml:mrow><mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msubsup><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mo>′</mml:mo></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>, the SVD of <italic>x</italic><sub><italic>ik</italic></sub> gives <italic>u</italic><sub><italic>ℓ</italic><italic>i</italic></sub> and <italic>u</italic><sub><italic>ℓ</italic><italic>k</italic></sub> that are obtained when SVD is applied to <italic>x</italic><sub><italic>ij</italic></sub> and <italic>x</italic><sub><italic>kj</italic></sub>. Here <inline-formula id="pone.0275472.e084"><alternatives><graphic id="pone.0275472.e084g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e084" xlink:type="simple"/><mml:math display="inline" id="M84"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:mn>2</mml:mn> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>mRNA</mml:mtext></mml:msubsup></mml:math></alternatives></inline-formula> is highly correlated with <inline-formula id="pone.0275472.e085"><alternatives><graphic id="pone.0275472.e085g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e085" xlink:type="simple"/><mml:math display="inline" id="M85"><mml:msubsup><mml:mi>u</mml:mi> <mml:mrow><mml:mn>2</mml:mn> <mml:mi>j</mml:mi></mml:mrow> <mml:mtext>miRNA</mml:mtext></mml:msubsup></mml:math></alternatives></inline-formula> [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>]. This is coincident with the requirement <inline-formula id="pone.0275472.e086"><alternatives><graphic id="pone.0275472.e086g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e086" xlink:type="simple"/><mml:math display="inline" id="M86"><mml:mrow><mml:msub><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:msubsup><mml:mi>v</mml:mi> <mml:mrow><mml:mi>ℓ</mml:mi> <mml:mi>j</mml:mi></mml:mrow> <mml:mo>′</mml:mo></mml:msubsup></mml:mrow></mml:math></alternatives></inline-formula>. As SVD is equivalent to PCA, this might explain why TD-based unsupervised FE works well even though the above rationalization is applied only to PCA.</p>
</sec>
<sec id="sec015">
<title>The third data set</title>
<p>Next, we would like to extend the above discussion to TD. Therefore, we consider a third data set analyzed in another study [<xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>] where we performed <italic>in silico</italic> drug discovery for SARS-CoV-2 by applying TD-based unsupervised FE to the gene expression profiles of human cell lines infected with SARS-CoV-2. The third data set comprises five cell lines infected with either mock (control) or SARS-Cov-2, including three biological replicates. It is formatted as tensor, <inline-formula id="pone.0275472.e087"><alternatives><graphic id="pone.0275472.e087g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e087" xlink:type="simple"/><mml:math display="inline" id="M87"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mn>5</mml:mn> <mml:mo>×</mml:mo> <mml:mn>2</mml:mn> <mml:mo>×</mml:mo> <mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>, that represents the expression of the <italic>i</italic>th gene of the <italic>j</italic>th cell line from the infected (k = 1) or control (k = 2) group in the <italic>m</italic>th biological replicate. HOSVD was applied to <italic>x</italic><sub><italic>ijkm</italic></sub> and we got
<disp-formula id="pone.0275472.e088"><alternatives><graphic id="pone.0275472.e088g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e088" xlink:type="simple"/><mml:math display="block" id="M88"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:mo>=</mml:mo> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mn>5</mml:mn></mml:munderover> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mn>2</mml:mn></mml:munderover> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mn>3</mml:mn></mml:munderover> <mml:munderover><mml:mo>∑</mml:mo> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>4</mml:mn></mml:msub> <mml:mo>=</mml:mo> <mml:mn>1</mml:mn></mml:mrow> <mml:mi>N</mml:mi></mml:munderover> <mml:mi>G</mml:mi> <mml:mrow><mml:mo>(</mml:mo> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>4</mml:mn></mml:msub> <mml:mo>)</mml:mo></mml:mrow> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>j</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>2</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>4</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(45)</label></disp-formula>
In this study, we selected <italic>ℓ</italic><sub>1</sub> = 1, <italic>ℓ</italic><sub>2</sub> = 2, <italic>ℓ</italic><sub>3</sub> = 1 based on biological discussions. We then realized that <italic>G</italic>(5, 1, 2, 1) has the largest absolute value given <italic>ℓ</italic><sub>1</sub> = 1, <italic>ℓ</italic><sub>2</sub> = 2, <italic>ℓ</italic><sub>3</sub> = 1. Thus, <italic>u</italic><sub>5<italic>i</italic></sub> was used to attribute <italic>P</italic>-values to gene <italic>i</italic> using
<disp-formula id="pone.0275472.e089"><alternatives><graphic id="pone.0275472.e089g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e089" xlink:type="simple"/><mml:math display="block" id="M89"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:mn>5</mml:mn> <mml:mi>i</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mn>5</mml:mn></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(46)</label></disp-formula>
and the obtained <italic>P</italic>-values were corrected using the by BH criterion; further, 163 genes associated with adjusted <italic>P</italic>-values less than 0.01 were selected. We now relate TD to the above discussion about PCA. Because of the HOSVD algorithm, <inline-formula id="pone.0275472.e090"><alternatives><graphic id="pone.0275472.e090g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e090" xlink:type="simple"/><mml:math display="inline" id="M90"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>4</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> can also be obtained by applying SVD to the unfolded matrix, <inline-formula id="pone.0275472.e091"><alternatives><graphic id="pone.0275472.e091g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e091" xlink:type="simple"/><mml:math display="inline" id="M91"><mml:mrow><mml:mi>X</mml:mi> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">R</mml:mi> <mml:mrow><mml:mi>N</mml:mi> <mml:mo>×</mml:mo> <mml:mn>30</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>. Here 30 columns correspond to one of 30 combinations of <italic>j</italic>, <italic>k</italic>, <italic>m</italic>. Here we select <italic>ℓ</italic><sub>1</sub> = 1, <italic>ℓ</italic><sub>2</sub> = 2, <italic>ℓ</italic><sub>3</sub> = 1 so that the gene expression is independent of the cell lines and biological replicates and has opposite signs between the control and infected cells. Thus, two clusters are expected, each of which corresponds to either the control or infected cell lines. The reason why <italic>ℓ</italic><sub>4</sub> = 5 is selected is simply because <italic>u</italic><sub>5<italic>i</italic></sub> is composed of the centroid subspace coincident with two clusters. Thus, in this sense, the above discussion about PCA can be directly applied to this result.</p>
<p>To confirm this, <italic>y</italic><sub><italic>j</italic></sub> was taken to be
<disp-formula id="pone.0275472.e092"><alternatives><graphic id="pone.0275472.e092g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e092" xlink:type="simple"/><mml:math display="block" id="M92"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>y</mml:mi> <mml:mrow><mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>α</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:msub><mml:mi>β</mml:mi> <mml:mi>k</mml:mi></mml:msub> <mml:msub><mml:mi>γ</mml:mi> <mml:mi>m</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(47)</label></disp-formula> <disp-formula id="pone.0275472.e093"><alternatives><graphic id="pone.0275472.e093g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e093" xlink:type="simple"/><mml:math display="block" id="M93"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>α</mml:mi> <mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(48)</label></disp-formula> <disp-formula id="pone.0275472.e094"><alternatives><graphic id="pone.0275472.e094g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e094" xlink:type="simple"/><mml:math display="block" id="M94"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>β</mml:mi> <mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mo>-</mml:mo> <mml:mn>1</mml:mn> <mml:mo>)</mml:mo></mml:mrow> <mml:mi>k</mml:mi></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(49)</label></disp-formula> <disp-formula id="pone.0275472.e095"><alternatives><graphic id="pone.0275472.e095g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e095" xlink:type="simple"/><mml:math display="block" id="M95"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="left"><mml:msub><mml:mi>γ</mml:mi> <mml:mi>m</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(50)</label></disp-formula>
such that it represented the distinction between <italic>k</italic> = 1 and <italic>k</italic> = 2 (i.e. that between infected and control cell lines), where <inline-formula id="pone.0275472.e096"><alternatives><graphic id="pone.0275472.e096g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e096" xlink:type="simple"/><mml:math display="inline" id="M96"><mml:mrow><mml:msub><mml:mi>y</mml:mi> <mml:mrow><mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">N</mml:mi> <mml:mrow><mml:mn>5</mml:mn> <mml:mo>×</mml:mo> <mml:mn>2</mml:mn> <mml:mo>×</mml:mo> <mml:mn>3</mml:mn></mml:mrow></mml:msup> <mml:mo>,</mml:mo> <mml:msub><mml:mi>α</mml:mi> <mml:mi>j</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">N</mml:mi> <mml:mn>5</mml:mn></mml:msup> <mml:mo>,</mml:mo> <mml:msub><mml:mi>β</mml:mi> <mml:mi>k</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">N</mml:mi> <mml:mn>2</mml:mn></mml:msup> <mml:mo>,</mml:mo></mml:mrow></mml:math></alternatives></inline-formula> and <inline-formula id="pone.0275472.e097"><alternatives><graphic id="pone.0275472.e097g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e097" xlink:type="simple"/><mml:math display="inline" id="M97"><mml:mrow><mml:msub><mml:mi>γ</mml:mi> <mml:mi>m</mml:mi></mml:msub> <mml:mo>∈</mml:mo> <mml:msup><mml:mi mathvariant="double-struck">N</mml:mi> <mml:mn>3</mml:mn></mml:msup></mml:mrow></mml:math></alternatives></inline-formula>. Then PP was performed as
<disp-formula id="pone.0275472.e098"><alternatives><graphic id="pone.0275472.e098g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e098" xlink:type="simple"/><mml:math display="block" id="M98"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>b</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:munder><mml:mo>∑</mml:mo> <mml:mrow><mml:mi>j</mml:mi> <mml:mo>,</mml:mo> <mml:mi>k</mml:mi> <mml:mo>,</mml:mo> <mml:mi>m</mml:mi></mml:mrow></mml:munder> <mml:msub><mml:mi>x</mml:mi> <mml:mrow><mml:mi>i</mml:mi> <mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:msub><mml:mi>y</mml:mi> <mml:mrow><mml:mi>j</mml:mi> <mml:mi>k</mml:mi> <mml:mi>m</mml:mi></mml:mrow></mml:msub> <mml:mrow> <mml:mo>.</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(51)</label></disp-formula> <italic>P</italic>-values were attributed to genes as
<disp-formula id="pone.0275472.e099"><alternatives><graphic id="pone.0275472.e099g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e099" xlink:type="simple"/><mml:math display="block" id="M99"><mml:mtable displaystyle="true"><mml:mtr><mml:mtd columnalign="right"><mml:mrow><mml:msub><mml:mi>P</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:mo>=</mml:mo> <mml:msub><mml:mi>P</mml:mi> <mml:msup><mml:mi>χ</mml:mi> <mml:mn>2</mml:mn></mml:msup></mml:msub> <mml:mo>[</mml:mo> <mml:mo>&gt;</mml:mo> <mml:msup><mml:mrow><mml:mo>(</mml:mo> <mml:mfrac><mml:msub><mml:mi>b</mml:mi> <mml:mi>i</mml:mi></mml:msub> <mml:msub><mml:mi>σ</mml:mi> <mml:mi>b</mml:mi></mml:msub></mml:mfrac> <mml:mo>)</mml:mo></mml:mrow> <mml:mn>2</mml:mn></mml:msup> <mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></alternatives> <label>(52)</label></disp-formula>
and 155 genes associated with corrected <italic>P</italic>-values less than 0.01 were selected, where <italic>b</italic><sub><italic>i</italic></sub> is expected to play a role of <italic>u</italic><sub>5<italic>i</italic></sub> in <xref ref-type="disp-formula" rid="pone.0275472.e089">Eq (46)</xref>. <xref ref-type="table" rid="pone.0275472.t005">Table 5</xref> lists high coincidence of selected genes between TD-based unsupervised FE and PP. <xref ref-type="fig" rid="pone.0275472.g005">Fig 5</xref> shows the overall coincidence of distributions of <italic>P</italic>-values between TD-based unsupervised FE and PP. Thus, why TD based unsupervised FE can work well is explained by the ability of singular value vectors to generate a centroid subspace of clusters coincident with control and infected cell lines.</p>
<table-wrap id="pone.0275472.t005" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t005</object-id>
<label>Table 5</label>
<caption>
<title>Confusion matrix of selected genes between TD-based unsupervised FE and PP in the third data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 1.40 × 10<sup>−241</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t005g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t005" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">PP</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">21582</td>
<td align="center">52</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">60</td>
<td align="center">103</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>One might wonder why TD is needed if <inline-formula id="pone.0275472.e100"><alternatives><graphic id="pone.0275472.e100g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e100" xlink:type="simple"/><mml:math display="inline" id="M100"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>4</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> can be computed by applying SVD to the unfolded matrix. To understand this, we compared <italic>v</italic><sub>5(<italic>ijk</italic>)</sub> obtained by applying SVD to an unfolded matrix, and corresponding to <italic>u</italic><sub>5<italic>i</italic></sub> as well as <italic>u</italic><sub>1<italic>j</italic></sub><italic>u</italic><sub>2<italic>k</italic></sub><italic>u</italic><sub>1<italic>m</italic></sub> with <italic>y</italic><sub><italic>ikm</italic></sub>. While <italic>u</italic><sub>1<italic>j</italic></sub><italic>u</italic><sub>2<italic>k</italic></sub><italic>u</italic><sub>1<italic>m</italic></sub> is well coincident with <italic>y</italic><sub><italic>jkm</italic></sub>, <italic>v</italic><sub>5(<italic>jkm</italic>)</sub> is not (<xref ref-type="fig" rid="pone.0275472.g008">Fig 8</xref>). Thus, we need to apply TD to <italic>x</italic><sub><italic>ijkm</italic></sub> to obtain singular value vectors attributed to samples, which are coincident with two clusters but cannot be obtained when SVD is applied to an unfolded matrix.</p>
<fig id="pone.0275472.g008" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.g008</object-id>
<label>Fig 8</label>
<caption>
<title>Comparisons between <italic>y</italic><sub><italic>jkm</italic></sub> and either <italic>v</italic><sub>5(<italic>jkm</italic>)</sub> or <italic>u</italic><sub>1<italic>j</italic></sub><italic>u</italic><sub>2<italic>k</italic></sub><italic>u</italic><sub>1<italic>m</italic></sub>.</title>
<p>Red straight lines indicate linear regressions.</p>
</caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.g008" xlink:type="simple"/>
</fig>
</sec>
<sec id="sec016">
<title>Rationalization of threshold <italic>P</italic>-values</title>
<p>As we have successfully shown that TD as well as PCA are equivalent to PP that aims to maximize projection onto the subspace centroid of clusters coincident with the desired distinction (cancer vs. normal tissue or control vs. infected cell lines), we would next like to rationalize the <italic>P</italic>-values computed by the <italic>χ</italic><sup>2</sup> distribution and threshold values of <italic>P</italic> = 0.01, which have long been employed to select DEGs with PCA- and TD-based unsupervised FE. Because distribution of projection in the infinite sample number limits is proven to be always Gaussian [<xref ref-type="bibr" rid="pone.0275472.ref006">6</xref>], this null hypothesis might seem reasonable. Nonetheless, the individual distribution of gene expression is far from Gaussian and is rather close to negative signed binomial distribution and when the number of samples is not large enough, the distribution of projection does not converge with a Gaussian distribution at all. Thus, a more straightforward rationalization is needed. Therefore, we generated a null distribution by shuffling <italic>i</italic> in each sample and recomputed the singular value vectors, <inline-formula id="pone.0275472.e101"><alternatives><graphic id="pone.0275472.e101g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e101" xlink:type="simple"/><mml:math display="inline" id="M101"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>1</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> (for mRNA in the first and the second data sets), <inline-formula id="pone.0275472.e102"><alternatives><graphic id="pone.0275472.e102g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e102" xlink:type="simple"/><mml:math display="inline" id="M102"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>3</mml:mn></mml:msub> <mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula> (for miRNA in the first and the second data sets), and <inline-formula id="pone.0275472.e103"><alternatives><graphic id="pone.0275472.e103g" mimetype="image" position="anchor" xlink:href="info:doi/10.1371/journal.pone.0275472.e103" xlink:type="simple"/><mml:math display="inline" id="M103"><mml:mrow><mml:msub><mml:mi>u</mml:mi> <mml:mrow><mml:msub><mml:mi>ℓ</mml:mi> <mml:mn>5</mml:mn></mml:msub> <mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></alternatives></inline-formula>(for genes in the third data set). Then <italic>P</italic>-values were recomputed using the generated null distribution and were corrected using the BH criterion to obtain genes associated with significant adjusted <italic>P</italic>-values. In the following, we apply the shuffling to three data sets, the first, the second, and the third data set, and select genes using <italic>P</italic>-values obtained by shuffling. Coincidence of selected genes and distribution of <italic>P</italic>-values between PCA or TD and shuffling is estimated. These evaluations enable us to discuss the suitability of threshold <italic>P</italic>-values.</p>
<p><xref ref-type="fig" rid="pone.0275472.g001">Fig 1(A)</xref> shows the histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling one hundred times when the miRNAs in the first data set were considered. As it is obvious that there are too many <italic>P</italic>-values near 1, we excluded some miRNAs with low values to obtain a <italic>P</italic>-value distribution more coincident with the null distribution. <xref ref-type="fig" rid="pone.0275472.g001">Fig 1(B)</xref> shows the histogram of raw <italic>P</italic>-values computed to be restricted to the top 500 more expressive miRNAs; this seems more coincident with the null distribution. We then found that twelve miRNAs are associated with adjusted <italic>P</italic>-values less than 0.1. <xref ref-type="table" rid="pone.0275472.t006">Table 6</xref> lists the comparison of selected miRNAs between TD-based unsupervised FE and the null distribution generated by shuffling. Although the threshold <italic>P</italic>-values differ between the two, the selected miRNAs are quite coincident. A threshold <italic>P</italic>-value 0.01 was empirically employed for PCA- and TD-based unsupervised FE as it often gave us biologically reasonable results. <italic>P</italic> = 0.01 in Gaussian distribution is assumed as the null hypothesis corresponding to <italic>P</italic> = 0.1 when the null distribution is generated by shuffling. Although this discrepancy must be fulfilled in the future, we conclude that their performances are quite similar.</p>
<table-wrap id="pone.0275472.t006" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t006</object-id>
<label>Table 6</label>
<caption>
<title>Confusion matrix of selected miRNAs between TD-based unsupervised FE and shuffling in the first data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 1.28 × 10<sup>−21</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t006g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t006" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">shuffling</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.1</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.1</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &gt; 0.01</td>
<td align="center">488</td>
<td align="center">1</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>k</italic></sub> &lt; 0.01</td>
<td align="center">0</td>
<td align="center">11</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>
<xref ref-type="fig" rid="pone.0275472.g002">Fig 2(A)</xref> shows the histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling one hundred times when mRNAs in the first data set were considered. As it is obvious that there are too many <italic>P</italic>-values near 1, we excluded some mRNAs with low values to obtain a <italic>P</italic>-value distribution more coincident with the null distribution. <xref ref-type="fig" rid="pone.0275472.g002">Fig 2(B)</xref> shows the histogram of raw <italic>P</italic>-values computed to be restricted to the top 3000 more expressive mRNAs; this seems more coincident with the null distribution. We then found that 69 mRNAs are associated with adjusted <italic>P</italic>-values less than 0.1. <xref ref-type="table" rid="pone.0275472.t007">Table 7</xref> lists the comparison of selected mRNAs between TD-based unsupervised FE and the null distribution generated by shuffling. Although threshold <italic>P</italic>-values differ between the two, the selected mRNAs are quite coincident. A threshold <italic>P</italic>-value 0.01 was empirically employed for PCA- and TD-based unsupervised FE as it often gave us biologically reasonable results. <italic>P</italic> = 0.01 in a Gaussian distribution is assumed as the null hypothesis corresponding to <italic>P</italic> = 0.1 when the null distribution is generated by shuffling. Although this discrepancy must be fulfilled in the future, we conclude that their performances are quite similar.</p>
<table-wrap id="pone.0275472.t007" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t007</object-id>
<label>Table 7</label>
<caption>
<title>Confusion matrix of selected mRNAs between TD-based unsupervised FE and shuffling in the first data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 2.69 × 10<sup>−137</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t007g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t007" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">shuffling</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.1</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.1</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">2928</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">3</td>
<td align="center">69</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>
<xref ref-type="fig" rid="pone.0275472.g006">Fig 6(A)</xref> shows the histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling one hundred times when miRNAs in the second data set were considered. As it is unlikely to get significant <italic>P</italic>-values, we did not select miRNAs associated with significant <italic>P</italic>-values. <xref ref-type="fig" rid="pone.0275472.g006">Fig 6(B)</xref> shows the histogram of raw <italic>P</italic>-values computed for mRNAs in the second data set; there are no peaks around <italic>P</italic> = 1. We then found that 262 mRNAs are associated with adjusted <italic>P</italic>-values less than 0.1. <xref ref-type="table" rid="pone.0275472.t008">Table 8</xref> lists the comparison of selected mRNAs between TD-based unsupervised FE and the null distribution generated by shuffling. Although threshold <italic>P</italic>-values differ between the two, the selected mRNAs are well coincident. A threshold <italic>P</italic>-value 0.01 was empirically employed for PCA- and TD-based unsupervised FE as it often gave us biologically reasonable results. <italic>P</italic> = 0.01 in a Gaussian distribution is assumed as the null hypothesis corresponding to <italic>P</italic> = 0.1 when the null distribution is generated by shuffling. Although this discrepancy must be fulfilled in the future, we conclude that their performances are quite similar.</p>
<table-wrap id="pone.0275472.t008" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t008</object-id>
<label>Table 8</label>
<caption>
<title>Confusion matrix of selected mRNAs between TD-based unsupervised FE and shuffling in the second data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 0.0 within numerical accuracy (i.e., smaller than the possible smallest number given numerical accuracy).</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t008g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t008" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">shuffling</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.1</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.1</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">33736</td>
<td align="center">53</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">0</td>
<td align="center">209</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
<p>
<xref ref-type="fig" rid="pone.0275472.g003">Fig 3(A)</xref> shows the histogram of raw <italic>P</italic>-values computed using the null distribution generated by shuffling one hundred times when considering the genes in the third data set. As there were too many <italic>P</italic>-values less than 0.2, we excluded some mRNAs with low values to obtain a <italic>P</italic>-value distribution more coincident with the null distribution. <xref ref-type="fig" rid="pone.0275472.g003">Fig 3(B)</xref> shows the histogram of raw <italic>P</italic>-values computed to be restricted to the top 2780 more expressive mRNAs; this seems more coincident with the null distribution. We then found that 48 mRNAs are associated with adjusted <italic>P</italic>-values less than 0.1. <xref ref-type="table" rid="pone.0275472.t009">Table 9</xref> lists the comparison of selected mRNAs between TD-based unsupervised FE and the null distribution generated by shuffling. Although threshold <italic>P</italic>-values differ between two, selected mRNAs are well coincident. A threshold <italic>P</italic>-value 0.01 was empirically employed for PCA- and TD-based unsupervised FE as it often gave us biologically reasonable results. <italic>P</italic> = 0.01 in a Gaussian distribution is assumed as the null hypothesis corresponding to <italic>P</italic> = 0.1 when the null distribution is generated by shuffling. Although this discrepancy must be fulfilled in the future, we conclude that their performances are quite similar.</p>
<table-wrap id="pone.0275472.t009" position="float">
<object-id pub-id-type="doi">10.1371/journal.pone.0275472.t009</object-id>
<label>Table 9</label>
<caption>
<title>Confusion matrix of selected genes between TD-based unsupervised FE and shuffling in the third data set.</title>
<p><italic>P</italic>-value computed by Fisher’s exact test is 5.00 × 10<sup>−63</sup>.</p>
</caption>
<alternatives>
<graphic id="pone.0275472.t009g" mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.t009" xlink:type="simple"/>
<table border="0" frame="box" rules="all">
<colgroup>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
<col align="left" valign="middle"/>
</colgroup>
<tbody>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center" colspan="2">shuffling</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.1</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.1</td>
</tr>
<tr>
<td align="center">TD based unsupervised FE</td>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &gt; 0.01</td>
<td align="center">2617</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center"/>
<td align="center">adjusted <italic>P</italic><sub><italic>i</italic></sub> &lt; 0.01</td>
<td align="center">115</td>
<td align="center">48</td>
</tr>
</tbody>
</table>
</alternatives>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="sec017" sec-type="conclusions">
<title>Discussion</title>
<p>In the previous section, we explained why PCA- and TD-based unsupervised FE work well (because singular value vectors correspond to projection onto the centroid subspace obtained by K-means) and how the criterion to select genes associated with adjusted <italic>P</italic>-values less than 0.01, which was computed assuming the null hypothesis that singular value vectors obey Gaussian distribution, is empirically coincident with another criterion to select the genes associated with adjusted <italic>P</italic>-values less than 0.1, which are computed assuming the null distribution generated by shuffling.</p>
<p>There are many points to be discussed. In the above example, we only dealt with the case wherein only two clusters could be distinguished in a one-dimensional space (i.e., only one singular value vector). Considering cases with more clusters might be challenging, projections onto subspace centroids do not have a one-to-one correspondence with singular value vectors as the coincidence between the projection to the subspace centroid and singular value vectors stands only between the spaces spanned by them, and not between themselves. Despite this, TD- and PCA-based unsupervised FE applied to more than two classes is known to work rather as well as in the case with only two clusters [<xref ref-type="bibr" rid="pone.0275472.ref016">16</xref>].</p>
<p>On the contrary, although we could only discuss cases with a finite number of clusters, PCA- and TD-based unsupervised FE are also known to work in detecting parameter dependence, e.g., time development [<xref ref-type="bibr" rid="pone.0275472.ref017">17</xref>, <xref ref-type="bibr" rid="pone.0275472.ref018">18</xref>]. Extending the discussion here to regression analysis without any clusters will be the next step.</p>
<p>One might also wonder whether we need TD if singular value vectors attributed to genes are common between TD and PCA. At first, in the integrated analysis of mRNA and miRNA, TD-based unsupervised FE could outperform PCA-based unsupervised FE [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>]. Similarly, TD-based unsupervised FE outperformed PCA-based unsupervised FE in the integrated analysis of gene expression and DNA methylation [<xref ref-type="bibr" rid="pone.0275472.ref019">19</xref>]. Thus, TD-based unsupervised FE is required when integrated analysis is targeted. Even when no integrated analysis was targeted, TD based unsupervised FE can give singular value vectors that are more coincident with biological clusters (<xref ref-type="fig" rid="pone.0275472.g008">Fig 8</xref>). Thus, despite the apparent equality of singular value vectors attributed to genes between TD and PCA, TD-based unsupervised FE is a more useful strategy than PCA-based unsupervised FE.</p>
<p>Although we did not clearly denote this, conventional gene selection strategies based on statistical tests are known to fail when applied to the first, second, and third data sets [<xref ref-type="bibr" rid="pone.0275472.ref012">12</xref>, <xref ref-type="bibr" rid="pone.0275472.ref013">13</xref>]; they always selected too many or too few genes, mRNAs, and miRNA, which is in contrast to TD-based unsupervised FE that could always select a restricted number of genes, from tens to hundreds.</p>
<p>One might also wonder why we did not employ the null distribution generated by shuffling instead of the un-justified Gaussian distribution, with PCA- and TD-based unsupervised FE. As can be seen above, employment of null distribution generated by shuffling is not straightforward; in some cases, e.g, the first and the third data sets mentioned above, we needed to exclude low expressed genes manually whereas this was not required for the second data set. No miRNAs that were significantly expressed distinctly between controls and cancers in the second data sets were detected with the null distribution generated by shuffling. In addition, the number of low expressed genes to be removed cannot be decided uniquely. On the contrary, the criterion that genes associated with adjusted <italic>P</italic>-values less than 0.01 assuming the null hypothesis that singular value vectors obey a Gaussian distribution is more robust. This often can give a restricted number of genes without excluding low expressed genes. Although why this works so well must be explored in the future, it is an empirically more useful strategy than the null distributions generated by shuffling.</p>
<p>One may also wonder why we did not employ the centroid subspace, <italic>S</italic><sub><italic>b</italic></sub>, instead of singular value vectors if these two are equivalent for optimal clusters and the meaning of centroid subspace is easier to understand compared to singular value vectors. At first, we needed to apply K-means which often fail in unbalanced data sets composed of clusters with a very distinct number of samples. Next, K-means always identifies the primary cluster. Nevertheless, in the case of SARS-CoV-2 (the third data set), distinction between infected cell lines and control cell lines was detected using the fifth singular value vectors whose contribution will probably be neglected by K-means because of its too small contribution. In addition, singular value vectors can be computed in a fully unsupervised manner that does not require any labeling. Considering these advantages, it is reasonable to use singular value vectors instead of a centroid subspace despite its apparent usefulness. Further, as the <italic>y</italic><sub><italic>j</italic></sub> used to compute projection <bold><italic>b</italic></bold> is decided manually, even if some biological features that <italic>y</italic><sub><italic>j</italic></sub> assumes, such as clusters, do not exist, <bold><italic>b</italic></bold> can be computed. This might result in wrong conclusions. However, if there are no clusters at all, because no corresponding singular value vectors attributed to samples and coincident with <italic>y</italic><sub><italic>j</italic></sub> are obtained, we can have an opportunity to realize any misunderstanding. Thus, usage of singular value vectors but not projection <bold><italic>b</italic></bold> might be advantageous.</p>
<p>One might also wonder why other more frequently used TD such as CP decomposition [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>] have not been employed instead of HOSVD. This might be understood as follows. In the above description, we could relate the singular value vectors obtained by HOSVD to the centroid subspace, because singular value vectors attributed to genes are common between HOSVD and PCA. This equivalence will be broken if HOSVD is replaced with other TDs. When we invented TD-based unsupervised FE, though we also tested other TDs [<xref ref-type="bibr" rid="pone.0275472.ref003">3</xref>], HOSVD always outperformed other TDs when used for feature selections. The equivalence of HOSVD and PCA might explain why HOSVD could outperform other popular TDs as a feature selection tool.</p>
<p>Another possible concern is that only one hundred times shuffling was performed for the computation in Figs <xref ref-type="fig" rid="pone.0275472.g001">1</xref> to <xref ref-type="fig" rid="pone.0275472.g003">3</xref> whereas we considered <italic>P</italic>-values equal to 0.01; nevertheless, it is not problematic at all because of the following two reasons. First of all, the <italic>P</italic>-values we considered were not raw <italic>P</italic>-values but corrected <italic>P</italic>-values. Thus total number of probabilities computed are much larger than one hundred. Since the numbers of computed <italic>P</italic>-values are as many as those of mRNAs and miRNAs, they are as many as 10<sup>3</sup> or 10<sup>4</sup>. Thus, the number of shuffling, one hundred, is not directly related to <italic>P</italic>-values of 0.01 at all. Second, individual <italic>P</italic>-values are not related to the number of shuffling at all; what we have performed was to generate <italic>P</italic>-values whose number is equal to that of miRNAs or mRNAs, i.e., 10<sup>3</sup> or 10<sup>4</sup>. Thus, individual <italic>P</italic>-values can take much smaller values than 0.01, say 10<sup>−3</sup> and 10<sup>−4</sup> for miRNAs and mRNAs, respectively. Increasing or decreasing the number of shuffling does not affect the absolute values of <italic>P</italic>-values at all. The number of shuffling is only related to the reproducibility; if we can compute <italic>P</italic>-values based upon only one shuffling, it might heavily fluctuate. On the other hand, if we take average of <italic>P</italic>-values over one hundred shuffling, their outcome is expected to be more stable. The purpose of taking average over one hundred shuffling is simply because of stability of outcome. Apparent relationship between <italic>P</italic> = 0.01 and one hundred times shuffling does not make any sense. In conclusion, even if we take <italic>P</italic> = 0.01 as a threshold for one hundred times shuffling, it is not a problem at all.</p>
<p>Based upon the studies presented in the above, we emphasize that the usages of PCA or TD based unsupervised FE are recommended, since generally we do not know to which direction we project the data sets. PCA and TD turned out to have ability to give the directions of projections in an unsupervised manner. When projections directions are trivial, e.g., distinction between two classes, PCA and TD can correctly give us the directions. Even if the data sets are more complicated, we can employ higher mode tensors to tackle more complicated data sets. PCA and TD based unsupervised methods will be promising methods.</p>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="pone.0275472.ref001">
<label>1</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Fang</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Martin</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>Z</given-names></name>. <article-title>Statistical methods for identifying differentially expressed genes in RNA-Seq experiments</article-title>. <source>Cell &amp; Bioscience</source>. <year>2012</year>;<volume>2</volume>(<issue>1</issue>):<fpage>26</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/2045-3701-2-26" xlink:type="simple">10.1186/2045-3701-2-26</ext-link></comment> <object-id pub-id-type="pmid">22849430</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref002">
<label>2</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Chen</surname> <given-names>JJ</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>SJ</given-names></name>, <name name-style="western"><surname>Tsai</surname> <given-names>CA</given-names></name>, <name name-style="western"><surname>Lin</surname> <given-names>CJ</given-names></name>. <article-title>Selection of differentially expressed genes in microarray data analysis</article-title>. <source>The Pharmacogenomics Journal</source>. <year>2006</year>;<volume>7</volume>(<issue>3</issue>):<fpage>212</fpage>–<lpage>220</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/sj.tpj.6500412" xlink:type="simple">10.1038/sj.tpj.6500412</ext-link></comment> <object-id pub-id-type="pmid">16940966</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref003">
<label>3</label>
<mixed-citation publication-type="other" xlink:type="simple">Taguchi YH. Unsupervised Feature Extraction Applied to Bioinformatics. Springer International Publishing; 2020. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/978-3-030-22456-1" xlink:type="simple">https://doi.org/10.1007/978-3-030-22456-1</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0275472.ref004">
<label>4</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Tibshirani</surname> <given-names>R</given-names></name>. <article-title>Regression Shrinkage and Selection Via the Lasso</article-title>. <source>JOURNAL OF THE ROYAL STATISTICAL SOCIETY, SERIES B</source>. <year>1994</year>;<volume>58</volume>:<fpage>267</fpage>–<lpage>288</lpage>.</mixed-citation>
</ref>
<ref id="pone.0275472.ref005">
<label>5</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Huber</surname> <given-names>PJ</given-names></name>. <article-title>Projection Pursuit</article-title>. <source>The Annals of Statistics</source>. <year>1985</year>;<volume>13</volume>(<issue>2</issue>):<fpage>435</fpage>–<lpage>475</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1214/aos/1176349519" xlink:type="simple">10.1214/aos/1176349519</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0275472.ref006">
<label>6</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Bickel</surname> <given-names>PJ</given-names></name>, <name name-style="western"><surname>Kur</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Nadler</surname> <given-names>B</given-names></name>. <article-title>Projection pursuit in high dimensions</article-title>. <source>Proceedings of the National Academy of Sciences</source>. <year>2018</year>;<volume>115</volume>(<issue>37</issue>):<fpage>9151</fpage>–<lpage>9156</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1801177115" xlink:type="simple">10.1073/pnas.1801177115</ext-link></comment> <object-id pub-id-type="pmid">30150379</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref007">
<label>7</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ospina</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>López-Kleine</surname> <given-names>L</given-names></name>. <article-title>Identification of differentially expressed genes in microarray data in a principal component space</article-title>. <source>SpringerPlus</source>. <year>2013</year>;<volume>2</volume>(<issue>1</issue>):<fpage>60</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/2193-1801-2-60" xlink:type="simple">10.1186/2193-1801-2-60</ext-link></comment> <object-id pub-id-type="pmid">23539565</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref008">
<label>8</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Clark</surname> <given-names>NR</given-names></name>, <name name-style="western"><surname>Hu</surname> <given-names>KS</given-names></name>, <name name-style="western"><surname>Feldmann</surname> <given-names>AS</given-names></name>, <name name-style="western"><surname>Kou</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Chen</surname> <given-names>EY</given-names></name>, <name name-style="western"><surname>Duan</surname> <given-names>Q</given-names></name>, <etal>et al</etal>. <article-title>The characteristic direction: a geometrical approach to identify differentially expressed genes</article-title>. <source>BMC Bioinformatics</source>. <year>2014</year>;<volume>15</volume>(<issue>1</issue>):<fpage>79</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-15-79" xlink:type="simple">10.1186/1471-2105-15-79</ext-link></comment> <object-id pub-id-type="pmid">24650281</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref009">
<label>9</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Shahbazi</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Monfared</surname> <given-names>MS</given-names></name>, <name name-style="western"><surname>Thiruchelvam</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Ka Fei</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Babasafari</surname> <given-names>AA</given-names></name>. <article-title>Integration of knowledge-based seismic inversion and sedimentological investigations for heterogeneous reservoir</article-title>. <source>Journal of Asian Earth Sciences</source>. <year>2020</year>;<volume>202</volume>:<fpage>104541</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jseaes.2020.104541" xlink:type="simple">10.1016/j.jseaes.2020.104541</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0275472.ref010">
<label>10</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Khayer</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Kahoo</surname> <given-names>AR</given-names></name>, <name name-style="western"><surname>Monfared</surname> <given-names>MS</given-names></name>, <name name-style="western"><surname>Tokhmechi</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Kavousi</surname> <given-names>K</given-names></name>. <article-title>Target-Oriented Fusion of Attributes in Data Level for Salt Dome Geobody Delineation in Seismic Data</article-title>. <source>Natural Resources Research</source>. <year>2022</year>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/s11053-022-10086-z" xlink:type="simple">10.1007/s11053-022-10086-z</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0275472.ref011">
<label>11</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Khayer</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Roshandel-Kahoo</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Soleimani-Monfared</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Kavoosi</surname> <given-names>K</given-names></name>. <article-title>Combination of seismic attributes using graph-based methods to identify the salt dome boundary</article-title>. <source>Journal of Petroleum Science and Engineering</source>. <year>2022</year>;<volume>215</volume>:<fpage>110625</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.petrol.2022.110625" xlink:type="simple">10.1016/j.petrol.2022.110625</ext-link></comment></mixed-citation>
</ref>
<ref id="pone.0275472.ref012">
<label>12</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Ng</surname> <given-names>KL</given-names></name>, <name name-style="western"><surname>Taguchi</surname> <given-names>YH</given-names></name>. <article-title>Identification of miRNA signatures for kidney renal clear cell carcinoma using the tensor-decomposition method</article-title>. <source>Scientific Reports</source>. <year>2020</year>;<volume>10</volume>(<issue>1</issue>). <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41598-020-71997-6" xlink:type="simple">10.1038/s41598-020-71997-6</ext-link></comment> <object-id pub-id-type="pmid">32938959</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref013">
<label>13</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Taguchi</surname> <given-names>Yh</given-names></name>, <name name-style="western"><surname>Turki</surname> <given-names>T</given-names></name>. <article-title>A new advanced in silico drug discovery method for novel coronavirus (SARS-CoV-2) with tensor decomposition-based unsupervised feature extraction</article-title>. <source>PLOS ONE</source>. <year>2020</year>;<volume>15</volume>(<issue>9</issue>):<fpage>1</fpage>–<lpage>16</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0238907" xlink:type="simple">10.1371/journal.pone.0238907</ext-link></comment> <object-id pub-id-type="pmid">32915876</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref014">
<label>14</label>
<mixed-citation publication-type="book" xlink:type="simple">
<name name-style="western"><surname>Dodge</surname> <given-names>Y</given-names></name>. <chapter-title>Q-Q Plot (Quantile to Quantile Plot)</chapter-title>. In: <source>The Concise Encyclopedia of Statistics</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer New York</publisher-name>; <year>2008</year>. p. <fpage>437</fpage>–<lpage>439</lpage>.</mixed-citation>
</ref>
<ref id="pone.0275472.ref015">
<label>15</label>
<mixed-citation publication-type="other" xlink:type="simple">R Core Team. R: A Language and Environment for Statistical Computing; 2019. Available from: <ext-link ext-link-type="uri" xlink:href="https://www.R-project.org/" xlink:type="simple">https://www.R-project.org/</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0275472.ref016">
<label>16</label>
<mixed-citation publication-type="other" xlink:type="simple">Ding C, He X. K-Means Clustering via Principal Component Analysis. In: Proceedings of the Twenty-First International Conference on Machine Learning. ICML’04. New York, NY, USA: Association for Computing Machinery; 2004. p. 29. Available from: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1145/1015330.1015408" xlink:type="simple">https://doi.org/10.1145/1015330.1015408</ext-link>.</mixed-citation>
</ref>
<ref id="pone.0275472.ref017">
<label>17</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Taguchi</surname> <given-names>YH</given-names></name>. <article-title>Principal component analysis based unsupervised feature extraction applied to budding yeast temporally periodic gene expression</article-title>. <source>BioData Mining</source>. <year>2016</year>;<volume>9</volume>(<issue>1</issue>). <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/s13040-016-0101-9" xlink:type="simple">10.1186/s13040-016-0101-9</ext-link></comment> <object-id pub-id-type="pmid">27366210</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref018">
<label>18</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Taguchi</surname> <given-names>Yh</given-names></name>, <name name-style="western"><surname>Turki</surname> <given-names>T</given-names></name>. <article-title>Tensor Decomposition-Based Unsupervised Feature Extraction Applied to Single-Cell Gene Expression Analysis</article-title>. <source>Frontiers in Genetics</source>. <year>2019</year>;<volume>10</volume>:<fpage>864</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2019.00864" xlink:type="simple">10.3389/fgene.2019.00864</ext-link></comment> <object-id pub-id-type="pmid">31608111</object-id></mixed-citation>
</ref>
<ref id="pone.0275472.ref019">
<label>19</label>
<mixed-citation publication-type="journal" xlink:type="simple">
<name name-style="western"><surname>Taguchi</surname> <given-names>YH</given-names></name>. <article-title>Tensor decomposition-based and principal-component-analysis-based unsupervised feature extraction applied to the gene expression and methylation profiles in the brains of social insects with multiple castes</article-title>. <source>BMC Bioinformatics</source>. <year>2018</year>;<volume>19</volume>(<issue>S4</issue>). <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/s12859-018-2068-7" xlink:type="simple">10.1186/s12859-018-2068-7</ext-link></comment> <object-id pub-id-type="pmid">29745827</object-id></mixed-citation>
</ref>
</ref-list>
</back>
<sub-article article-type="aggregated-review-documents" id="pone.0275472.r001" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0275472.r001</article-id>
<title-group>
<article-title>Decision Letter 0</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Chen</surname>
<given-names>Chi-Hua</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2022</copyright-year>
<copyright-holder>Chi-Hua Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pone.0275472" document-id-type="doi" document-type="article" id="rel-obj001" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>0</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">23 Aug 2022</named-content>
</p>
<p><!-- <div> -->PONE-D-22-20332<!-- </div> --><!-- <div> -->Projection in genomic analysis: A theoretical basis to rationalize tensor decomposition and principal component analysis as feature selection tools<!-- </div> --><!-- <div> -->PLOS ONE</p>
<p>Dear Dr. Taguchi,</p>
<p>Thank you for submitting your manuscript to PLOS ONE. After careful consideration, we feel that it has merit but does not fully meet PLOS ONE’s publication criteria as it currently stands. Therefore, we invite you to submit a revised version of the manuscript that addresses the points raised during the review process.</p>
<p>Please submit your revised manuscript by Oct 07 2022 11:59PM. If you will need more time than this to complete your revisions, please reply to this message or contact the journal office at <email xlink:type="simple">plosone@plos.org</email>. When you're ready to submit your revision, log on to <ext-link ext-link-type="uri" xlink:href="https://www.editorialmanager.com/pone/" xlink:type="simple">https://www.editorialmanager.com/pone/</ext-link> and select the 'Submissions Needing Revision' folder to locate your manuscript file.</p>
<p>Please include the following items when submitting your revised manuscript:<!-- </div> --><list list-type="bullet"><list-item><p>A rebuttal letter that responds to each point raised by the academic editor and reviewer(s). You should upload this letter as a separate file labeled 'Response to Reviewers'.</p></list-item><list-item><p>A marked-up copy of your manuscript that highlights changes made to the original version. You should upload this as a separate file labeled 'Revised Manuscript with Track Changes'.</p></list-item><list-item><p>An unmarked version of your revised paper without tracked changes. You should upload this as a separate file labeled 'Manuscript'.</p></list-item></list></p>
<p>If you would like to make changes to your financial disclosure, please include your updated statement in your cover letter. Guidelines for resubmitting your figure files are available below the reviewer comments at the end of this letter.</p>
<p><!-- <div> -->If applicable, we recommend that you deposit your laboratory protocols in protocols.io to enhance the reproducibility of your results. Protocols.io assigns your protocol its own identifier (DOI) so that it can be cited independently in the future. For instructions see: <ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols" xlink:type="simple">https://journals.plos.org/plosone/s/submission-guidelines#loc-laboratory-protocols</ext-link>. Additionally, PLOS ONE offers an option for publishing peer-reviewed Lab Protocol articles, which describe protocols hosted on protocols.io. Read more information on sharing protocols at <ext-link ext-link-type="uri" xlink:href="https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols" xlink:type="simple">https://plos.org/protocols?utm_medium=editorial-email&amp;utm_source=authorletters&amp;utm_campaign=protocols</ext-link>.</p>
<p>We look forward to receiving your revised manuscript.</p>
<p>Kind regards,</p>
<p>Chi-Hua Chen, Ph.D.</p>
<p>Academic Editor</p>
<p>PLOS ONE</p>
<p>Journal Requirements:</p>
<p>When submitting your revision, we need you to address these additional requirements.</p>
<p>1. Please ensure that your manuscript meets PLOS ONE's style requirements, including those for file naming. The PLOS ONE style templates can be found at </p>
<p><ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=wjVg/PLOSOne_formatting_sample_main_body.pdf</ext-link> and </p>
<p><ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf" xlink:type="simple">https://journals.plos.org/plosone/s/file?id=ba62/PLOSOne_formatting_sample_title_authors_affiliations.pdf</ext-link></p>
<p>2. Please update your submission to use the PLOS LaTeX template. The template and more information on our requirements for LaTeX submissions can be found at <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/latex" xlink:type="simple">http://journals.plos.org/plosone/s/latex</ext-link>.</p>
<p>3. We note that the grant information you provided in the ‘Funding Information’ and ‘Financial Disclosure’ sections do not match. </p>
<p>When you resubmit, please ensure that you provide the correct grant numbers for the awards you received for your study in the ‘Funding Information’ section.</p>
<p>4. Thank you for stating the following in the Acknowledgments Section of your manuscript: </p>
<p>"This work was supported by KAKENHI [grant numbers 19H05270, 20H04848, and 20K12067] to YT and Institutional Fund Project (IFPIP) from the Ministry of Education and King Abdulaziz University (DSR), Jeddah, Saudi Arabia [grant number IFPIP: 924-611-1442] to TT."</p>
<p>We note that you have provided funding information that is not currently declared in your Funding Statement. However, funding information should not appear in the Acknowledgments section or other areas of your manuscript. We will only publish funding information present in the Funding Statement section of the online submission form. </p>
<p>Please remove any funding-related text from the manuscript and let us know how you would like to update your Funding Statement. Currently, your Funding Statement reads as follows: </p>
<p>"The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript."</p>
<p>Please include your amended statements within your cover letter; we will change the online submission form on your behalf.</p>
<p>5. In your Data Availability statement, you have not specified where the minimal data set underlying the results described in your manuscript can be found. PLOS defines a study's minimal data set as the underlying data used to reach the conclusions drawn in the manuscript and any additional data required to replicate the reported study findings in their entirety. All PLOS journals require that the minimal data set be made fully available. For more information about our data policy, please see <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/data-availability" xlink:type="simple">http://journals.plos.org/plosone/s/data-availability</ext-link>.</p>
<p>Upon re-submitting your revised manuscript, please upload your study’s minimal underlying data set as either Supporting Information files or to a stable, public repository and include the relevant URLs, DOIs, or accession numbers within your revised cover letter. For a list of acceptable repositories, please see <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/data-availability#loc-recommended-repositories" xlink:type="simple">http://journals.plos.org/plosone/s/data-availability#loc-recommended-repositories</ext-link>. Any potentially identifying patient information must be fully anonymized.</p>
<p>Important: If there are ethical or legal restrictions to sharing your data publicly, please explain these restrictions in detail. Please see our guidelines for more information on what we consider unacceptable restrictions to publicly sharing data: <ext-link ext-link-type="uri" xlink:href="http://journals.plos.org/plosone/s/data-availability#loc-unacceptable-data-access-restrictions" xlink:type="simple">http://journals.plos.org/plosone/s/data-availability#loc-unacceptable-data-access-restrictions</ext-link>. Note that it is not acceptable for the authors to be the sole named individuals responsible for ensuring data access.</p>
<p>We will update your Data Availability statement to reflect the information you provide in your cover letter.</p>
<p>6. Please review your reference list to ensure that it is complete and correct. If you have cited papers that have been retracted, please include the rationale for doing so in the manuscript text, or remove these references and replace them with relevant current references. Any changes to the reference list should be mentioned in the rebuttal letter that accompanies your revised manuscript. If you need to cite a retracted article, indicate the article’s retracted status in the References list and also include a citation and full reference for the retraction notice.</p>
<p>[Note: HTML markup is below. Please do not edit.]</p>
<p>Reviewers' comments:</p>
<p>Reviewer's Responses to Questions</p>
<p><!-- <font color="black"> --><bold>Comments to the Author</bold></p>
<p>1. Is the manuscript technically sound, and do the data support the conclusions?</p>
<p>The manuscript must describe a technically sound piece of scientific research with data that supports the conclusions. Experiments must have been conducted rigorously, with appropriate controls, replication, and sample sizes. The conclusions must be drawn appropriately based on the data presented. <!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>********** </p>
<p><!-- <font color="black"> -->2. Has the statistical analysis been performed appropriately and rigorously? <!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>********** </p>
<p><!-- <font color="black"> -->3. Have the authors made all data underlying the findings in their manuscript fully available?</p>
<p>The <ext-link ext-link-type="uri" xlink:href="http://www.plosone.org/static/policies.action#sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.<!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: No</p>
<p>********** </p>
<p><!-- <font color="black"> -->4. Is the manuscript presented in an intelligible fashion and written in standard English?</p>
<p>PLOS ONE does not copyedit accepted manuscripts, so the language in submitted articles must be clear, correct, and unambiguous. Any typographical or grammatical errors should be corrected at revision, so please note any specific errors here.<!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>********** </p>
<p><!-- <font color="black"> -->5. Review Comments to the Author</p>
<p>Please use the space provided to explain your answers to the questions above. You may also include additional comments for the author, including concerns about dual publication, research ethics, or publication ethics. (Please upload your review as an attachment if it exceeds 20,000 characters)<!-- </font> --></p>
<p>Reviewer #1: The paper analyzes the reason why the recently proposed principal component analysis (PCA) and tensor decomposition (TD)-based unsupervised feature extraction (FE) has often outperformed these statistical test-based methods in the context of projection pursuit that was proposed a long time ago. Some findings in this paper rationalize the success of PCA- and TD-based unsupervised FE for the first time. I have the following suggestions for this manuscript. Other comments can see the attached file.</p>
<p>Reviewer #2: General Comments:</p>
<p>Is the paper new, technically correct, and relevant?</p>
<p>Yes, the paper is new and technically sounds. Results somehow does support the methodology, but needed to be more cleared by the author in case of properties of the data.</p>
<p>Is the paper well organized?</p>
<p>The paper is properly organized, good literature review, suitable motivation and clear explanation on results are positive points to that.</p>
<p>Is the abstract concise?</p>
<p>Yes, but I think it needs to be rephrased after revision to add some comments about any artifacts or negative points in the method, if exist.</p>
<p>Is the introduction motivating?</p>
<p>Yes, Introduction section is motivating.</p>
<p>Are the methodology, results, and conclusions completely developed?</p>
<p>No, they need to be modified and developed according to the technical comments.</p>
<p>Are there language, mathematics, reference, or style errors? There is no mathematical, reference or style error.</p>
<p>Technical Comments:</p>
<p>Are the codes available for this research? As I found, there is no code available for this study, e. g. in Github. If the authors could make the codes available, the manuscript could be much better evaluated, not only for reviewers, but also for possible readers. When it is not possible to upload the code for public access, such as in Github, could they be provided for reviewer for better assessment of the study?</p>
<p>The study is comprehensive and requires large time to be read carefully and being reviewed. The theoretical background has been well explained in details, and the experiments and related models are presented and the algorithm in Fig. 1 is also well presented. I think more explanation about the steps and the parameters in Fig. 1 is required.</p>
<p>The result comparison parts are well organized and presented. The display way is good. But quantitative evaluation is somehow too much that one can get lost in that. I think it would be better that you add more explanation to that.</p>
<p>How did you evaluate the final result? How did you consider to finally selection a methodology for the most complicate problem?</p>
<p>What about when the models are more complex?</p>
<p>The introduction section is a nice one. It is architected very beautifully, while written fully academic and comprehend. I assume that any change in the introduction section is not necessary, but one of the important tasks after publishing a study is to increase its chance to be seen by the most possible number of researchers, so I would like to give two recommendations. First, to get your published study in the list of searched for papers based on keywords, I propose to increase variety of your keywords. In my viewpoint, they do not cover the whole topic of the study and are not widely searched words. I propose to add at least the keyword “data analysis”. Second, one of the methods in the publisher’s website that brings a publication on to the researchers, is based on the similar publications that they have read before. So, the more you cite similar publication, the more the chance that the search engine in the publisher website propose your paper to the researcher. Besides of that, it will also complete your introduction section. As another advantage, it rises new ideas to the researchers by combining various methods, or resolving drawback of one seen paper by reading the similar one, or extending the methodology to a fully automatic one. So, based on these points, I would like to ask to cite to the following similar publication in the manuscript which used PCA and feature selection for deep learning, but in different field of study. The first proposed publication is: Shahbazi, A., Soleimani Monfared, M., Thiruchelvam, V., Ka Fei, T., Babasafari, A.A., (2020). Integration of knowledge-based seismic inversion and sedimentological investigations for heterogeneous reservoir. Journal of Asian Earth Sciences. The second publication for citation is: Khayer, K., Kahoo, A.R., Soleimani Monfared, M., Tokhmechi, B., and Kavousi, K., (2022). Target-Oriented Fusion of Attributes in Data Level for Salt Dome Geobody Delineation in Seismic Data. Natural resource research, and the other publication could be: Khayer, K., Kahoo, A.R., Soleimani Monfared, M., and Kavouosi, K., (2022). Combination of seismic attributes using graph-based methods to identify the salt dome boundary. Journal of Petroleum Science and Engineering. 215, Part A, 110625,</p>
<p>The abstract focusses mainly on the general problem and ignores the other items of the abstract such as the methodology, good introduction, results and conclusion.</p>
<p>The authors should explain what limitations did they find out about the proposed method.</p>
<p>Best Regard</p>
<p>********** </p>
<p><!-- <font color="black"> -->6. PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.<!-- </font> --></p>
<p>Reviewer #1: No</p>
<p>Reviewer #2: No</p>
<p>**********</p>
<p>[NOTE: If reviewer comments were submitted as an attachment file, they will be attached to this email and accessible via the submission site. Please log into your account, locate the manuscript record, and check for the action link "View Attachments". If this link does not appear, there are no attachment files.]</p>
<p>While revising your submission, please upload your figure files to the Preflight Analysis and Conversion Engine (PACE) digital diagnostic tool, <ext-link ext-link-type="uri" xlink:href="https://pacev2.apexcovantage.com/" xlink:type="simple">https://pacev2.apexcovantage.com/</ext-link>. PACE helps ensure that figures meet PLOS requirements. To use PACE, you must first register as a user. Registration is free. Then, login and navigate to the UPLOAD tab, where you will find detailed instructions on how to use the tool. If you encounter any issues or have any questions when using PACE, please email PLOS at <email xlink:type="simple">figures@plos.org</email>. Please note that Supporting Information files do not need this step.<!-- </div> --></p>
<supplementary-material id="pone.0275472.s001" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.s001" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">PONE-D-22-20332_reviewer.docx</named-content></p>
</caption>
</supplementary-material>
<supplementary-material id="pone.0275472.s002" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.s002" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">Comments-PONE-D-22-20332.pdf</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="author-comment" id="pone.0275472.r002">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0275472.r002</article-id>
<title-group>
<article-title>Author response to Decision Letter 0</article-title>
</title-group>
<related-object document-id="10.1371/journal.pone.0275472" document-id-type="doi" document-type="peer-reviewed-article" id="rel-obj002" link-type="rebutted-decision-letter" object-id="10.1371/journal.pone.0275472.r001" object-id-type="doi" object-type="decision-letter"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="author-response-date">6 Sep 2022</named-content>
</p>
<p>See attached</p>
<supplementary-material id="pone.0275472.s003" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pone.0275472.s003" xlink:type="simple">
<label>Attachment</label>
<caption>
<p>Submitted filename: <named-content content-type="submitted-filename">Replies_to_reviewers.docx</named-content></p>
</caption>
</supplementary-material>
</body>
</sub-article>
<sub-article article-type="aggregated-review-documents" id="pone.0275472.r003" specific-use="decision-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0275472.r003</article-id>
<title-group>
<article-title>Decision Letter 1</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Chen</surname>
<given-names>Chi-Hua</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2022</copyright-year>
<copyright-holder>Chi-Hua Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pone.0275472" document-id-type="doi" document-type="article" id="rel-obj003" link-type="peer-reviewed-article"/>
<custom-meta-group>
<custom-meta>
<meta-name>Submission Version</meta-name>
<meta-value>1</meta-value>
</custom-meta>
</custom-meta-group>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">19 Sep 2022</named-content>
</p>
<p>Projection in genomic analysis: A theoretical basis to rationalize tensor decomposition and principal component analysis as feature selection tools</p>
<p>PONE-D-22-20332R1</p>
<p>Dear Dr. Taguchi,</p>
<p>We’re pleased to inform you that your manuscript has been judged scientifically suitable for publication and will be formally accepted for publication once it meets all outstanding technical requirements.</p>
<p>Within one week, you’ll receive an e-mail detailing the required amendments. When these have been addressed, you’ll receive a formal acceptance letter and your manuscript will be scheduled for publication.</p>
<p>An invoice for payment will follow shortly after the formal acceptance. To ensure an efficient process, please log into Editorial Manager at <ext-link ext-link-type="uri" xlink:href="http://www.editorialmanager.com/pone/" xlink:type="simple">http://www.editorialmanager.com/pone/</ext-link>, click the 'Update My Information' link at the top of the page, and double check that your user information is up-to-date. If you have any billing related questions, please contact our Author Billing department directly at <email xlink:type="simple">authorbilling@plos.org</email>.</p>
<p>If your institution or institutions have a press office, please notify them about your upcoming paper to help maximize its impact. If they’ll be preparing press materials, please inform our press team as soon as possible -- no later than 48 hours after receiving the formal acceptance. Your manuscript will remain under strict press embargo until 2 pm Eastern Time on the date of publication. For more information, please contact <email xlink:type="simple">onepress@plos.org</email>.</p>
<p>Kind regards,</p>
<p>Chi-Hua Chen, Ph.D.</p>
<p>Academic Editor</p>
<p>PLOS ONE</p>
<p>Additional Editor Comments (optional):</p>
<p>Reviewers' comments:</p>
<p>Reviewer's Responses to Questions</p>
<p><!-- <font color="black"> --><bold>Comments to the Author</bold></p>
<p>1. If the authors have adequately addressed your comments raised in a previous round of review and you feel that this manuscript is now acceptable for publication, you may indicate that here to bypass the “Comments to the Author” section, enter your conflict of interest statement in the “Confidential to Editor” section, and submit your "Accept" recommendation.<!-- </font> --></p>
<p>Reviewer #1: All comments have been addressed</p>
<p>Reviewer #2: All comments have been addressed</p>
<p>**********</p>
<p><!-- <font color="black"> -->2. Is the manuscript technically sound, and do the data support the conclusions?</p>
<p>The manuscript must describe a technically sound piece of scientific research with data that supports the conclusions. Experiments must have been conducted rigorously, with appropriate controls, replication, and sample sizes. The conclusions must be drawn appropriately based on the data presented. <!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p><!-- <font color="black"> -->3. Has the statistical analysis been performed appropriately and rigorously? <!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p><!-- <font color="black"> -->4. Have the authors made all data underlying the findings in their manuscript fully available?</p>
<p>The <ext-link ext-link-type="uri" xlink:href="http://www.plosone.org/static/policies.action#sharing" xlink:type="simple">PLOS Data policy</ext-link> requires authors to make all data underlying the findings described in their manuscript fully available without restriction, with rare exception (please refer to the Data Availability Statement in the manuscript PDF file). The data should be provided as part of the manuscript or its supporting information, or deposited to a public repository. For example, in addition to summary statistics, the data points behind means, medians and variance measures should be available. If there are restrictions on publicly sharing data—e.g. participant privacy or use of data from a third party—those must be specified.<!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p><!-- <font color="black"> -->5. Is the manuscript presented in an intelligible fashion and written in standard English?</p>
<p>PLOS ONE does not copyedit accepted manuscripts, so the language in submitted articles must be clear, correct, and unambiguous. Any typographical or grammatical errors should be corrected at revision, so please note any specific errors here.<!-- </font> --></p>
<p>Reviewer #1: Yes</p>
<p>Reviewer #2: Yes</p>
<p>**********</p>
<p><!-- <font color="black"> -->6. Review Comments to the Author</p>
<p>Please use the space provided to explain your answers to the questions above. You may also include additional comments for the author, including concerns about dual publication, research ethics, or publication ethics. (Please upload your review as an attachment if it exceeds 20,000 characters)<!-- </font> --></p>
<p>Reviewer #1: This manuscript has enriched the content of the article and enhanced the readability of the article through modification, but there are still some small problems.</p>
<p>1. It is suggested that the paragraphs of the full paper should be aligned at both ends, which may make the article look more beautiful.</p>
<p>2. In line 201 on page 8, u3k does not exist in (26).</p>
<p>3. In line 284 on page 11, a sentence uses two verbs, “P-values were attributed to genes as... 155 genes associated with corrected P-values less than 0.01 were selected, bi is expected to play a role of u5i in eq. (46).”</p>
<p>4. Please check the references carefully. For example, reference [3], [10], [12], [14], [15], [16], and [19] etc.</p>
<p>Reviewer #2: Dear Authors;</p>
<p>I have read your response and edited manuscript carefully and I was pleased with your answers and the way of developing the research and the manuscript.</p>
<p>So, I have no further comment for you.</p>
<p>Best Regards</p>
<p>**********</p>
<p><!-- <font color="black"> -->7. PLOS authors have the option to publish the peer review history of their article (<ext-link ext-link-type="uri" xlink:href="https://journals.plos.org/plosone/s/editorial-and-peer-review-process#loc-peer-review-history" xlink:type="simple">what does this mean?</ext-link>). If published, this will include your full peer review and any attached files.</p>
<p>If you choose “no”, your identity will remain anonymous but your review may still be made public.</p>
<p><bold>Do you want your identity to be public for this peer review?</bold> For information about this choice, including consent withdrawal, please see our <ext-link ext-link-type="uri" xlink:href="https://www.plos.org/privacy-policy" xlink:type="simple">Privacy Policy</ext-link>.<!-- </font> --></p>
<p>Reviewer #1: No</p>
<p>Reviewer #2: No</p>
<p>**********</p>
</body>
</sub-article>
<sub-article article-type="editor-report" id="pone.0275472.r004" specific-use="acceptance-letter">
<front-stub>
<article-id pub-id-type="doi">10.1371/journal.pone.0275472.r004</article-id>
<title-group>
<article-title>Acceptance letter</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name name-style="western">
<surname>Chen</surname>
<given-names>Chi-Hua</given-names>
</name>
<role>Academic Editor</role>
</contrib>
</contrib-group>
<permissions>
<copyright-year>2022</copyright-year>
<copyright-holder>Chi-Hua Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
</license>
</permissions>
<related-object document-id="10.1371/journal.pone.0275472" document-id-type="doi" document-type="article" id="rel-obj004" link-type="peer-reviewed-article"/>
</front-stub>
<body>
<p>
<named-content content-type="letter-date">20 Sep 2022</named-content>
</p>
<p>PONE-D-22-20332R1 </p>
<p>Projection in genomic analysis: A theoretical basis to rationalize tensor decomposition and principal component analysis as feature selection tools </p>
<p>Dear Dr. Taguchi:</p>
<p>I'm pleased to inform you that your manuscript has been deemed suitable for publication in PLOS ONE. Congratulations! Your manuscript is now with our production department. </p>
<p>If your institution or institutions have a press office, please let them know about your upcoming paper now to help maximize its impact. If they'll be preparing press materials, please inform our press team within the next 48 hours. Your manuscript will remain under strict press embargo until 2 pm Eastern Time on the date of publication. For more information please contact <email xlink:type="simple">onepress@plos.org</email>.</p>
<p>If we can help with anything else, please email us at <email xlink:type="simple">plosone@plos.org</email>. </p>
<p>Thank you for submitting your work to PLOS ONE and supporting open access. </p>
<p>Kind regards, </p>
<p>PLOS ONE Editorial Office Staff</p>
<p>on behalf of</p>
<p>Professor Chi-Hua Chen </p>
<p>Academic Editor</p>
<p>PLOS ONE</p>
</body>
</sub-article>
</article>