<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "http://jats.nlm.nih.gov/publishing/1.3/JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<processing-meta>
<custom-meta-group content-type="composition">
<custom-meta specific-use="newgen" xlink:href="https://www.newgen.co/">
<meta-name>Composition Vendor</meta-name>
<meta-value>Newgen KnowledgeWorks (P) Ltd.</meta-value>
</custom-meta>
</custom-meta-group>
</processing-meta>
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Comput Biol</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">ploscomp</journal-id>
<journal-title-group>
<journal-title>PLOS Computational Biology</journal-title>
</journal-title-group>
<issn pub-type="ppub">1553-734X</issn>
<issn pub-type="epub">1553-7358</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pcbi.1014500</article-id>
<article-id pub-id-type="publisher-id">PCOMPBIOL-D-25-01772</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Software</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Oncology</subject><subj-group><subject>Cancers and neoplasms</subject><subj-group><subject>Lung and intrathoracic tumors</subject><subj-group><subject>Mesothelioma</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Cell biology</subject><subj-group><subject>Cellular types</subject><subj-group><subject>Animal cells</subject><subj-group><subject>Blood cells</subject><subj-group><subject>White blood cells</subject><subj-group><subject>Neutrophils</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Cell biology</subject><subj-group><subject>Cellular types</subject><subj-group><subject>Animal cells</subject><subj-group><subject>Immune cells</subject><subj-group><subject>White blood cells</subject><subj-group><subject>Neutrophils</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune cells</subject><subj-group><subject>White blood cells</subject><subj-group><subject>Neutrophils</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune cells</subject><subj-group><subject>White blood cells</subject><subj-group><subject>Neutrophils</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Computational biology</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Genome annotation</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Genome annotation</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Clinical medicine</subject><subj-group><subject>Clinical immunology</subject><subj-group><subject>Immunotherapy</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Clinical immunology</subject><subj-group><subject>Immunotherapy</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Clinical immunology</subject><subj-group><subject>Immunotherapy</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune response</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Immunology</subject><subj-group><subject>Immune response</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Biochemistry</subject><subj-group><subject>Metabolism</subject><subj-group><subject>Carbohydrate metabolism</subject><subj-group><subject>Glucose metabolism</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Computational biology</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Gene ontologies</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Genetics</subject><subj-group><subject>Genomics</subject><subj-group><subject>Genome analysis</subject><subj-group><subject>Gene ontologies</subject></subj-group></subj-group></subj-group></subj-group></subj-group></article-categories>
<title-group>
<article-title>GeneInsight: Condensing gene set knowledge via language models</article-title>
<alt-title alt-title-type="running-head">Gene set analysis via language models</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-2549-3601</contrib-id>
<name name-style="western">
<surname>Chin</surname>
<given-names>Wee Loong</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role content-type="http://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role content-type="http://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/validation/">Validation</role>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>Chen</surname>
<given-names>Kevin</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/software/">Software</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0138-2691</contrib-id>
<name name-style="western">
<surname>Lassmann</surname>
<given-names>Timo</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role content-type="http://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role content-type="http://credit.niso.org/contributor-roles/resources/">Resources</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
</contrib-group>
<aff id="aff001"><label>1</label> <addr-line>National Centre for Asbestos-Related Diseases, University of Western Australia, Perth, Western Australia, Australia</addr-line></aff>
<aff id="aff002"><label>2</label> <addr-line>Department of Medical Oncology, Sir Charles Gairdner Hospital, Perth, Western Australia, Australia</addr-line></aff>
<aff id="aff003"><label>3</label> <addr-line>The Kids Research Institute Australia, University of Western Australia, Nedlands, Western Australia, Australia</addr-line></aff>
<contrib-group>
<contrib contrib-type="editor" xlink:type="simple">
<name name-style="western">
<surname>Ziemann</surname>
<given-names>Mark</given-names>
</name>
<role>Editor</role>
<xref ref-type="aff" rid="edit1"/></contrib>
</contrib-group>
<aff id="edit1"><addr-line>Burnet Institute, AUSTRALIA</addr-line></aff>
<author-notes>
<fn fn-type="conflict" id="coi001">
<p>The authors have declared that no competing interests exist.</p>
</fn>
<corresp id="cor001">* E-mail: <email xlink:type="simple">melvin.chin@uwa.edu.au</email></corresp>
</author-notes>
<pub-date pub-type="epub"><day>5</day><month>8</month><year>2026</year></pub-date>
<pub-date pub-type="collection"><month>8</month><year>2026</year></pub-date>
<volume>22</volume>
<issue>8</issue>
<elocation-id>e1014500</elocation-id>
<history>
<date date-type="received"><day>2</day><month>9</month><year>2025</year></date>
<date date-type="accepted"><day>26</day><month>6</month><year>2026</year></date>
</history>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Chin et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pcbi.1014500"/>
<abstract>
<p>Gene set analysis often returns extensive annotations from multiple sources, requiring manual effort to identify coherent biological themes. We developed GeneInsight, an AI-powered tool that automates this by retrieving functional annotations from STRING-DB, clustering semantically related terms using sentence embeddings, and generating thematic summaries via large language model prompting. This enables researchers to identify biological themes that may be obscured when annotation sources are examined separately.</p>
</abstract>
<funding-group>
<award-group id="award001">
<funding-source><institution>Cancer Research Trust</institution>
</funding-source><principal-award-recipient><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0138-2691</contrib-id><name name-style="western">
<surname>Lassmann</surname><given-names>Timo</given-names></name></principal-award-recipient></award-group>
<award-group id="award002">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100023840</institution-id>
<institution>Feilman Foundation</institution>
</institution-wrap>
</funding-source><principal-award-recipient><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0138-2691</contrib-id><name name-style="western">
<surname>Lassmann</surname><given-names>Timo</given-names></name></principal-award-recipient></award-group>
<award-group id="award003">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="funder-id">http://dx.doi.org/10.13039/501100020265</institution-id>
<institution>Stan Perron Charitable Foundation</institution>
</institution-wrap>
</funding-source><principal-award-recipient><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0138-2691</contrib-id><name name-style="western">
<surname>Lassmann</surname><given-names>Timo</given-names></name></principal-award-recipient></award-group>
<funding-statement>This work was supported by a collaborative cancer research grant from the Cancer Research Trust (“Enabling advanced single cell cancer genomics in Western Australia”), awarded to T.L. T.L. was additionally supported by fellowships from the Feilman Foundation and the Stan Perron Charitable Foundation. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<page-count count="9"/>
</counts>
<custom-meta-group>
<custom-meta id="data-availability">
<meta-name>Data Availability</meta-name>
<meta-value>All code used for running experiments, plotting is available on a GitHub repository at <ext-link ext-link-type="uri" xlink:href="https://github.com/wlchin/geneinsight/" xlink:type="simple">https://github.com/wlchin/geneinsight/</ext-link>.</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Gene set interpretation is a fundamental task in functional genomics, where researchers must derive biological insights from lists of genes identified in high-throughput experiments. Current approaches utilise statistical enrichment methods that query predefined functional databases, such as Gene Ontology [<xref ref-type="bibr" rid="pcbi.1014500.ref001">1</xref>] and KEGG pathways [<xref ref-type="bibr" rid="pcbi.1014500.ref002">2</xref>], to identify overrepresented biological processes [<xref ref-type="bibr" rid="pcbi.1014500.ref003">3</xref>,<xref ref-type="bibr" rid="pcbi.1014500.ref004">4</xref>]. Although powerful, current methods yield fragmented outputs, such as lists of enriched terms from various ontologies, leaving researchers to manually integrate these results to achieve functional insights, a process that is both inefficient and error-prone.</p>
<p>Exploratory gene set analysis has become increasingly challenging as the volume of annotated datasets grows. Researchers compare their findings not only with gene ontology term enrichments but also with signatures from knockdown experiments and diverse resources such as LINCS (Library of Integrated Network-based Cellular Signatures) [<xref ref-type="bibr" rid="pcbi.1014500.ref005">5</xref>] and the STRING database (STRING-DB) [<xref ref-type="bibr" rid="pcbi.1014500.ref006">6</xref>]. The main issue is that enrichment analysis tests for the over-representation of genes associated with specific terms. When gene lists overlap, the same genes often appear under many different functional terms. As a result, enrichment analysis can return many distinct-sounding terms that all point to the same underlying biology, making interpretation more difficult.</p>
<p>However, the challenge extends beyond simply removing duplicate information. Simplistic filtering of overlapping gene sets would obscure important biological relationships that only become apparent when analysing genes across the diverse resources mentioned above. These relationships often represent biological processes that bridge multiple databases and reveal insights not captured by any single resource. Consequently, manually curating enrichment outputs to identify both redundancies and meaningful biological patterns is not only error-prone and prohibitively time-consuming but also risks overlooking crucial biological connections.</p>
<p>We hypothesise that recently introduced topic modelling techniques can address this problem. By analysing how terms co-occur across texts, topic modelling reveals underlying themes without requiring predefined categories. These unsupervised statistical methods include Latent Dirichlet Allocation (LDA) [<xref ref-type="bibr" rid="pcbi.1014500.ref007">7</xref>] and Non-negative Matrix Factorization (NMF) [<xref ref-type="bibr" rid="pcbi.1014500.ref008">8</xref>], which identify recurring patterns in document collections. The resulting latent themes represent collections of related terms that frequently appear together and may correspond to biological processes, pathways, or functional modules not explicitly defined in current annotation databases.</p>
<p>Recent advances in large language models [<xref ref-type="bibr" rid="pcbi.1014500.ref009">9</xref>] (LLMs) create powerful new opportunities for gene set interpretation. LLMs have demonstrated remarkable capabilities in contextual understanding and natural language generation, potentially enabling automated synthesis of distributed biological knowledge. Several studies have explored this direction. Hu et al. [<xref ref-type="bibr" rid="pcbi.1014500.ref010">10</xref>] evaluated LLMs as “compressed databases” of biological literature, querying internal model weights to infer gene set function with external search used for post-hoc verification. GeneAgent [<xref ref-type="bibr" rid="pcbi.1014500.ref011">11</xref>] generates functional hypotheses from internal knowledge before fact-checking extracted claims against databases. llm2geneset [<xref ref-type="bibr" rid="pcbi.1014500.ref012">12</xref>] uses internal knowledge to dynamically construct gene set categories, with statistical testing applied afterwards.</p>
<p>Here we present GeneInsight, a tool that integrates LLMs with topic modelling to automate gene set interpretation. Our tool aggregates gene-specific annotations from the STRING database, applies topic modelling to identify coherent biological themes, and employs LLM-based summarisation to generate contextual interpretations of these themes. While existing LLM-based tools query the model’s internal knowledge to infer gene function, GeneInsight takes a different approach. We retrieve annotations directly from STRING and combine topic modelling with LLM summarisation to distil these into coherent biological themes, grounding interpretations in expert-curated database content.</p>
</sec>
<sec id="sec002" sec-type="materials|methods">
<title>Design and implementation</title>
<p>GeneInsight uses a two-stage approach to extract and organise biological information from gene sets (<xref ref-type="fig" rid="pcbi.1014500.g001">Fig 1a</xref>). In the biological theme generation stage, the system collects functional annotations from the STRING database for each input gene, creating a collection of gene-specific descriptions. This textual corpus is subjected to cluster-based topic modelling, which groups similar annotations into clusters (topics) and identifies key terms for each cluster. An LLM then converts representative annotations from each cluster into interpretable biological themes. Themes are then prioritised using an overlap-ratio threshold combined with an empirical p-value calibrated against a size-stratified random-query null (<xref ref-type="supplementary-material" rid="pcbi.1014500.s005">S1 File</xref>, Materials and Methods).</p>
<fig id="pcbi.1014500.g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1014500.g001</object-id><label>Fig 1</label><caption><title>GeneInsight system architecture and workflow.</title><p><bold>(a)</bold> Schematic overview of the GeneInsight framework. The pipeline processes gene sets through two sequential stages: (1) Theme generation and (2) summarisation generates comprehensive reports with interactive visualisations. <bold>(b)</bold> Web interface components with components referenced in <bold>(a)</bold>. <bold>(c)</bold> Performance metrics used in benchmarking showing theme diversity using average pairwise distance, set-level alignment using soft cardinality overlap and summarisation accuracy using top-k metric. RAG, Retrieval augmented generation; LLM, Large language model; HTML, Hypertext markup language. Created in BioRender. Chin, W. (2026) <ext-link ext-link-type="uri" xlink:href="https://BioRender.com/u37x4" xlink:type="simple">https://BioRender.com/u37x4</ext-link>ls.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.g001" xlink:type="simple"/></fig>
<p>The second summarisation stage begins with another round of cluster-based topic modelling to identify key themes. This approach refines these enriched themes by measuring how consistently they appear as cluster representatives across multiple runs of topic modelling. The software then extracts the final summary by selecting themes to include based on user-defined length preferences. A large language model creates a hierarchical summary where major biological themes appear as main headings with related subheadings grouped beneath them. The final interactive HTML report (<xref ref-type="fig" rid="pcbi.1014500.g001">Fig 1b</xref>) links theme descriptions to their corresponding gene annotations. This integration enables researchers to easily navigate between overarching biological processes and their specific components. Moreover, every theme is directly tied to the original STRING-derived annotation outputs, allowing users to trace each summary theme back to its supporting gene-level descriptions and source annotations.</p>
</sec>
<sec id="sec003" sec-type="results">
<title>Results</title>
<p>We first characterised GeneInsight’s theme-prioritisation stage by comparing its retained themes with the STRING database functional enrichment application programming interface (API). Using identical underlying gene-level information and 1,000 Molecular Signatures Database (MSigDB) [<xref ref-type="bibr" rid="pcbi.1014500.ref014">14</xref>] gene sets, this comparison directly assessed our tool’s effectiveness in identifying important biological themes. We measured performance through metrics (<xref ref-type="fig" rid="pcbi.1014500.g001">Fig 1c</xref>) that evaluated both the diversity of identified concepts and the degree of overlap between methods.</p>
<p>GeneInsight consistently identified a larger number of enriched gene sets than the STRING database functional enrichment API across all filtering thresholds examined (<xref ref-type="fig" rid="pcbi.1014500.g002">Figs 2a</xref>, <xref ref-type="supplementary-material" rid="pcbi.1014500.s005">S1</xref>). While the two methods show strong positive correlation (r = 0.69-0.87), GeneInsight retained 2–3 times more terms than the STRING-DB API at every matched setting examined (<xref ref-type="supplementary-material" rid="pcbi.1014500.s001">S1 Fig</xref>), indicating that topic modelling and LLM summarisation capture biological themes not identified by standard term-by-term enrichment (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2b</xref>).</p>
<fig id="pcbi.1014500.g002" position="float"><object-id pub-id-type="doi">10.1371/journal.pcbi.1014500.g002</object-id><label>Fig 2</label><caption><title>Evaluation of GeneInsight and STRING-DB.</title><p><bold>(a)</bold> Scatter plot showing the number of terms identified by GeneInsight versus STRING-DB API across 1,000 MSigDB gene sets. <bold>(b)</bold> Overlap of retained terms between GeneInsight and STRING-DB at various semantic similarity thresholds. <bold>(c)</bold> Average distance (AdPD) measurements comparing GeneInsight and STRING-DB enrichment results. <bold>(d)</bold> Top-k similarity scores for enriched gene sets across different user-defined summary levels (25, 50, 75, and 100 terms). <bold>(e)</bold> Top-k scores calculated using MoverScore for semantic similarity assessment. <bold>(f)</bold> Pearson correlation coefficients between cosine similarity and MoverScores at different levels of user-defined summarisation. <bold>(g)</bold> Top-k cosine similarity scores plotted against varying input corpus sizes (ranging from &lt;500 to &gt;3000 terms).</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.g002" xlink:type="simple"/></fig>
<p>To evaluate the conceptual breadth of identified terms, we measured semantic diversity using average pairwise distance between terms. This metric quantifies how conceptually distinct each term is from all others in the set, with higher values indicating coverage of a broader range of biological concepts rather than redundant or closely related processes. GeneInsight-derived terms demonstrated significantly greater semantic diversity compared to STRING-DB terms (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2c</xref>), indicating their ability to capture a wider spectrum of biological information.</p>
<p>Next, we assessed the summarisation stage of GeneInsight, which converts prioritised themes into user-defined summaries. To evaluate theme preservation during summarisation at different levels (25, 50, 75, and 100 terms), we used a Top-k semantic similarity metric that focuses on each source term’s strongest matches to measure how well summaries capture essential biological concepts without being diluted by less relevant relationships. This approach identifies the k-nearest semantic neighbours for each term, better handling the imbalance between comprehensive source material and length-constrained summaries. GeneInsight demonstrated robust summarisation of key themes across different final summary counts (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2d</xref>), with the stability of these metrics suggesting that our tool effectively identifies core biological concepts regardless of user-supplied summary length constraints.</p>
<p>To provide an orthogonal validation beyond the cosine similarity metric, which measures the directional similarity between text representations, we employed Earth Mover’s Distance [<xref ref-type="bibr" rid="pcbi.1014500.ref015">15</xref>] (MoverScore) (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2e</xref>). Like cosine distance, this complementary approach assesses semantic similarity by measuring the minimum cost required to transform one text into another in the embedding space, capturing different aspects of semantic relationships. All Top-k values demonstrated MoverScores greater than 0.5, indicating that each extracted theme successfully captured substantial semantic content from the original enriched gene sets. The results showed consistently high correlation values (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2f</xref>) between MoverScores and cosine similarity scores (0.93 - 0.94) across all theme configurations, confirming the robustness of our semantic similarity assessments.</p>
<p>We also confirmed that summarisation performance remained stable across documents of varying sizes (from &lt;500 to &gt;3000 terms), demonstrating that our tool’s summarisation capability is independent of input size (<xref ref-type="fig" rid="pcbi.1014500.g002">Fig 2g</xref>).</p>
<p>Next, we applied GeneInsight to an RNA-seq dataset from a murine model of mesothelioma treated with immunotherapy, specifically focusing on genes differentially expressed in treatment responders [<xref ref-type="bibr" rid="pcbi.1014500.ref013">13</xref>]. GeneInsight prioritised biological themes centred around Type I interferon signalling and monocyte-macrophage axis activation (<xref ref-type="supplementary-material" rid="pcbi.1014500.s002">S1 Data</xref>) which were not detected through standard Gene Ontology (GO) enrichment analysis. The importance of Type I interferon signalling was subsequently validated through mouse models using antibody-mediated interferon blockade experiments, which confirmed the functional relevance of these pathways in treatment response. The monocyte involvement themes identified by our tool were further validated through single-cell analysis, confirming GeneInsight’s capacity to identify biologically relevant signatures that effectively bridge bulk and single-cell approaches.</p>
<p>We then evaluated GeneInsight on a multi-omics dataset from the DREAM study [<xref ref-type="bibr" rid="pcbi.1014500.ref016">16</xref>,<xref ref-type="bibr" rid="pcbi.1014500.ref017">17</xref>], which included mesothelioma patients undergoing chemoimmunotherapy treatment. Using responder-specific genes identified through a NanoString panel and bulk RNA-seq time course data, GeneInsight successfully extracted stem cell-like signatures in T-cells (<xref ref-type="supplementary-material" rid="pcbi.1014500.s003">S2 Data</xref>) that were independently confirmed through orthogonal single-cell validation. In the original analysis, these stem-like signatures were only discovered after weeks of analysis involving differential abundance testing, manual inspection of CD8<sup>+</sup> T cell subclusters, differential expression analysis and marker analysis comparing responders and non-responder populations [<xref ref-type="bibr" rid="pcbi.1014500.ref017">17</xref>]. GeneInsight streamlined this process by automatically identifying these key biological themes in a single analysis of 30 minutes, demonstrating how it can reduce analytical complexity and accelerate hypothesis generation from multi-omic datasets.</p>
<p>Finally, we evaluated GeneInsight using a published gene set [<xref ref-type="bibr" rid="pcbi.1014500.ref018">18</xref>] associated with the transcriptional response of neutrophils to Francisella tularensis infection. GeneInsight identified distinct biological themes from differentially expressed genes involved in glucose metabolism (<xref ref-type="supplementary-material" rid="pcbi.1014500.s004">S3 Data</xref>), suggesting a metabolic shift that aligns with neutrophil functional alterations during infection. This metabolic reprogramming pattern, particularly involving key glycolytic regulators such as PFKL, was validated in a follow-up study [<xref ref-type="bibr" rid="pcbi.1014500.ref019">19</xref>] using the same dataset 9 years later. This case demonstrates GeneInsight’s capacity to derive novel biological insights from existing datasets, potentially accelerating discovery timelines from functional genomics data.</p>
</sec>
<sec id="sec004">
<title>Availability and future directions</title>
<p>GeneInsight is freely available as an open-source Python package under the MIT License at <ext-link ext-link-type="uri" xlink:href="https://github.com/wlchin/geneinsight" xlink:type="simple">https://github.com/wlchin/geneinsight</ext-link>. The complete source code, documentation, installation instructions, and analysis workflows are accessible through this repository. The <xref ref-type="supplementary-material" rid="pcbi.1014500.s005">S1 File</xref> describes the methods associated with these analysis workflows in detail.</p>
<p>Several limitations should be considered when interpreting GeneInsight results. GeneInsight uses empirical p-values, calibrated against a size-stratified random-query null, as ranking scores for prioritising themes. The procedure assumes that the query genes can be treated as a random sample from a user-specified background gene set [<xref ref-type="bibr" rid="pcbi.1014500.ref020">20</xref>–<xref ref-type="bibr" rid="pcbi.1014500.ref022">22</xref>]. In practice, many biological inputs depart from this assumption. Differential-expression lists, curated pathways, STRING-derived neighbourhoods, and STRING-DB functional enrichment outputs often carry experimental, annotation, or network-derived structure that the random-query null does not model [<xref ref-type="bibr" rid="pcbi.1014500.ref006">6</xref>,<xref ref-type="bibr" rid="pcbi.1014500.ref020">20</xref>]. Gene–gene correlation, uneven annotation across genes, and overlap among functional categories can also affect calibration [<xref ref-type="bibr" rid="pcbi.1014500.ref020">20</xref>–<xref ref-type="bibr" rid="pcbi.1014500.ref022">22</xref>]. These are known limitations of over-representation analysis more broadly [<xref ref-type="bibr" rid="pcbi.1014500.ref023">23</xref>], and the empirical procedure used here calibrates the random-sampling component but not these additional sources of structure. We do not claim calibration against alternative nulls such as expression-matched or network-degree-matched randomisation. The empirical p-values should therefore be interpreted as a calibrated ranking score for prioritising themes, not as significance tests for individual genes or pathways.</p>
<p>GeneInsight also relies on LLM-generated summaries. Although the LLM is constrained to summarise only retrieved STRING-DB terms and the final HTML report links each theme back to its source annotations, hallucination or over-interpretation cannot be eliminated. Users should therefore verify that generated summaries accurately reflect the displayed source annotations and should treat GeneInsight outputs as hypothesis-generating summaries rather than definitive biological conclusions.</p>
<p>Looking ahead, GeneInsight’s algorithm is designed to process text-based annotations regardless of their source. While our current implementation uses the STRING database, the framework can readily incorporate PubMed abstracts, Gene Ontology terms, pathway descriptions from Reactome [<xref ref-type="bibr" rid="pcbi.1014500.ref024">24</xref>], and unpublished private datasets. This source-agnostic design allows GeneInsight to incorporate any text-based annotation source, providing an automated aggregation layer that synthesises diverse annotations into interpretable biological themes.</p>
</sec>
<sec id="sec005" sec-type="supplementary-material">
<title>Supporting information</title>
<supplementary-material id="pcbi.1014500.s001" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s001" xlink:type="simple">
<label>S1 Fig</label>
<caption>
<title>Correlation between GeneInsight and STRING database (STRING-DB) retained gene sets across a range of filtering cut-offs, shown as a sensitivity diagnostic.</title>
<p>The grid shows pairwise comparisons of enriched gene sets identified by GeneInsight (x-axis) versus terms from the STRING database (y-axis) at varying BH-adjusted hypergeometric p-value cut-offs (filtering thresholds, not calibrated FDR estimates). Each subplot represents a different combination of p-value thresholds, with F indicating the BH-adjusted hypergeometric p-value cut-off applied to GeneInsight results and E indicating the STRING-reported FDR threshold. Thresholds range from stringent (0.001) to permissive (0.05). Blue dots represent individual gene sets with the number of enriched terms plotted for each tool. Dashed lines show the linear regression trend. Correlation coefficients (r) are displayed in the upper left of each panel.</p>
<p>(PDF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s002" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s002" xlink:type="simple">
<label>S1 Data</label>
<caption>
<title>Murine mesothelioma immunotherapy response analysis.</title>
<p>HTML reports and raw files from a GeneInsight analysis of the murine mesothelioma immunotherapy response dataset, highlighting biological themes related to Type I interferons and monocyte-macrophage activation.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s003" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s003" xlink:type="simple">
<label>S2 Data</label>
<caption>
<title>DREAM study mesothelioma patient analysis.</title>
<p>HTML reports and raw files from a GeneInsight analysis of the DREAM study mesothelioma patient dataset, highlighting stem cell-like (lymphocyte proliferation and differentiation) signatures in T-cells of responders to chemoimmunotherapy treatment.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s004" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s004" xlink:type="simple">
<label>S3 Data</label>
<caption>
<title>Neutrophil transcriptional response analysis.</title>
<p>HTML reports and raw files from a GeneInsight analysis of neutrophil transcriptional response to Francisella tularensis infection, featuring identified metabolic reprogramming signatures in glucose metabolism.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s005" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s005" xlink:type="simple">
<label>S1 File</label>
<caption>
<title>Analysis workflow methods documentation.</title>
<p>Detailed methods associated with analysis workflows.</p>
<p>(DOCX)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s006" mimetype="application/pdf" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s006" xlink:type="simple">
<label>S2 File</label>
<caption>
<title>LLM prompts.</title>
<p>Complete prompts used for theme generation and summarisation.</p>
<p>(PDF)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s007" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s007" xlink:type="simple">
<label>S4 Data</label>
<caption>
<title>Murine mesothelioma immunotherapy response analysis with overlap filter.</title>
<p>Raw files from a GeneInsight analysis with overlap ratio filter (threshold 0.25) applied to the murine mesothelioma immunotherapy response dataset.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s008" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s008" xlink:type="simple">
<label>S5 Data</label>
<caption>
<title>DREAM study mesothelioma patient analysis with overlap filter.</title>
<p>Raw files from a GeneInsight analysis with overlap ratio filter (threshold 0.25) applied to the DREAM study mesothelioma patient dataset.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
<supplementary-material id="pcbi.1014500.s009" mimetype="application/zip" position="float" xlink:href="info:doi/10.1371/journal.pcbi.1014500.s009" xlink:type="simple">
<label>S6 Data</label>
<caption>
<title>Neutrophil transcriptional response analysis with overlap filter.</title>
<p>Raw files from a GeneInsight analysis with overlap ratio filter (threshold 0.25) applied to the neutrophil transcriptional response dataset.</p>
<p>(ZIP)</p>
</caption>
</supplementary-material>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="pcbi.1014500.ref001"><label>1</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ashburner</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Ball</surname> <given-names>CA</given-names></name>, <name name-style="western"><surname>Blake</surname> <given-names>JA</given-names></name>, <name name-style="western"><surname>Botstein</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Butler</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Cherry</surname> <given-names>JM</given-names></name>. <article-title>Gene Ontology: tool for the unification of biology</article-title>. <source>Nat Genet</source>. <year>2000</year>;<volume>25</volume>:<fpage>25</fpage>–<lpage>9</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/75556" xlink:type="simple">10.1038/75556</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref002"><label>2</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kanehisa</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Goto</surname> <given-names>S</given-names></name>. <article-title>KEGG: kyoto encyclopedia of genes and genomes</article-title>. <source>Nucleic Acids Res</source>. <year>2000</year>;<volume>28</volume>(<issue>1</issue>):<fpage>27</fpage>–<lpage>30</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/28.1.27" xlink:type="simple">10.1093/nar/28.1.27</ext-link></comment> <object-id pub-id-type="pmid">10592173</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref003"><label>3</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wu</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Hu</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Xu</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Chen</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Guo</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Dai</surname> <given-names>Z</given-names></name>. <article-title>clusterProfiler 4.0: A universal enrichment tool for interpreting omics data</article-title>. <source>Innovation (Camb)</source>. <year>2021</year>;<volume>2</volume>:<fpage>100141</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.xinn.2021.100141" xlink:type="simple">10.1016/j.xinn.2021.100141</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Kuleshov</surname> <given-names>MV</given-names></name>, <name name-style="western"><surname>Jones</surname> <given-names>MR</given-names></name>, <name name-style="western"><surname>Rouillard</surname> <given-names>AD</given-names></name>, <name name-style="western"><surname>Fernandez</surname> <given-names>NF</given-names></name>, <name name-style="western"><surname>Duan</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>Z</given-names></name>, <etal>et al</etal>. <article-title>Enrichr: a comprehensive gene set enrichment analysis web server 2016 update</article-title>. <source>Nucleic Acids Res</source>. <year>2016</year>;<volume>44</volume>(W1):W90-7. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkw377" xlink:type="simple">10.1093/nar/gkw377</ext-link></comment> <object-id pub-id-type="pmid">27141961</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Stathias</surname> <given-names>V</given-names></name>, <name name-style="western"><surname>Turner</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Koleti</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Vidovic</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Cooper</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Fazel-Najafabadi</surname> <given-names>M</given-names></name>. <article-title>LINCS Data Portal 2.0: next generation access point for perturbation-response signatures</article-title>. <source>Nucleic Acids Research</source>. <year>2020</year>;<volume>48</volume>:D431–<lpage>9</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkz1023" xlink:type="simple">10.1093/nar/gkz1023</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Szklarczyk</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Kirsch</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Koutrouli</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Nastou</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Mehryary</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Hachilif</surname> <given-names>R</given-names></name>, <etal>et al</etal>. <article-title>The STRING database in 2023: protein-protein association networks and functional enrichment analyses for any sequenced genome of interest</article-title>. <source>Nucleic Acids Res</source>. <year>2023</year>;<volume>51</volume>(D1):D638–<lpage>46</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkac1000" xlink:type="simple">10.1093/nar/gkac1000</ext-link></comment> <object-id pub-id-type="pmid">36370105</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref007"><label>7</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Lewis</surname> <given-names>CM</given-names></name>, <name name-style="western"><surname>Grossetti</surname> <given-names>F</given-names></name>. <article-title>A statistical approach for optimal topic model identification</article-title>. <source>J Mach Learn Res</source>. <year>2022</year>;<volume>23</volume>(<issue>58</issue>):<fpage>1</fpage>–<lpage>20</lpage>.</mixed-citation></ref>
<ref id="pcbi.1014500.ref008"><label>8</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Egger</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Yu</surname> <given-names>J</given-names></name>. <article-title>A topic modeling comparison between LDA, NMF, Top2Vec, and BERTopic to demystify Twitter posts</article-title>. <source>Front Sociol</source>. <year>2022</year>;<volume>7</volume>:<fpage>886498</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fsoc.2022.886498" xlink:type="simple">10.3389/fsoc.2022.886498</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref009"><label>9</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Naveed</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Khan</surname> <given-names>AU</given-names></name>, <name name-style="western"><surname>Qiu</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Saqib</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Anwar</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Usman</surname> <given-names>M</given-names></name>, <etal>et al</etal>. A Comprehensive Overview of Large Language Models. <year>2024</year>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/arXiv.2307.06435" xlink:type="simple">10.48550/arXiv.2307.06435</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref010"><label>10</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Hu</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Alkhairy</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Lee</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Pillich</surname> <given-names>RT</given-names></name>, <name name-style="western"><surname>Fong</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Smith</surname> <given-names>K</given-names></name>, <etal>et al</etal>. <article-title>Evaluation of large language models for discovery of gene set function</article-title>. <source>Nat Methods</source>. <year>2025</year>;<volume>22</volume>(<issue>1</issue>):<fpage>82</fpage>–<lpage>91</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41592-024-02525-x" xlink:type="simple">10.1038/s41592-024-02525-x</ext-link></comment> <object-id pub-id-type="pmid">39609565</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Wang</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Jin</surname> <given-names>Q</given-names></name>, <name name-style="western"><surname>Wei</surname> <given-names>C-H</given-names></name>, <name name-style="western"><surname>Tian</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Lai</surname> <given-names>P-T</given-names></name>, <name name-style="western"><surname>Zhu</surname> <given-names>Q</given-names></name>, <etal>et al</etal>. <article-title>GeneAgent: self-verification language agent for gene-set analysis using domain databases</article-title>. <source>Nat Methods</source>. <year>2025</year>;<volume>22</volume>(<issue>8</issue>):<fpage>1677</fpage>–<lpage>85</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41592-025-02748-6" xlink:type="simple">10.1038/s41592-025-02748-6</ext-link></comment> <object-id pub-id-type="pmid">40721871</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zhu</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>RY</given-names></name>, <name name-style="western"><surname>Wang</surname> <given-names>X</given-names></name>, <name name-style="western"><surname>Azevedo</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Moreno</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Kuhn</surname> <given-names>JA</given-names></name>, <etal>et al</etal>. <article-title>Enhancing gene set overrepresentation analysis with large language models</article-title>. <source>Bioinform Adv</source>. <year>2025</year>;<volume>5</volume>(<issue>1</issue>):vbaf054. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioadv/vbaf054" xlink:type="simple">10.1093/bioadv/vbaf054</ext-link></comment> <object-id pub-id-type="pmid">40401046</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref013"><label>13</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zemek</surname> <given-names>RM</given-names></name>, <name name-style="western"><surname>Chin</surname> <given-names>WL</given-names></name>, <name name-style="western"><surname>Fear</surname> <given-names>VS</given-names></name>, <name name-style="western"><surname>Wylie</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Casey</surname> <given-names>TH</given-names></name>, <name name-style="western"><surname>Forbes</surname> <given-names>C</given-names></name>, <etal>et al</etal>. <article-title>Temporally restricted activation of IFNβ signaling underlies response to immune checkpoint therapy in mice</article-title>. <source>Nat Commun</source>. <year>2022</year>;<volume>13</volume>(<issue>1</issue>):<fpage>4895</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41467-022-32567-8" xlink:type="simple">10.1038/s41467-022-32567-8</ext-link></comment> <object-id pub-id-type="pmid">35986006</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref014"><label>14</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Liberzon</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Birger</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Thorvaldsdóttir</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Ghandi</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Mesirov</surname> <given-names>JP</given-names></name>, <name name-style="western"><surname>Tamayo</surname> <given-names>P</given-names></name>. <article-title>The Molecular Signatures Database (MSigDB) hallmark gene set collection</article-title>. <source>Cell Syst</source>. <year>2015</year>;<volume>1</volume>(<issue>6</issue>):<fpage>417</fpage>–<lpage>25</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.cels.2015.12.004" xlink:type="simple">10.1016/j.cels.2015.12.004</ext-link></comment> <object-id pub-id-type="pmid">26771021</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref015"><label>15</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Zhao</surname> <given-names>W</given-names></name>, <name name-style="western"><surname>Peyrard</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Liu</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Gao</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Meyer</surname> <given-names>CM</given-names></name>, <name name-style="western"><surname>Eger</surname> <given-names>S</given-names></name>. MoverScore: Text Generation Evaluating with Contextualized Embeddings and Earth Mover Distance. <year>2019</year>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/arXiv.1909.02622" xlink:type="simple">10.48550/arXiv.1909.02622</ext-link></comment></mixed-citation></ref>
<ref id="pcbi.1014500.ref016"><label>16</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Nowak</surname> <given-names>AK</given-names></name>, <name name-style="western"><surname>Lesterhuis</surname> <given-names>WJ</given-names></name>, <name name-style="western"><surname>Kok</surname> <given-names>P-S</given-names></name>, <name name-style="western"><surname>Brown</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Hughes</surname> <given-names>BG</given-names></name>, <name name-style="western"><surname>Karikios</surname> <given-names>DJ</given-names></name>, <etal>et al</etal>. <article-title>Durvalumab with first-line chemotherapy in previously untreated malignant pleural mesothelioma (DREAM): a multicentre, single-arm, phase 2 trial with a safety run-in</article-title>. <source>Lancet Oncol</source>. <year>2020</year>;<volume>21</volume>(<issue>9</issue>):<fpage>1213</fpage>–<lpage>23</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/S1470-2045(20)30462-9" xlink:type="simple">10.1016/S1470-2045(20)30462-9</ext-link></comment> <object-id pub-id-type="pmid">32888453</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref017"><label>17</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Chin</surname> <given-names>WL</given-names></name>, <name name-style="western"><surname>Cook</surname> <given-names>AM</given-names></name>, <name name-style="western"><surname>Chee</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Principe</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Hoang</surname> <given-names>TS</given-names></name>, <name name-style="western"><surname>Kidman</surname> <given-names>J</given-names></name>, <etal>et al</etal>. <article-title>Coupling of response biomarkers between tumor and peripheral blood in patients undergoing chemoimmunotherapy</article-title>. <source>Cell Rep Med</source>. <year>2025</year>;<volume>6</volume>(<issue>1</issue>):<fpage>101882</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.xcrm.2024.101882" xlink:type="simple">10.1016/j.xcrm.2024.101882</ext-link></comment> <object-id pub-id-type="pmid">39731918</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref018"><label>18</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schwartz</surname> <given-names>JT</given-names></name>, <name name-style="western"><surname>Bandyopadhyay</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Kobayashi</surname> <given-names>SD</given-names></name>, <name name-style="western"><surname>McCracken</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Whitney</surname> <given-names>AR</given-names></name>, <name name-style="western"><surname>Deleo</surname> <given-names>FR</given-names></name>, <etal>et al</etal>. <article-title>Francisella tularensis alters human neutrophil gene expression: insights into the molecular basis of delayed neutrophil apoptosis</article-title>. <source>J Innate Immun</source>. <year>2013</year>;<volume>5</volume>(<issue>2</issue>):<fpage>124</fpage>–<lpage>36</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1159/000342430" xlink:type="simple">10.1159/000342430</ext-link></comment> <object-id pub-id-type="pmid">22986450</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref019"><label>19</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Krysa</surname> <given-names>SJ</given-names></name>, <name name-style="western"><surname>Allen</surname> <given-names>L-AH</given-names></name>. <article-title>Metabolic Reprogramming Mediates Delayed Apoptosis of Human Neutrophils Infected With Francisella tularensis</article-title>. <source>Front Immunol</source>. <year>2022</year>;<volume>13</volume>:<fpage>836754</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fimmu.2022.836754" xlink:type="simple">10.3389/fimmu.2022.836754</ext-link></comment> <object-id pub-id-type="pmid">35693822</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref020"><label>20</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Geistlinger</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Csaba</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Santarelli</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Ramos</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Schiffer</surname> <given-names>L</given-names></name>, <name name-style="western"><surname>Turaga</surname> <given-names>N</given-names></name>, <etal>et al</etal>. <article-title>Toward a gold standard for benchmarking gene set enrichment analysis</article-title>. <source>Brief Bioinform</source>. <year>2021</year>;<volume>22</volume>(<issue>1</issue>):<fpage>545</fpage>–<lpage>56</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bib/bbz158" xlink:type="simple">10.1093/bib/bbz158</ext-link></comment> <object-id pub-id-type="pmid">32026945</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref021"><label>21</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Cao</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>S</given-names></name>. <article-title>A Bayesian extension of the hypergeometric test for functional enrichment analysis</article-title>. <source>Biometrics</source>. <year>2014</year>;<volume>70</volume>(<issue>1</issue>):<fpage>84</fpage>–<lpage>94</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1111/biom.12122" xlink:type="simple">10.1111/biom.12122</ext-link></comment> <object-id pub-id-type="pmid">24320951</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref022"><label>22</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Goeman</surname> <given-names>JJ</given-names></name>, <name name-style="western"><surname>Bühlmann</surname> <given-names>P</given-names></name>. <article-title>Analyzing gene expression data in terms of gene sets: methodological issues</article-title>. <source>Bioinformatics</source>. <year>2007</year>;<volume>23</volume>(<issue>8</issue>):<fpage>980</fpage>–<lpage>7</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btm051" xlink:type="simple">10.1093/bioinformatics/btm051</ext-link></comment> <object-id pub-id-type="pmid">17303618</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref023"><label>23</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ziemann</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Schroeter</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Bora</surname> <given-names>A</given-names></name>. <article-title>Two subtle problems with overrepresentation analysis</article-title>. <source>Bioinform Adv</source>. <year>2024</year>;<volume>4</volume>(<issue>1</issue>):vbae159. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioadv/vbae159" xlink:type="simple">10.1093/bioadv/vbae159</ext-link></comment> <object-id pub-id-type="pmid">39539946</object-id></mixed-citation></ref>
<ref id="pcbi.1014500.ref024"><label>24</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Milacic</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Beavers</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Conley</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Gong</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Gillespie</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Griss</surname> <given-names>J</given-names></name>, <etal>et al</etal>. <article-title>The Reactome Pathway Knowledgebase 2024</article-title>. <source>Nucleic Acids Res</source>. <year>2024</year>;<volume>52</volume>:D672–<lpage>8</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkad1025" xlink:type="simple">10.1093/nar/gkad1025</ext-link></comment></mixed-citation></ref>
</ref-list>
</back>
</article>