<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "http://jats.nlm.nih.gov/publishing/1.3/JATS-journalpublishing1-3.dtd">
<article article-type="discussion" dtd-version="1.3" xml:lang="en" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<processing-meta>
<custom-meta-group content-type="composition">
<custom-meta specific-use="newgen" xlink:href="https://www.newgen.co/">
<meta-name>Composition Vendor</meta-name>
<meta-value>Newgen KnowledgeWorks (P) Ltd.</meta-value>
</custom-meta>
</custom-meta-group>
</processing-meta>
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">PLoS Med</journal-id>
<journal-id journal-id-type="publisher-id">plos</journal-id>
<journal-id journal-id-type="pmc">plosmed</journal-id>
<journal-title-group>
<journal-title>PLOS Medicine</journal-title>
</journal-title-group>
<issn pub-type="ppub">1549-1277</issn>
<issn pub-type="epub">1549-1676</issn>
<publisher>
<publisher-name>Public Library of Science</publisher-name>
<publisher-loc>San Francisco, CA USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.1371/journal.pmed.1005170</article-id>
<article-id pub-id-type="publisher-id">PMEDICINE-D-26-01836</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Perspective</subject>
</subj-group>
<subj-group subj-group-type="Discipline-v3">
<subject>Computer and information sciences</subject><subj-group><subject>Artificial intelligence</subject></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Decision making</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Decision making</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Decision making</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognition</subject><subj-group><subject>Decision making</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Health care</subject><subj-group><subject>Health care policy</subject><subj-group><subject>Treatment guidelines</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Recreation</subject><subj-group><subject>Games</subject><subj-group><subject>Video games</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Behavior</subject><subj-group><subject>Recreation</subject><subj-group><subject>Games</subject><subj-group><subject>Video games</subject></subj-group></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Neuroscience</subject><subj-group><subject>Cognitive science</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Biology and life sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Social sciences</subject><subj-group><subject>Psychology</subject><subj-group><subject>Cognitive psychology</subject><subj-group><subject>Language</subject></subj-group></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Medicine and health sciences</subject><subj-group><subject>Oncology</subject><subj-group><subject>Cancer treatment</subject></subj-group></subj-group></subj-group><subj-group subj-group-type="Discipline-v3">
<subject>Research and analysis methods</subject><subj-group><subject>Research assessment</subject></subj-group></subj-group></article-categories>
<title-group>
<article-title>How to benchmark medical AI agents</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0123-2239</contrib-id>
<name name-style="western">
<surname>Ruhrberg Estévez</surname>
<given-names>Silas</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0009-0006-6195-9276</contrib-id>
<name name-style="western">
<surname>Ferber</surname>
<given-names>Dyke</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" xlink:type="simple">
<name name-style="western">
<surname>van der Schaar</surname>
<given-names>Mihaela</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff002"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff003"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes" xlink:type="simple">
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3730-5348</contrib-id>
<name name-style="western">
<surname>Kather</surname>
<given-names>Jakob Nikolas</given-names>
</name>
<role content-type="http://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role content-type="http://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role content-type="http://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
<xref ref-type="aff" rid="aff001"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff004"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff005"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff006"><sup>6</sup></xref>
<xref ref-type="corresp" rid="cor001">*</xref>
</contrib>
</contrib-group>
<aff id="aff001"><label>1</label> <addr-line>Else Kroener Fresenius Center for Digital Health, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology, Dresden, Germany</addr-line></aff>
<aff id="aff002"><label>2</label> <addr-line>Cambridge Centre for AI in Medicine, University of Cambridge, Cambridge, United Kingdom</addr-line></aff>
<aff id="aff003"><label>3</label> <addr-line>Department of Applied Mathematics and Theoretical Physics, University of Cambridge, Cambridge, United Kingdom</addr-line></aff>
<aff id="aff004"><label>4</label> <addr-line>Department of Medicine I, Faculty of Medicine and University Hospital Carl Gustav Carus, TUD Dresden University of Technology, Dresden, Germany</addr-line></aff>
<aff id="aff005"><label>5</label> <addr-line>Medical Oncology, National Center for Tumor Diseases (NCT), University Hospital Heidelberg, Heidelberg, Germany</addr-line></aff>
<aff id="aff006"><label>6</label> <addr-line>Pathology &amp; Data Analytics, Leeds Institute of Medical Research at St James’s, University of Leeds, Leeds, United Kingdom</addr-line></aff>
<author-notes>
<corresp id="cor001">* E-mail: <email xlink:type="simple">kather.jn@tu-dresden.de</email></corresp>
<fn fn-type="conflict" id="coi001">
<p>I have read the journal’s policy and the authors of this manuscript have the following competing interests: JNK holds shares in StratifAI, Synagen, Spira Labs, Tremont AI, and Saterra AI; is Co-PI on institutional research grants from GSK and AstraZeneca, and declares honoraria or consulting fees from AstraZeneca, Bayer, Bioptimus, Daiichi Sankyo, Eisai, Janssen, Merck, MSD, Novartis, BMS, Roche, and Pfizer. The Cambridge Centre for AI in Medicine (CCAIM) receives funding from GSK, Boehringer-Ingelheim, AstraZeneca, Sanofi and Quantum Black, AI by McKinsey. D.F. holds shares in and is employed at Synagen AI GmbH. D.F. has received a research grant from OpenAI.</p>
</fn>
</author-notes>
<pub-date pub-type="epub"><day>9</day><month>7</month><year>2026</year></pub-date>
<pub-date pub-type="collection"><month>7</month><year>2026</year></pub-date>
<volume>23</volume>
<issue>7</issue>
<elocation-id>e1005170</elocation-id>
<permissions>
<copyright-year>2026</copyright-year>
<copyright-holder>Ruhrberg Estévez et al</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
<license-p>This is an open access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">Creative Commons Attribution License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p></license>
</permissions>
<self-uri content-type="pdf" xlink:href="info:doi/10.1371/journal.pmed.1005170"/>
<abstract abstract-type="teaser">
<p>Medical artificial intelligence research is shifting from single-task models toward multimodal large language model-based agents for complex clinical workflows, requiring benchmarks that assess clinical reasoning, process safety, and resource stewardship rather than final outputs alone.</p>
</abstract>
<abstract abstract-type="toc">
<p>In this Perspective article, Silas Ruhrberg Estévez and colleagues discuss why, as medical AI research shifts toward multimodal large language model-based agents for complex clinical workflows, benchmarks that assess clinical reasoning, process safety, and resource stewardship—rather than final outputs alone—are required.</p>
</abstract>
<funding-group>
<funding-statement>The author(s) received no specific funding for this work.</funding-statement>
</funding-group>
<counts>
<fig-count count="1"/>
<table-count count="0"/>
<page-count count="5"/>
</counts>
</article-meta>
</front>
<body>
<sec id="sec001" sec-type="intro">
<title>Introduction</title>
<p>Before deployment in high-stakes settings, artificial intelligence (AI) systems must be shown to be accurate, safe, generalizable, and useful for their intended tasks [<xref ref-type="bibr" rid="pmed.1005170.ref001">1</xref>]. In machine learning, standardized benchmarks allow algorithms to be compared under shared conditions. They do more than measure performance: They define the problem, determine what counts as success, and shape scientific progress. ImageNet illustrates this effect by providing a large-scale, standardized task that helped drive modern computer vision [<xref ref-type="bibr" rid="pmed.1005170.ref002">2</xref>]. Medical AI has followed the same benchmarking logic; radiology datasets have enabled systematic comparison of imaging models [<xref ref-type="bibr" rid="pmed.1005170.ref003">3</xref>], while question-answering benchmarks such as MedQA [<xref ref-type="bibr" rid="pmed.1005170.ref004">4</xref>] assess clinical reasoning. These benchmarks share a simple evaluation structure: a fixed input, a single response, and a reference answer against which predictions are scored. This structure has been effective in measuring and accelerating progress on bounded clinical tasks, from image interpretation and histopathology classification to knowledge-based medical licensing examinations, with some systems reaching expert-level performance under controlled benchmark conditions [<xref ref-type="bibr" rid="pmed.1005170.ref005">5</xref>,<xref ref-type="bibr" rid="pmed.1005170.ref006">6</xref>].</p>
<p>As illustrated in <xref ref-type="fig" rid="pmed.1005170.g001">Fig 1A</xref>, classical machine-learning models typically map fixed clinical inputs to prediction outputs, while large language models (LLMs) generate text responses to clinical questions. Medical AI agents differ because they operate within a clinical workflow: they actively gather information, use tools, and adapt subsequent decisions as new information becomes available. Final-output matching is therefore insufficient for agentic workflows, because clinical care often allows multiple safe trajectories. Competence depends on how information is gathered, actions are sequenced, and decisions adapt over time. Emerging benchmarks have begun to reflect this through simulated patient encounters and iterative diagnostic tasks [<xref ref-type="bibr" rid="pmed.1005170.ref007">7</xref>]. In these settings, performance is more variable, suggesting that earlier benchmarks primarily rewarded static answers rather than longitudinal decision-making [<xref ref-type="bibr" rid="pmed.1005170.ref006">6</xref>,<xref ref-type="bibr" rid="pmed.1005170.ref008">8</xref>].</p>
<fig id="pmed.1005170.g001" position="float"><object-id pub-id-type="doi">10.1371/journal.pmed.1005170.g001</object-id><label>Fig 1</label><caption><title>Evaluation settings for medical AI models.</title><p><bold>(A)</bold> Schematic contrast between classical machine-learning models, LLM-based clinical question answering, and autonomous medical AI agents embedded in clinical workflows. <bold>(B)</bold> Overview of benchmark domains across model generations, highlighting the additional evaluation dimensions required for agentic systems: action appropriateness, process safety, and resource stewardship. Figure was created in BioRender. Ruhrberg Estévez, S. (2026) <ext-link ext-link-type="uri" xlink:href="https://BioRender.com/s6mb56s" xlink:type="simple">https://BioRender.com/s6mb56s</ext-link>.</p></caption>
<graphic mimetype="image" position="float" xlink:href="info:doi/10.1371/journal.pmed.1005170.g001" xlink:type="simple"/></fig>
</sec>
<sec id="sec002">
<title>Benchmarking sequential decision-making</title>
<p>Similar evaluation paradigms already exist outside medicine. In interactive environments such as video games, evaluation extends beyond recalling facts about the environment and instead assesses whether agents can observe the current state, select actions, adapt to feedback, and pursue long-horizon objectives [<xref ref-type="bibr" rid="pmed.1005170.ref009">9</xref>]. These environments distinguish an agent’s strategic choices from the mechanics of action execution. For example, an agent playing a video game such as Pokémon must decide when to explore, collect characters, interact with non-player characters, or enter battles; success depends on how these decisions are sequenced as the game state evolves.</p>
<p>The same distinction applies to medical agents operating within an electronic health record. Clinical decision-making is sequential, multimodal, and constrained by cost, time, resource availability, and patient safety. The agent must decide whether to ask further history questions, perform examination steps, order laboratory tests or imaging, request specialist input, consult guidelines, or initiate treatment. Existing benchmarks remain valuable for evaluating these component capabilities. In agentic systems, however, the same models increasingly function as tools within a broader decision-making workflow rather than as standalone predictors [<xref ref-type="bibr" rid="pmed.1005170.ref005">5</xref>,<xref ref-type="bibr" rid="pmed.1005170.ref010">10</xref>].</p>
<p>Early clinical agent benchmarks illustrate why workflow-level evaluation is necessary. Multi-step procedures create failure modes invisible to endpoint scoring, including shortcut learning, premature closure, unnecessary tool use, and clinically implausible paths to otherwise correct answers [<xref ref-type="bibr" rid="pmed.1005170.ref008">8</xref>,<xref ref-type="bibr" rid="pmed.1005170.ref011">11</xref>]. Recent evaluations of LLM-based clinical agents show that adding tools does not automatically produce reliable clinical behavior, with only modest gains over baseline models, as well as persistently low performance in several diagnostic and multimodal settings, and substantially increased resource use [<xref ref-type="bibr" rid="pmed.1005170.ref012">12</xref>]. Process-aware evaluation can address these limitations by measuring deviation from reference workups, completion of safety checks, guideline adherence, and proportionality of resource use. Such benchmarks would direct model development toward clinically meaningful behavior and provide developers, clinicians, and regulators with more interpretable evidence for assessing readiness before deployment.</p>
</sec>
<sec id="sec003">
<title>Benchmarking medical AI agents</title>
<p>In an emergency department workup, a patient presenting with abdominal pain is assessed iteratively: history-taking informs laboratory testing, which guides imaging and treatment decisions. Each step depends on prior actions, and errors in ordering, sequencing, or overuse of tests carry both clinical and economic consequences. Similarly, tumor board decision-making requires integrating pathology, CT and MRI findings, clinical guidelines, comorbidities, and trial eligibility into a coordinated treatment plan across specialties. In these settings, correctness depends on the pathway as well as the endpoint. A correct diagnosis reached through excessive or misordered interventions is not clinically equivalent to an efficient, guideline-concordant workup.</p>
<p>A process-aware benchmark for medical AI agents should assess the entire decision trajectory (see <xref ref-type="fig" rid="pmed.1005170.g001">Fig 1B</xref>). Three dimensions are particularly important. Action appropriateness captures whether the system selects clinically sensible actions at the right point in the workflow. Process safety captures whether it avoids unsafe steps, such as invasive investigations without indication, omitted red-flag screening, or treatment recommendations made without checking contraindications. Resource stewardship captures whether it avoids unnecessary tests, procedures, referrals, or costs.</p>
<p>Because clinical trajectories rarely have a single ground truth, future benchmarks should define acceptable ranges of practice rather than one fixed answer. Reference trajectories should be derived from expert consensus, clinical guidelines, simulated patient encounters, and, where appropriate, observed care pathways. Multidisciplinary clinical panels and formal consensus methods could help bound acceptable variation while allowing benchmarks to be updated as evidence and practice evolve. Such benchmarks should capture the full structure of clinical decision-making, including the sequence of clinical findings, laboratory and imaging requests, referrals, treatment decisions, and safety-critical omissions.</p>
<p>Building these benchmarks will be technically and financially demanding, requiring robust simulated clinical environments and sustained expert input. Their design must reflect the intended clinical role of the agent. In the foreseeable future, medical AI agents are likely to act primarily in clinician-supporting roles rather than replacing them. In tumor board settings, agents may be most useful when they broaden multidisciplinary review by identifying relevant trials, off-label treatments and additional evidence. More autonomous systems that execute clinical workflows would require stricter evaluation. Benchmark metrics must also be designed carefully to avoid reward hacking, where agents optimize benchmark scores without improving clinical utility. The goal is to permit legitimate variation in clinical practice while identifying trajectories that are unsafe, inefficient, or poorly justified. As medical AI moves from prediction to action, benchmarks must evolve from scoring isolated answers to evaluating clinically plausible decision trajectories.</p>
</sec>
</body>
<back>
<glossary>
<title>Abbreviations</title>
<def-list><def-item><term>AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term>LLM</term><def><p>large language model</p></def></def-item></def-list>
</glossary>
<ack>
<p><xref ref-type="fig" rid="pmed.1005170.g001">Fig 1</xref> was created in BioRender. Ruhrberg Estévez, S. (2026) <ext-link ext-link-type="uri" xlink:href="https://BioRender.com/s6mb56s" xlink:type="simple">https://BioRender.com/s6mb56s</ext-link>.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="pmed.1005170.ref001"><label>1</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Vasey</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Nagendran</surname> <given-names>M</given-names></name>, <name name-style="western"><surname>Campbell</surname> <given-names>B</given-names></name>, <name name-style="western"><surname>Clifton</surname> <given-names>DA</given-names></name>, <name name-style="western"><surname>Collins</surname> <given-names>GS</given-names></name>, <name name-style="western"><surname>Denaxas</surname> <given-names>S</given-names></name>, <etal>et al</etal>. <article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title>. <source>Nat Med</source>. <year>2022</year>;<volume>28</volume>(<issue>5</issue>):<fpage>924</fpage>–<lpage>33</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41591-022-01772-9" xlink:type="simple">10.1038/s41591-022-01772-9</ext-link></comment> <object-id pub-id-type="pmid">35585198</object-id></mixed-citation></ref>
<ref id="pmed.1005170.ref002"><label>2</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Krizhevsky</surname> <given-names>A</given-names></name>, <name name-style="western"><surname>Sutskever</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Hinton</surname> <given-names>GE</given-names></name>. <source>ImageNet classification with deep convolutional neural networks. The Twenty-Fifth Annual Conference on Neural Information Processing Systems</source>. <year>2012</year></mixed-citation></ref>
<ref id="pmed.1005170.ref003"><label>3</label><mixed-citation publication-type="other" xlink:type="simple">Irvin J, Rajpurkar P, Ko M, Yu Y, Ciurea-Ilcus S, Chute C, et al. CheXpert: a large chest radiograph dataset with uncertainty labels and expert comparison. In: Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI. 2019.</mixed-citation></ref>
<ref id="pmed.1005170.ref004"><label>4</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Jin</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Pan</surname> <given-names>E</given-names></name>, <name name-style="western"><surname>Oufattole</surname> <given-names>N</given-names></name>, <name name-style="western"><surname>Weng</surname> <given-names>WH</given-names></name>, <name name-style="western"><surname>Fang</surname> <given-names>H</given-names></name>, <name name-style="western"><surname>Szolovits</surname> <given-names>P</given-names></name>. <article-title>What disease does this patient have? A large-scale open domain question answering dataset from medical exams</article-title>. <source>Appl Sci</source>. <year>2021</year>;<volume>11</volume>(<issue>14</issue>):<fpage>6421</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/app11146421" xlink:type="simple">10.3390/app11146421</ext-link></comment></mixed-citation></ref>
<ref id="pmed.1005170.ref005"><label>5</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Schmidgall</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Ziaei</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Harris</surname> <given-names>C</given-names></name>, <name name-style="western"><surname>Kim</surname> <given-names>JW</given-names></name>, <name name-style="western"><surname>Reis</surname> <given-names>EP</given-names></name>, <name name-style="western"><surname>Jopling</surname> <given-names>J</given-names></name>. <article-title>AgentClinic: a multimodal benchmark for tool-using clinical AI agents</article-title>. <source>npj Digital Medicine</source>. <year>2026</year>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41746-026-02674-7" xlink:type="simple">10.1038/s41746-026-02674-7</ext-link></comment></mixed-citation></ref>
<ref id="pmed.1005170.ref006"><label>6</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Bean</surname> <given-names>AM</given-names></name>, <name name-style="western"><surname>Payne</surname> <given-names>RE</given-names></name>, <name name-style="western"><surname>Parsons</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Kirk</surname> <given-names>HR</given-names></name>, <name name-style="western"><surname>Ciro</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Mosquera-Gómez</surname> <given-names>R</given-names></name>, <etal>et al</etal>. <article-title>Reliability of LLMs as medical assistants for the general public: a randomized preregistered study</article-title>. <source>Nat Med</source>. <year>2026</year>;<volume>32</volume>(<issue>2</issue>):<fpage>609</fpage>–<lpage>15</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41591-025-04074-y" xlink:type="simple">10.1038/s41591-025-04074-y</ext-link></comment> <object-id pub-id-type="pmid">41663592</object-id></mixed-citation></ref>
<ref id="pmed.1005170.ref007"><label>7</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Hager</surname> <given-names>P</given-names></name>, <name name-style="western"><surname>Jungmann</surname> <given-names>F</given-names></name>, <name name-style="western"><surname>Holland</surname> <given-names>R</given-names></name>, <name name-style="western"><surname>Bhagat</surname> <given-names>K</given-names></name>, <name name-style="western"><surname>Hubrecht</surname> <given-names>I</given-names></name>, <name name-style="western"><surname>Knauer</surname> <given-names>M</given-names></name>, <etal>et al</etal>. <article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title>. <source>Nat Med</source>. <year>2024</year>;<volume>30</volume>(<issue>9</issue>):<fpage>2613</fpage>–<lpage>22</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41591-024-03097-1" xlink:type="simple">10.1038/s41591-024-03097-1</ext-link></comment> <object-id pub-id-type="pmid">38965432</object-id></mixed-citation></ref>
<ref id="pmed.1005170.ref008"><label>8</label><mixed-citation publication-type="other" xlink:type="simple">Chiu C, Pitis S, van der Schaar M. Simulating viva voce examinations to evaluate clinical reasoning in large language models. In: The Thirty-ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track. 2025.</mixed-citation></ref>
<ref id="pmed.1005170.ref009"><label>9</label><mixed-citation publication-type="other" xlink:type="simple">Park D, Kim M, Choi B, Kim J, Lee K, Lee J, et al. Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games. In: 2026.</mixed-citation></ref>
<ref id="pmed.1005170.ref010"><label>10</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Ferber</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>El Nahhas</surname> <given-names>OSM</given-names></name>, <name name-style="western"><surname>Wölflein</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Wiest</surname> <given-names>IC</given-names></name>, <name name-style="western"><surname>Clusmann</surname> <given-names>J</given-names></name>, <name name-style="western"><surname>Leßmann</surname> <given-names>M-E</given-names></name>, <etal>et al</etal>. <article-title>Development and validation of an autonomous artificial intelligence agent for clinical decision-making in oncology</article-title>. <source>Nat Cancer</source>. <year>2025</year>;<volume>6</volume>(<issue>8</issue>):<fpage>1337</fpage>–<lpage>49</lpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s43018-025-00991-6" xlink:type="simple">10.1038/s43018-025-00991-6</ext-link></comment> <object-id pub-id-type="pmid">40481323</object-id></mixed-citation></ref>
<ref id="pmed.1005170.ref011"><label>11</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Safavi-Naini</surname> <given-names>SAA</given-names></name>, <name name-style="western"><surname>Ali</surname> <given-names>S</given-names></name>, <name name-style="western"><surname>Shahab</surname> <given-names>O</given-names></name>, <name name-style="western"><surname>Shahhoseini</surname> <given-names>Z</given-names></name>, <name name-style="western"><surname>Savage</surname> <given-names>T</given-names></name>, <name name-style="western"><surname>Rafiee</surname> <given-names>S</given-names></name>, <etal>et al</etal>. <article-title>Benchmarking proprietary and open-source language and vision-language models for gastroenterology clinical reasoning</article-title>. <source>NPJ Digit Med</source>. <year>2025</year>;<volume>8</volume>(<issue>1</issue>):<fpage>797</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41746-025-02174-0" xlink:type="simple">10.1038/s41746-025-02174-0</ext-link></comment> <object-id pub-id-type="pmid">41310206</object-id></mixed-citation></ref>
<ref id="pmed.1005170.ref012"><label>12</label><mixed-citation publication-type="journal" xlink:type="simple"><name name-style="western"><surname>Liu</surname> <given-names>Y</given-names></name>, <name name-style="western"><surname>Carrero</surname> <given-names>ZI</given-names></name>, <name name-style="western"><surname>Jiang</surname> <given-names>X</given-names></name>, <name name-style="western"><surname>Ferber</surname> <given-names>D</given-names></name>, <name name-style="western"><surname>Wölflein</surname> <given-names>G</given-names></name>, <name name-style="western"><surname>Zhang</surname> <given-names>L</given-names></name>, <etal>et al</etal>. <article-title>Benchmarking large language model-based agent systems for clinical decision tasks</article-title>. <source>NPJ Digit Med</source>. <year>2026</year>;<volume>9</volume>(<issue>1</issue>):<fpage>259</fpage>. <comment>doi: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41746-026-02443-6" xlink:type="simple">10.1038/s41746-026-02443-6</ext-link></comment> <object-id pub-id-type="pmid">41708802</object-id></mixed-citation></ref>
</ref-list>
</back>
</article>