<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e92584</article-id><article-id pub-id-type="doi">10.2196/92584</article-id><article-categories><subj-group subj-group-type="heading"><subject>Viewpoint</subject></subj-group></article-categories><title-group><article-title>Medical AI Agents for Clinical Decision Support: Viewpoint Using the Planning, Action, Reflection, and Memory (PARM) Analytical Lens</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Dinc</surname><given-names>Rasit</given-names></name><degrees>Prof Dr, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ardic</surname><given-names>Nurittin</given-names></name><degrees>MD, Prof Dr</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>INVAMED Medical Innovation Institute</institution><addr-line>One World Trade Centre, 85th Floor 285 Fulton Street</addr-line><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff2"><institution>Med-International UK Health Agency Ltd.</institution><addr-line>Nuneaton</addr-line><addr-line>Warwickshire</addr-line><country>United Kingdom</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Newton</surname><given-names>Nicki</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Thomson</surname><given-names>Riley</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Palama</surname><given-names>Valentina</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Rasit Dinc, Prof Dr, PhD, INVAMED Medical Innovation Institute, One World Trade Centre, 85th Floor 285 Fulton Street, New York, NY, 10007, United States, 1 3475350630; <email>rasitdinc@hotmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e92584</elocation-id><history><date date-type="received"><day>31</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>11</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>17</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Rasit Dinc, Nurittin Ardic. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 21.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e92584"/><abstract><p>Medical AI agents are emerging as a new generation of clinical decision support systems, moving beyond static prediction toward multistep, workflow-oriented assistance. This Viewpoint argues that agentic architectures incorporating planning, action, reflection, and memory (PARM) represent a meaningful evolution beyond traditional rule-based, machine learning, and multimodal clinical decision support systems. Using PARM as an analytical lens, we examine how medical AI agents can support diagnostic reasoning, treatment planning, and longitudinal monitoring while remaining constrained by human oversight. We further discuss the governance mechanisms required for responsible implementation, including bounded autonomy, auditability, verification protocols, postdeployment surveillance, and clear accountability structures. Rather than proposing autonomous modification of clinical judgment, this Viewpoint emphasizes agentic AI as a supervised workflow support paradigm. Safe implementation will require technical safeguards, institutional governance, regulatory clarity, and evaluation approaches that assess end-to-end task reliability, escalation behavior, and performance under deployment shifts.</p></abstract><kwd-group><kwd>medical AI</kwd><kwd>clinical decision support</kwd><kwd>AI agents</kwd><kwd>multimodal AI</kwd><kwd>human oversight</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Clinical decision support systems (CDSSs) have long been positioned as a cornerstone of digital medicine, aiming to improve diagnostic accuracy, treatment planning, and patient follow-up by leveraging the computational analysis of clinical data. Early CDSS implementations were largely rule-based, encoding expert knowledge into deterministic logic that operated within narrowly defined clinical domains [<xref ref-type="bibr" rid="ref1">1</xref>]. Although effective in specific contexts, these systems often struggled to scale with the increasing complexity, heterogeneity, and volume of modern health care data. Subsequent generations of data-driven approaches, including statistical machine learning and deep learning, significantly improved predictive performance but remained fundamentally reactive, offering recommendations or risk scores without the ability to autonomously plan, act, or adapt over time [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>The last decade has seen rapid advancements in multimodal AI, enabling the integration of various data types, such as clinical text, medical imaging, physiological signals, laboratory results, and increasingly, patient-generated data. Multimodal approaches have improved performance by integrating different data types [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Despite these advances, many currently reported multimodal CDSS applications remain limited to single-step inference or static estimation. They often lack the capacity for persistent memory, goal-directed behavior, and the translation of insights into coordinated clinical actions, thus limiting their impact on real-world clinical workflows. Recent advances in foundation models and tool-assisted architectures have begun to address these limitations [<xref ref-type="bibr" rid="ref8">8</xref>]. Large language models (LLMs) and multimodal foundation models now exhibit enhanced reasoning capabilities, natural language understanding, and the ability to interact with external tools such as calculators, databases, and application programming interfaces. Frameworks integrating reasoning with action selection have shown that models can iteratively parse complex goals, call appropriate tools, and interpret results to inform further reasoning [<xref ref-type="bibr" rid="ref9">9</xref>]. In parallel, retrieval-augmented generation has enabled dynamic access to external information sources by mitigating the limitations of static model training and improving the factual basis of decision-making in clinical contexts [<xref ref-type="bibr" rid="ref10">10</xref>]. These developments have accelerated the emergence of medical AI agent systems designed not only to predict or advise but also to achieve clinical goals through iterative sensing, reasoning, action, and learning.</p><p>In health care settings, medical AI agents are increasingly conceptualized as autonomous or semiautonomous computational entities operating within extended clinical workflows while remaining subject to appropriate human oversight. In this study, the term &#x201C;agent&#x201D; refers to structured workflow support with limited autonomy and explicit clinical approval for high-impact clinical decisions. Rather than generating isolated outputs, agent systems can generate diagnostic or treatment plans, execute subtasks through controlled tool use, monitor outcomes, and adapt strategies based on feedback from the clinical setting. Recent studies have converged on a recurring set of functional components underlying such systems, often defined by 4 core components: planning, action, reflection, and memory (PARM) [<xref ref-type="bibr" rid="ref11">11</xref>]. Planning refers to the decomposition of higher-level clinical goals into executable steps, action encompasses the execution of these steps through interaction with clinical information systems or decision support tools, reflection involves the evaluation of outcomes and system performance, and memory enables the storage and retrieval of contextual information over time [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. In subsequent sections of this study, the 4 core components will be referred to collectively using the acronym PARM for clarity. Although these components have been defined across various branches of the literature, including multimodal AI, autonomous agents, and clinical informatics, they are often discussed in isolation or within domain-specific applications. Consequently, the field lacks a unified synthesis that positions medical AI agents within the broader evolution of CDSS and that explains how agent architectures expand upon, rather than replace, existing multimodal approaches [<xref ref-type="bibr" rid="ref14">14</xref>]. This fragmentation makes it difficult for clinicians, researchers, and regulators to evaluate agent systems; compare designs; and reason about safety, accountability, and clinical value.</p><p>The aim of this Viewpoint is to synthesize current research on medical AI agents for clinical decision support by organizing the literature around the architectural dimensions of PARM. Rather than proposing a new standard or formal framework, this study uses these components as an analytical lens to examine how agent capabilities are incorporated into contemporary CDSSs. This Viewpoint aims to provide conceptual clarity and practical guidance for the responsible development and deployment of medical AI agents in clinical practice by tracing the transition from multimodal predictive systems to agent-based architectures, highlighting representative clinical applications, and discussing key safety and governance considerations.</p><p>Throughout this Viewpoint, PARM is used as an analytical lens to examine how agentic architectures extend traditional CDSSs. The purpose is not to provide an exhaustive review of all published studies but to offer a conceptual perspective on the emerging role of medical AI agents in clinical workflows.</p></sec><sec id="s2"><title>The Shift Toward Agentic Clinical Decision Support</title><p>Although multimodal AI enables the integration of medical imaging, clinical text, laboratory data, and physiological signals, many clinical applications remain limited to single-step inference. For example, a chest X-ray model may generate diagnostic predictions, while a clinical language model may summarize notes. These systems produce outputs but cannot autonomously pursue clinical goals across lengthy workflows.</p><p>Medical AI agents represent a qualitative shift. Instead of providing static predictions, agent-based systems iteratively plan sequences of actions, execute them through tool use, monitor outcomes, and adapt strategies based on feedback [<xref ref-type="bibr" rid="ref9">9</xref>]. Recent studies have shown that LLMs can perform complex clinical tasks, access calculators and databases, interpret results, and refine their approaches through reflective reasoning [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. This enables goal-directed behavior: an agent tasked with &#x201C;optimize anticoagulation for this patient&#x201D; can retrieve guidelines, assess bleeding risk, check for drug interactions, and recommend monitoring programs&#x2014;tasks that previously required coordination between multiple independent systems or manual processes.</p><p><xref ref-type="fig" rid="figure1">Figure 1</xref> schematically illustrates the PARM architecture for an agentic CDSS with integrated clinical supervision.</p><p>On the basis of this evolutionary perspective, the following section examines the fundamental architectural components of PARM that enable agent behavior in CDSSs as interdependent functional elements.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Evolution of clinical decision support systems (CDSSs) from traditional rule-based to agent-based architectures. Rule-based CDSSs rely on deterministic IF-THEN logic and expert-encoded information to generate alerts or recommendations from single data types, exhibiting static behavior with no learning or memory capabilities. Multimodal CDSSs integrate various data sources (clinical text, imaging, laboratory results, and physiological signals) through deep learning models to generate predictions or risk scores. Although multimodal fusion increases accuracy, these systems remain limited to single-step inference and reactive responses. Agent-based CDSSs extend multimodal capabilities by iteratively traversing the stages of planning (goal decomposition), action (tool use and execution), reflection (outcome evaluation), and memory (contextual retention). This PARM (Planning, Action, Reflection, and Memory)&#x2013;based arrangement enables goal-directed, multistep reasoning and adaptive behavior. Human-supervised checkpoints provide a form of limited autonomy by requiring clinician approval for high-risk decisions. The shift from static prediction to iterative, adaptive assistance represents a qualitative change in clinical AI capabilities.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e92584_fig01.png"/></fig></sec><sec id="s3"><title>Key Architectural Components of Medical AI Agents</title><sec id="s3-1"><title>Overview</title><p>Medical AI agents differ from traditional CDSSs through the integration of architectural components that enable goal-oriented, adaptive behavior in clinical workflows. In this section, PARM is considered a set of distinct yet interdependent elements, referred to as PARM components, that serve as an analytical lens rather than a prescriptive framework. Recent literature has revealed 4 interdependent components of central importance in the design of agent systems in health care. These components are not unique to medicine, but their applications and limitations are shaped by clinical uncertainty, safety requirements, and the need for human oversight. In this section, each component is examined in turn, with an emphasis on its role in clinical decision support. To provide a structured overview of these components and their roles in clinical decision support, <xref ref-type="table" rid="table1">Table 1</xref> summarizes the functional goals, representational capabilities, and key considerations related to PARM in medical AI agents.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Architectural components of medical AI<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> agents in clinical decision support.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Components</td><td align="left" valign="bottom">Functional role in clinical decision support systems</td><td align="left" valign="bottom">Example capabilities</td><td align="left" valign="bottom">Key considerations</td><td align="left" valign="bottom">Technology enablers</td></tr></thead><tbody><tr><td align="left" valign="top">Planning</td><td align="left" valign="top">Goal decomposition and task sequencing</td><td align="left" valign="top">Diagnostic pathways and care planning</td><td align="left" valign="top">Transparency and clinician alignment</td><td align="left" valign="top">Large language models and reasoning frameworks</td></tr><tr><td align="left" valign="top">Action</td><td align="left" valign="top">Tool-assisted execution</td><td align="left" valign="top">Electronic health record queries and guideline retrieval</td><td align="left" valign="top">Authorization and accountability</td><td align="left" valign="top">APIs<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> and retrieval systems</td></tr><tr><td align="left" valign="top">Reflection</td><td align="left" valign="top">Outcome evaluation and performance monitoring</td><td align="left" valign="top">Error detection and performance drift monitoring</td><td align="left" valign="top">Auditability and traceability</td><td align="left" valign="top">Feedback loops and monitoring systems</td></tr><tr><td align="left" valign="top">Memory</td><td align="left" valign="top">Context retention over time</td><td align="left" valign="top">Longitudinal patient context</td><td align="left" valign="top">Privacy and governance</td><td align="left" valign="top">Vector databases and knowledge graphs</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AI: artificial intelligence.</p></fn><fn id="table1fn2"><p><sup>b</sup>API: application programming interface.</p></fn></table-wrap-foot></table-wrap><p>Although these components can be defined individually, their clinical benefits emerge from their coordinated work within real-world workflows. Therefore, each component is discussed in more detail below, considering both its independent function and its interaction with the broader agent system.</p></sec><sec id="s3-2"><title>Planning</title><p>Planning enables medical AI agents to break down high-level clinical goals into executable sequences of action. Given an objective such as &#x201C;Evaluate this patient for acute coronary syndrome,&#x201D; an agent must guide subsequent steps by determining which diagnostic tests to order, in what order, and how to interpret the results. Planning systems may leverage clinical guidelines, task-specific protocols, and contextual patient data to create sequences of action that balance comprehensiveness and efficiency. Advanced planning frameworks can generate conditional plans that predict multiple possible outcomes and define appropriate responses for each [<xref ref-type="bibr" rid="ref12">12</xref>]. This hybrid approach allows agents to remain flexible while adhering to domain-specific requirements.</p><p>However, the main implementation challenge is ensuring that automatically generated plans remain clinically sound, evidence-based, and appropriately constrained; that is, they should not exceed the agent&#x2019;s defined scope of autonomy without explicit clinical approval. Overly aggressive or autonomous planning risks conflicting with clinician intent or patient preferences, highlighting the importance of transparency and controllability. Because of this uncertainty and these ethical considerations, planning components in medical AI agents are typically designed to support collaborative decision-making rather than independent execution.</p></sec><sec id="s3-3"><title>Action</title><p>Action encompasses the execution of planned steps through interaction with clinical information systems, decision support tools, or external information sources. Unlike traditional CDSSs, whose outputs are often limited to warnings or recommendations, agent-based systems can initiate controlled actions such as querying electronic health records, retrieving clinical guidelines, performing calculations, or generating structured reports.</p><p>Recent studies on tool-assisted reasoning have shown how AI systems can intertwine reasoning with action selection, allowing agents to select appropriate tools, interpret the returned information, and update subsequent steps accordingly [<xref ref-type="bibr" rid="ref9">9</xref>]. In clinical settings, especially when interacting with live systems, such actions must be carefully limited to avoid unintended consequences. Consequently, many medical AI agents operate within predefined action domains and require explicit human approval for high-impact actions.</p><p>The distinction between recommendation and execution is critical. Although action-capable agents can streamline workflows and reduce cognitive load, their deployment must maintain clinician responsibility. Therefore, current applications focus on auxiliary actions, in which agents carry out preparatory or information-gathering tasks while final decisions and interventions remain under human control.</p></sec><sec id="s3-4"><title>Reflection</title><p>Reflection refers to an agent&#x2019;s capacity to evaluate outcomes under clinical management, monitor its own behavior, and support iterative improvement. Reflection is particularly important in medical AI agents because performance failures often stem not from isolated prediction errors but from workflow-level issues such as premature shutdown, inappropriate tool use, or failure to report ambiguity. In practice, reflective functions can operate at three levels: (1) encounter-level controls (eg, identifying missing data, inconsistencies, or low-reliability recommendations requiring clinician review), (2) operational monitoring (eg, monitoring alert payload, override frequency, and abnormal action patterns), and (3) postdeployment surveillance (eg, detecting performance deviation and subgroup performance differences). In clinical settings, reflection is typically designed to support auditability and safety rather than autonomous self-modification. Continuous adaptation without oversight can reduce reproducibility and complicate accountability; therefore, reflective outputs should be logged, auditable, and linked to predefined update policies. This approach provides more secure integration into clinical workflows while preserving the authority of clinical physicians and traceable decision-making processes [<xref ref-type="bibr" rid="ref13">13</xref>].</p></sec><sec id="s3-5"><title>Memory</title><p>Memory provides the structural foundation for continuity in agent-based CDSSs. By enabling the storage and retrieval of contextual information over time, it supports longitudinal reasoning and patient-specific adaptation. Memory in medical AI agents can encompass short-term context, such as the current clinical encounter, as well as longer-term representations, including previous decisions, outcomes, and relevant patient history.</p><p>Architecturally, memory systems can combine structured databases, vector-based retrieval mechanisms, and external information stores [<xref ref-type="bibr" rid="ref16">16</xref>]. Recall-enhanced approaches allow agents to access current clinical information or institutional protocols without relying solely on static model parameters [<xref ref-type="bibr" rid="ref10">10</xref>]. In patient-centered applications, memory supports more personalized and consistent decision support by enabling agents to be aware of evolving clinical trends.</p><p>At the same time, memory also raises significant concerns regarding privacy, data management, and bias [<xref ref-type="bibr" rid="ref17">17</xref>]. Decisions about what information to store, for how long, and how to reuse it must comply with regulatory requirements and ethical standards. Therefore, the memory components in medical AI agents are closely intertwined with governance frameworks and organizational policies.</p></sec></sec><sec id="s4"><title>Operationalizing Agentic Architectures in Clinical Decision Support</title><sec id="s4-1"><title>Overview</title><p>The examples in this section are illustrative scenarios intended to show how agentic capabilities can be operationalized in CDSSs; they are not presented as validated end-to-end clinical systems unless explicitly supported by the cited literature. Although architectural components such as PARM define the conceptual foundation of medical AI agents, their clinical significance depends on how these capabilities are embodied in real-world decision support workflows. Operationalization requires aligning agent behavior with the temporal structure of clinical care, existing health information systems, and accountability frameworks governing medical decision-making processes. In this section, we highlight both opportunities and limitations by examining how agent architectures are implemented in key clinical decision support contexts [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p></sec><sec id="s4-2"><title>Diagnostic Decision Support</title><p>Diagnostic reasoning is an inherently iterative process involving hypothesis generation, targeted data collection, interpretation, and refinement [<xref ref-type="bibr" rid="ref14">14</xref>]. An illustrative diagnostic AI agent might be directed by a command such as, &#x201C;72-year-old patient with chest pain, elevated troponin, and nonspecific findings on ECG.&#x201D; The agent plans differential diagnoses (eg, acute coronary syndrome, pulmonary embolism, and aortic dissection), retrieves relevant clinical guidelines, checks recent imaging and laboratory results, identifies missing data (eg, D-dimer status and previous cardiac history), and constructs a structured assessment with confidence estimates and suggested next steps [<xref ref-type="bibr" rid="ref20">20</xref>]. Through reflection, the agent flags conflicting findings or low-reliability elements requiring clinical review. Major risks include premature diagnostic closure, anchoring bias, or failure to recognize rare but critical diagnoses. Monitoring mechanisms include mandatory clinical review before any diagnostic conclusion is communicated to patients, confidence thresholds that trigger automated amplification, and audit trails documenting the agent&#x2019;s reasoning path. These trails are used for subsequent review and learning.</p></sec><sec id="s4-3"><title>Treatment Planning</title><p>In an illustrative scenario, treatment planning agents may extend beyond diagnosis, recommending therapeutic interventions tailored to individual patient contexts [<xref ref-type="bibr" rid="ref21">21</xref>]. Given a confirmed diagnosis such as &#x201C;newly diagnosed atrial fibrillation with moderate stroke risk,&#x201D; the agent uses current guidelines (eg, Congestive heart failure, Hypertension, Diabetes, Stroke/Transient Ischemic Attack (TIA)/Thromboembolism, Vascular disease, Age, Sex category scoring and anticoagulation options), assesses contraindications based on medication history and laboratory values, considers patient-specific factors such as renal function and bleeding risk, and recommends treatment options in order of appropriateness. The agent can simulate expected outcomes under different scenarios, flag potential drug-drug interactions, and suggest monitoring protocols. Through its memory component, the agent remains aware of previous treatment responses and side effects. Risks include inappropriate dosage recommendations, failure to consider rare contraindications, or inadequate assessment of patient preferences. The review process includes mandatory clinical approval before any treatment recommendation reaches the patient, real-time alerts for high-risk recommendations (eg, anticoagulation in patients with recent bleeding), and documentation of the rationale for each decision for further review.</p></sec><sec id="s4-4"><title>Monitoring and Surveillance</title><p>In a hypothetical but clinically plausible workflow, a monitoring agent may track a patient&#x2019;s condition over time, identifying clinically significant changes and triggering appropriate responses. For a patient receiving warfarin therapy, such an agent may continuously review incoming international normalized ratio values, compare them to target ranges, assess trends rather than isolated measurements, and identify factors that may explain deviations (eg, recent medication changes, missed doses, and dietary changes). When values fall outside acceptable limits, the agent generates alerts calibrated according to the level of urgency&#x2014;immediate notification for critical values and scheduled review for borderline results. Such an agent may maintain longitudinal context through memory and recognize patterns, such as recurring subtherapeutic levels, that may indicate noncompliance. Key risks include alert fatigue from overreporting, critical changes missed due to improperly calibrated thresholds, or failure to report emergencies. Monitoring mechanisms include adjustable sensitivity parameters reviewed by clinical leadership, mandatory human consent prior to dose adjustments, and regular review of alert response times and outcomes [<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s4-5"><title>Workflow Integration</title><p>In all diagnostic, treatment, and monitoring applications, successful implementation requires seamless integration into existing clinical workflows. This includes compatibility with electronic health record systems, minimal disruption to established practices, and clear protocols for transitioning between automated processes and human decision-making. Implementation must consider varying levels of technical infrastructure, clinicians&#x2019; familiarity with AI systems, and organizational governance frameworks that define appropriate use boundaries [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>].</p></sec></sec><sec id="s5" sec-type="discussion"><title>Discussion</title><sec id="s5-1"><title>Safety, Governance, and Human Oversight</title><p>This study identified PARM as recurring functional components in medical AI agents and highlighted bounded autonomy, auditability, and workflow integration as central implementation themes. The capabilities that enable medical AI agents to achieve clinical goals in multistep workflows (PARM) bring governance challenges not found in traditional decision support systems. Static predictive models provide discrete outputs for human review; agent systems initiate sequences of actions that can take hours or days, access multiple data sources, and modify their behavior based on intermediate results. Ensuring patient safety and maintaining appropriate human authority require governance frameworks tailored to these different operational characteristics [<xref ref-type="bibr" rid="ref25">25</xref>].</p><p>Bounded autonomy refers to the operational boundaries within which an agent can function independently and does not require explicit human authorization [<xref ref-type="bibr" rid="ref26">26</xref>]. Boundaries must be specified across multiple dimensions: clinical scope (eg, which situations and which patient populations), authority to act (eg, information retrieval and order entry), and decision risk (eg, routine monitoring and critical interventions). A monitoring agent tracking vital signs postoperatively can autonomously retrieve laboratory results and assess trends but must consult a clinician before adjusting drug dosages or ordering imaging studies. Implementation typically requires the clear specification of permitted actions, definite stop points preventing unauthorized actions regardless of the agent&#x2019;s recommendations, and clear escalation pathways when agents encounter situations outside their defined scope. Boundaries generally reflect not only clinical risk but also institutional capabilities, local practice patterns, and regulatory constraints. As agents demonstrate reliable performance in narrow scopes, boundaries can be gradually expanded through formal validation processes. In early deployment settings, a more restrictive operational scope with explicit clinician oversight may reduce safety and accountability risks [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>].</p><p>Effective oversight operates across multiple temporal phases: predeployment validation, real-time monitoring during operation, and postoperative review of outcomes [<xref ref-type="bibr" rid="ref29">29</xref>]. Predeployment validation extends beyond traditional model performance metrics to assess an agent&#x2019;s behavior across anticipated clinical scenarios, including edge cases, ambiguous presentations, and situations requiring appropriate escalation. Validation should involve domain experts who evaluate not only the final recommendations but also the intermediate reasoning steps and calls to action. During the study, real-time monitoring tracks agent actions against expected patterns and flags anomalies such as excessive calls to action, unusual query patterns, or recommendations that deviate from established guidelines without clear justification. Confidence scores and uncertainty estimates should trigger mandatory human review when agents encounter unusual situations. Postdeployment monitoring collects performance data across patient encounters and identifies systematic errors, shifts in behavior over time, or performance differences among patient subgroups. Together, these temporal layers contribute to a multilayered defense, ensuring that failures of a control mechanism do not compromise patient safety. <xref ref-type="fig" rid="figure2">Figure 2</xref> illustrates a multilayered human oversight framework for medical AI agents, where safety measures at the design stage, runtime oversight, and postdeployment management work concurrently to provide a deep-seated defense architecture that protects clinician authority while restricting agent behavior.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Multilayered human oversight framework for medical AI agents. The figure illustrates a multilayered defense-in-depth oversight architecture for agent-based clinical decision support systems (CDSSs). Agent behavior is primarily constrained by limited autonomy, technical security measures, and audit logs, while human oversight operates through clinician approval checkpoints, real-time audits, and postdeployment audit reviews. These layers are embedded within broader institutional, regulatory, and ethical governance structures, providing continuous oversight throughout the system life cycle.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e92584_fig02.png"/></fig><p>Complete auditability requires recording not only agent outputs but also the reasoning processes, data sources, and intermediate steps that led to those outputs [<xref ref-type="bibr" rid="ref30">30</xref>]. Audit trails should record which guidelines were consulted, how patient-specific factors influenced the recommendations, which alternative options were evaluated and rejected, and when human review was initiated. This transparency serves multiple functions: it allows clinicians to understand and validate the reasoning of agents, supports quality improvement by identifying recurring errors or suboptimal patterns, and ensures accountability when outcomes are negative. Most importantly, accountability remains with human clinicians, not AI systems, even when decision support becomes more agentic and workflow integrated [<xref ref-type="bibr" rid="ref31">31</xref>]. Agents function as assistive tools supporting the clinical decision-making process; ultimate responsibility for patient care rests with licensed professionals who have the authority to accept, modify, or reject agent recommendations. Documentation systems should clearly state which actions were initiated by agents, which were approved by clinicians, and the justification for any deviations from agent recommendations. This ensures the effective use of agent support while preserving established medical accountability frameworks.</p><p>Evaluating agent systems requires measures that go beyond predictive accuracy to assess task completion, behavioral appropriateness, and modes of failure [<xref ref-type="bibr" rid="ref29">29</xref>]. Task reliability measures whether agents successfully complete clinical workflows from start to finish, not just whether individual predictions are correct. An agent might correctly diagnose a condition but fail to order appropriate validation tests or inform the relevant clinician; this represents a task failure despite diagnostic accuracy. Constraint compliance monitoring tracks whether agents operate within defined limits and flags violations even if the results are acceptable. An agent ordering imaging studies outside its authorized scope raises managerial concerns regardless of whether the imaging is clinically appropriate. Recovery and decay models assess how agents respond to errors, ambiguous inputs, or missing data, determining whether they fail by escalating appropriately or produce unreliable outputs without signaling the ambiguity. Evaluation frameworks are expected to highlight not only what agents do correctly but also how they behave when faced with conditions outside those encountered during training, as real-world deployments inevitably involve scenarios not represented in development datasets.</p><p>As medical AI capabilities evolve toward agentic architectures, governance frameworks must evolve in parallel. Using PARM as an analytical lens may help developers, clinicians, and regulators design agentic systems that balance clinical benefit with appropriate constraints, transparency, and human authority. Future work should establish standardized evaluation protocols for agentic CDSSs, develop consensus guidelines for limited autonomy in clinical settings, and explore how human-agent collaboration patterns evolve as agents assume greater workflow responsibilities while maintaining patient safety and professional accountability.</p></sec><sec id="s5-2"><title>Open Challenges and Future Directions</title><p>Despite their promise, medical AI agents face significant challenges that must be addressed before they can move into widespread clinical use [<xref ref-type="bibr" rid="ref32">32</xref>]. A fundamental technical challenge involves managing context windows across lengthy patient encounters. Although current LLMs exhibit impressive reasoning capabilities, their safe and reliable clinical use remains an active area of investigation [<xref ref-type="bibr" rid="ref33">33</xref>]. Agents must integrate information that may arrive asynchronously (eg, laboratory results arriving hours after the initial assessment, imaging reports from external facilities, or expert consultations occurring days later) while maintaining diagnostic consistency and avoiding contradictory recommendations. The evaluation challenge extends beyond traditional benchmarks. Current benchmarks primarily evaluate isolated predictions on static datasets and fail to capture the performance of agents in multistep clinical workflows where intermediate decisions influence subsequent actions [<xref ref-type="bibr" rid="ref34">34</xref>]. Developing assessment frameworks that measure end-to-end task completion, appropriate escalation behavior, and incremental deterioration under uncertainty represents an important research priority. Such frameworks should also address distributional drift, as agents trained with data from specific institutions or populations may exhibit unexpected failures when deployed in different clinical contexts [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>Regulatory frameworks currently lack clarity regarding agent systems [<xref ref-type="bibr" rid="ref36">36</xref>]. Current medical device regulations address static algorithms that produce discrete outputs; agent systems that initiate sequences of actions, access multiple data sources, and modify their behavior based on intermediate outcomes challenge traditional concepts of validation and approval. Determining appropriate regulatory pathways (ie, whether agents should be validated as complete systems or whether their component capabilities should be independently validated) remains an open question requiring collaboration among regulators, developers, and clinical stakeholders.</p><p>Long-term opportunities include developing agents that can explain their reasoning in clinically meaningful terms, moving beyond attention maps and attention weights to generate medically informed rationalizations. Furthermore, enabling agents to work effectively in resource-constrained environments where access to comprehensive electronic health records or high-bandwidth connectivity may be limited could extend their benefits beyond well-resourced health care systems [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. As these technologies mature, research should also address how human-agent collaboration models evolve and ensure that agents support, rather than undermine, clinical expertise.</p></sec><sec id="s5-3"><title>Study Limitations</title><p>This conceptual perspective has several limitations. PARM, as an analytical lens in this paper, has not been empirically validated with controlled trials comparing agent-based and traditional CDSSs. The clinical case examples presented are illustrative examples rather than systems currently used in clinical practice, and their real-world performance must be determined through prospective trials. Our discussion emphasizes technical and managerial considerations without conducting a comprehensive analysis of implementation costs, organizational change management, or clinician training requirements that would significantly impact adoption. Additionally, the rapidly evolving nature of LLM capabilities means that certain technical limitations discussed may be addressed with newer model architectures, but fundamental managerial challenges are likely to persist regardless of the underlying technology.</p></sec><sec id="s5-4"><title>Conclusions</title><p>Medical AI agents extend CDSSs beyond isolated predictions, broadening them toward workflow-driven assistance in which systems can plan tasks, execute controlled actions, evaluate performance, and maintain context over time. Using PARM as an analytical lens, this Viewpoint clarifies how agent capabilities can be made functional across diagnostic reasoning, treatment planning, and longitudinal monitoring while remaining constrained by clinician oversight and institutional governance. From a computational standpoint, the practical impact of agent-based CDSSs will depend on interoperability with clinical information systems; transparent record and audit trails; and assessment approaches that capture end-to-end task reliability, constraint compliance, escalation behavior, and security under deployment shift. Future work should prioritize rigorous implementation studies, standardized reporting of agent behavior, and governance models that support responsible deployment without displacing clinical responsibility. Properly designed medical AI agents can reduce cognitive load and improve the coordination of information and tasks while preserving human judgment as the ultimate authority in patient care.</p></sec></sec></body><back><notes><sec><title>Funding</title><p>The authors declare that this research received no external funding.</p></sec><sec><title>Data Availability</title><p>No primary datasets were generated or analyzed in this Viewpoint. This study is based on published and publicly available sources.</p></sec></notes><fn-group><fn fn-type="con"><p>Data analysis and result interpretation: NA</p><p>Data collection: RD</p><p>Study conception and design: NA, RD</p><p>Writing&#x2014;original draft: NA, RD</p><p>All authors reviewed the results and approved the final version of the manuscript.</p></fn><fn fn-type="conflict"><p>NA is retired and works as a volunteer consultant for Med-International UK Health Agency Ltd. RD is the president of the INVAMED Institute for Medical Innovation.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CDSS</term><def><p>clinical decision support system</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">PARM</term><def><p>planning, action, reflection, and memory</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sutton</surname><given-names>RT</given-names> </name><name name-style="western"><surname>Pincock</surname><given-names>D</given-names> </name><name name-style="western"><surname>Baumgart</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Sadowski</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Fedorak</surname><given-names>RN</given-names> </name><name name-style="western"><surname>Kroeker</surname><given-names>KI</given-names> </name></person-group><article-title>An overview of clinical decision support systems: benefits, risks, and strategies for success</article-title><source>NPJ Digit Med</source><year>2020</year><volume>3</volume><fpage>17</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-0221-y</pub-id><pub-id pub-id-type="medline">32047862</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Esteva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chou</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yeung</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Deep learning-enabled medical computer vision</article-title><source>NPJ Digit Med</source><year>2021</year><month>01</month><day>8</day><volume>4</volume><issue>1</issue><fpage>5</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-00376-2</pub-id><pub-id pub-id-type="medline">33420381</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>Learning the language of life with AI</article-title><source>Science</source><year>2025</year><month>01</month><day>31</day><volume>387</volume><issue>6733</issue><fpage>eadv4414</fpage><pub-id pub-id-type="doi">10.1126/science.adv4414</pub-id><pub-id pub-id-type="medline">39883757</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kohane</surname><given-names>I</given-names> </name></person-group><article-title>Machine learning in medicine</article-title><source>N Engl J Med</source><year>2019</year><month>04</month><day>4</day><volume>380</volume><issue>14</issue><fpage>1347</fpage><lpage>1358</lpage><pub-id pub-id-type="doi">10.1056/NEJMra1814259</pub-id><pub-id pub-id-type="medline">30943338</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ardic</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dinc</surname><given-names>R</given-names> </name></person-group><article-title>Emerging trends in multi-modal artificial intelligence for clinical decision support: a narrative review</article-title><source>Health Informatics J</source><year>2025</year><volume>31</volume><issue>3</issue><fpage>14604582251366141</fpage><pub-id pub-id-type="doi">10.1177/14604582251366141</pub-id><pub-id pub-id-type="medline">40808352</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krones</surname><given-names>F</given-names> </name><name name-style="western"><surname>Marikkar</surname><given-names>U</given-names> </name><name name-style="western"><surname>Parsons</surname><given-names>G</given-names> </name><name name-style="western"><surname>Szmul</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mahdi</surname><given-names>A</given-names> </name></person-group><article-title>Review of multimodal machine learning approaches in healthcare</article-title><source>Inf Fusion</source><year>2025</year><month>02</month><volume>114</volume><fpage>None</fpage><pub-id pub-id-type="doi">10.1016/j.inffus.2024.102690</pub-id><pub-id pub-id-type="medline">41799790</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acosta</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Falcone</surname><given-names>GJ</given-names> </name><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>Multimodal biomedical AI</article-title><source>Nat Med</source><year>2022</year><month>09</month><volume>28</volume><issue>9</issue><fpage>1773</fpage><lpage>1784</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01981-2</pub-id><pub-id pub-id-type="medline">36109635</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>D</given-names> </name><etal/></person-group><article-title>ReAct: synergizing reasoning and acting in language models</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 6, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2210.03629</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>P</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Piktus</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for knowledge-intensive NLP tasks</article-title><source>NIPS &#x2019;20: Proceedings of the 34th International Conference on Neural Information Processing Systems</source><year>2020</year><publisher-name>Curran Associates Inc</publisher-name><fpage>9459</fpage><lpage>9474</lpage><pub-id pub-id-type="doi">10.5555/3495724.3496517</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>C</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>A survey on large language model based autonomous agents</article-title><source>Front Comput Sci</source><year>2024</year><month>03</month><volume>18</volume><fpage>186345</fpage><pub-id pub-id-type="doi">10.1007/s11704-024-40231-1</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>A foundational architecture for AI agents in healthcare</article-title><source>Cell Rep Med</source><year>2025</year><month>10</month><day>21</day><volume>6</volume><issue>10</issue><fpage>102374</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102374</pub-id><pub-id pub-id-type="medline">41015033</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Karunanayake</surname><given-names>N</given-names> </name></person-group><article-title>Next-generation agentic AI for transforming healthcare</article-title><source>Inform Health</source><year>2025</year><month>09</month><volume>2</volume><issue>2</issue><fpage>73</fpage><lpage>83</lpage><pub-id pub-id-type="doi">10.1016/j.infoh.2025.03.001</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hom</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language model influence on diagnostic reasoning: a randomized clinical trial</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2440969</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.40969</pub-id><pub-id pub-id-type="medline">39466245</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Driess</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Towards generalist biomedical AI</article-title><source>NEJM AI</source><year>2024</year><month>02</month><day>22</day><volume>1</volume><issue>3</issue><pub-id pub-id-type="doi">10.1056/AIoa2300138</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for large language models: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 18, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2312.10997</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>IY</given-names> </name><name name-style="western"><surname>Pierson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>S</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ferryman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name></person-group><article-title>Ethical machine learning in healthcare</article-title><source>Annu Rev Biomed Data Sci</source><year>2021</year><month>07</month><volume>4</volume><fpage>123</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1146/annurev-biodatasci-092820-114757</pub-id><pub-id pub-id-type="medline">34396058</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>K</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><etal/></person-group><article-title>LLM-based agentic systems in medicine and healthcare</article-title><source>Nat Mach Intell</source><year>2024</year><volume>6</volume><issue>12</issue><fpage>1418</fpage><lpage>1420</lpage><pub-id pub-id-type="doi">10.1038/s42256-024-00944-1</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aboy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Minssen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Vayena</surname><given-names>E</given-names> </name></person-group><article-title>Navigating the EU AI Act: implications for regulated digital medical products</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>6</day><volume>7</volume><issue>1</issue><fpage>237</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01232-3</pub-id><pub-id pub-id-type="medline">39242831</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McDuff</surname><given-names>D</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Towards accurate differential diagnosis with large language models</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>451</fpage><lpage>457</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08869-4</pub-id><pub-id pub-id-type="medline">40205049</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>O</given-names> </name><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>AI in health and medicine</article-title><source>Nat Med</source><year>2022</year><month>01</month><volume>28</volume><issue>1</issue><fpage>31</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1038/s41591-021-01614-0</pub-id><pub-id pub-id-type="medline">35058619</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaber</surname><given-names>F</given-names> </name><name name-style="western"><surname>Shaik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allega</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>9</day><volume>8</volume><issue>1</issue><fpage>263</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="medline">40346344</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dietrich</surname><given-names>N</given-names> </name></person-group><article-title>Agentic AI in radiology: emerging potential and unresolved challenges</article-title><source>Br J Radiol</source><year>2025</year><month>10</month><day>1</day><volume>98</volume><issue>1174</issue><fpage>1582</fpage><lpage>1584</lpage><pub-id pub-id-type="doi">10.1093/bjr/tqaf173</pub-id><pub-id pub-id-type="medline">40705666</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Kolfschooten</surname><given-names>H</given-names> </name><name name-style="western"><surname>van Oirschot</surname><given-names>J</given-names> </name></person-group><article-title>The EU Artificial Intelligence Act (2024): implications for healthcare</article-title><source>Health Policy</source><year>2024</year><month>11</month><volume>149</volume><fpage>105152</fpage><pub-id pub-id-type="doi">10.1016/j.healthpol.2024.105152</pub-id><pub-id pub-id-type="medline">39244818</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Labkoff</surname><given-names>S</given-names> </name><name name-style="western"><surname>Oladimeji</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kannry</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward a responsible future: recommendations for AI-enabled clinical decision support</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>11</month><day>1</day><volume>31</volume><issue>11</issue><fpage>2730</fpage><lpage>2739</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae209</pub-id><pub-id pub-id-type="medline">39325508</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hassan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Borycki</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Kushniruk</surname><given-names>AW</given-names> </name></person-group><article-title>Artificial intelligence governance framework for healthcare</article-title><source>Healthc Manage Forum</source><year>2025</year><month>03</month><volume>38</volume><issue>2</issue><fpage>125</fpage><lpage>130</lpage><pub-id pub-id-type="doi">10.1177/08404704241291226</pub-id><pub-id pub-id-type="medline">39470044</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shneiderman</surname><given-names>B</given-names> </name></person-group><article-title>Human-centered artificial intelligence: reliable, safe &#x0026; trustworthy</article-title><source>Int J Hum Comput Interact</source><year>2020</year><volume>36</volume><issue>6</issue><fpage>495</fpage><lpage>504</lpage><pub-id pub-id-type="doi">10.1080/10447318.2020.1741118</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Amershi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Weld</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vorvoreanu</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Guidelines for human-AI interaction</article-title><source>CHI &#x2019;19: Proceedings of the 2019 CHI Conference on Human Factors in Computing Systems</source><year>2019</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1145/3290605.3300233</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>Nat Med</source><year>2022</year><month>05</month><volume>28</volume><issue>5</issue><fpage>924</fpage><lpage>933</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id><pub-id pub-id-type="medline">35585198</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="web"><article-title>WHO releases AI ethics and governance guidance for large multi-modal models</article-title><source>World Health Organization</source><year>2024</year><access-date>2026-01-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/news/item/18-01-2024-who-releases-ai-ethics-and-governance-guidance-for-large-multi-modal-models">https://www.who.int/news/item/18-01-2024-who-releases-ai-ethics-and-governance-guidance-for-large-multi-modal-models</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lawton</surname><given-names>T</given-names> </name><name name-style="western"><surname>Morgan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Porter</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Clinicians risk becoming &#x201C;liability sinks&#x201D; for artificial intelligence</article-title><source>Future Healthc J</source><year>2024</year><month>02</month><volume>11</volume><issue>1</issue><fpage>100007</fpage><pub-id pub-id-type="doi">10.1016/j.fhj.2024.100007</pub-id><pub-id pub-id-type="medline">38646041</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Karthikesalingam</surname><given-names>A</given-names> </name><name name-style="western"><surname>Suleyman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Corrado</surname><given-names>G</given-names> </name><name name-style="western"><surname>King</surname><given-names>D</given-names> </name></person-group><article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title><source>BMC Med</source><year>2019</year><month>10</month><day>29</day><volume>17</volume><issue>1</issue><fpage>195</fpage><pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id><pub-id pub-id-type="medline">31665002</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DS</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Black</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Geng</surname><given-names>G</given-names> </name><etal/></person-group><article-title>MedAgentBench: a virtual EHR environment to benchmark medical LLM agents</article-title><source>NEJM AI</source><year>2025</year><month>08</month><day>14</day><volume>2</volume><issue>9</issue><pub-id pub-id-type="doi">10.1056/AIdbp2500144</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koch</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Baumgartner</surname><given-names>CF</given-names> </name><name name-style="western"><surname>Berens</surname><given-names>P</given-names> </name></person-group><article-title>Distribution shift detection for the postmarket surveillance of medical AI algorithms: a retrospective simulation study</article-title><source>NPJ Digit Med</source><year>2024</year><month>05</month><day>9</day><volume>7</volume><issue>1</issue><fpage>120</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01085-w</pub-id><pub-id pub-id-type="medline">38724581</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><article-title>2025 watch list: artificial intelligence in health care</article-title><source>Can J Health Technol</source><year>2025</year><volume>5</volume><issue>3</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.51731/cjht.2025.1103</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Oakden-Rayner</surname><given-names>L</given-names> </name><name name-style="western"><surname>Beam</surname><given-names>AL</given-names> </name></person-group><article-title>The false hope of current approaches to explainable artificial intelligence in health care</article-title><source>Lancet Digit Health</source><year>2021</year><month>11</month><volume>3</volume><issue>11</issue><fpage>e745</fpage><lpage>e750</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(21)00208-9</pub-id><pub-id pub-id-type="medline">34711379</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rudin</surname><given-names>C</given-names> </name></person-group><article-title>Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead</article-title><source>Nat Mach Intell</source><year>2019</year><month>05</month><volume>1</volume><issue>5</issue><fpage>206</fpage><lpage>215</lpage><pub-id pub-id-type="doi">10.1038/s42256-019-0048-x</pub-id><pub-id pub-id-type="medline">35603010</pub-id></nlm-citation></ref></ref-list></back></article>