<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e91618</article-id><article-id pub-id-type="doi">10.2196/91618</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Prediction Models for In-Hospital Delirium Using Routinely Collected Electronic Health Record Data: Systematic Review</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Huang</surname><given-names>Hung-Min</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lu</surname><given-names>Chun-Shun</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chang</surname><given-names>Geng-Wei</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tien</surname><given-names>Ming-Hsu</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hsu</surname><given-names>Yu-Kai</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Institute of Health Informatics, University College London</institution><addr-line>Gower Street</addr-line><addr-line>London</addr-line><addr-line>England</addr-line><country>United Kingdom</country></aff><aff id="aff2"><institution>Department of General Medicine, MacKay Memorial Hospital</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><aff id="aff3"><institution>Department of General Medicine, Chang Gung Memorial Hospital</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><aff id="aff4"><institution>Department of General Medicine, Far Eastern Memorial Hospital</institution><addr-line>Taipei</addr-line><country>Taiwan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Tai</surname><given-names>Andy</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Tian</surname><given-names>Fangying</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Hung-Min Huang, MD, Institute of Health Informatics, University College London, Gower Street, London, England, WC1E 6BT, United Kingdom, 44 7570909461; <email>rmhihhu@ucl.ac.uk</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>16</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e91618</elocation-id><history><date date-type="received"><day>17</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>18</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>14</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Hung-Min Huang, Chun-Shun Lu, Geng-Wei Chang, Ming-Hsu Tien, Yu-Kai Hsu. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 16.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e91618"/><abstract><sec><title>Background</title><p>Delirium is a common and clinically important form of acute in-hospital mental status deterioration. Electronic health record (EHR)&#x2013;based prediction models may support early identification and targeted prevention, but their methodological quality, validation rigor, and clinical readiness remain uncertain.</p></sec><sec><title>Objective</title><p>This systematic review aimed to synthesize and critically evaluate prediction models for in-hospital delirium developed using routinely collected EHR data, focusing on model characteristics, validation strategies, performance, risk of bias, and clinical applicability.</p></sec><sec sec-type="methods"><title>Methods</title><p>We searched PubMed, MEDLINE, Embase, PsycINFO, and Web of Science from inception to November 11, 2025. Eligible studies developed, validated, or evaluated multivariable prediction models using routinely collected EHR or administrative data to predict acute mental status deterioration during adult hospital admissions. Although eligibility criteria were broad, all included studies operationalized deterioration as delirium. Data extraction was informed by CHARMS (Checklist for Critical Appraisal and Data Extraction for Systematic Reviews of Prediction Modeling Studies) and TRIPOD (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis) or TRIPOD&#x2013;artificial intelligence guidance. Model performance, validation, calibration, and implementation features were synthesized narratively. Risk of bias and applicability were assessed using PROBAST (Prediction Model Risk of Bias Assessment Tool).</p></sec><sec sec-type="results"><title>Results</title><p>Twenty-nine studies met the inclusion criteria. The evidence clustered into 4 overlapping prediction tasks: admission or early-stay risk stratification, perioperative or postoperative prediction, dynamic intensive care unit prediction, and external validation or workflow evaluation of existing tools. Most studies were retrospective cohorts (20/29, 69%) and were conducted in general ward, mixed ward&#x2013;intensive care unit, intensive care unit, or emergency department settings. Machine learning or hybrid approaches were common (18/29, 62%), but more complex models did not consistently outperform statistical or rule-based approaches. Of 29 studies, internal discrimination was reported in 24 (83%; area under the receiver operating characteristic curve range 0.77-0.97) studies, whereas external discrimination was reported in 12 studies and calibration in 15 studies. Decision curve analysis was reported in 3 studies, and prospective evaluation or workflow integration remained limited. Overall risk of bias was low in 8 studies, unclear in 10 studies, and high in 11 studies, mainly because of analysis-domain limitations.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Routinely collected EHR data can support delirium risk prediction across hospital settings, and many models show moderate to high discrimination. However, no single algorithm is ready for routine adoption. The field remains limited by heterogeneous prediction tasks, inconsistent outcome ascertainment, weak calibration and decision-analytic reporting, and insufficient external or prospective evaluation. Future studies should define the intended clinical use case before model development, evaluate calibration and clinical usefulness alongside discrimination, and test models across institutions, time periods, and workflows before deployment.</p></sec></abstract><kwd-group><kwd>delirium</kwd><kwd>risk prediction</kwd><kwd>electronic health records</kwd><kwd>machine learning</kwd><kwd>clinical decision support</kwd><kwd>inpatients</kwd><kwd>systematic review</kwd><kwd>PROBAST</kwd><kwd>TRIPOD</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Acute in-hospital mental status deterioration is a clinically important manifestation of acute brain dysfunction, encompassing disturbances such as confusion, inattention, agitation, and reduced consciousness. In current hospital-based prediction modeling research using routinely collected electronic health record (EHR) data, however, this construct has been operationalized almost exclusively as delirium. Delirium is common across hospital settings and is associated with increased morbidity, mortality, prolonged hospitalization, institutionalization, and persistent cognitive impairment following discharge [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>].</p><p>Delirium is the most clinically and methodologically established target in this literature. Its prominence reflects both its prognostic significance and the availability of validated bedside assessment tools, most notably the Confusion Assessment Method (CAM) and its intensive care unit (ICU) adaptations [<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>Prediction modeling for in-hospital delirium is motivated by the need to support early identification and prevention in resource-constrained clinical environments. Universal application of intensive preventive strategies is rarely feasible, and risk stratification tools that identify patients at elevated risk early in the hospital course offer a pragmatic approach to targeting preventive interventions [<xref ref-type="bibr" rid="ref4">4</xref>]. Advances in EHR-based modeling, including machine learning and natural language processing (NLP), have enabled scalable development of delirium prediction models using routinely collected clinical data [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>However, recent systematic reviews have highlighted important limitations in the existing evidence base. Although many delirium prediction models report moderate to high discrimination, substantial heterogeneity exists in outcome definitions, predictor handling, validation strategies, and reporting quality. External validation, calibration assessment, and prospective evaluation remain inconsistently performed, limiting confidence in clinical generalizability and real-world usefulness [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Against this background, the present systematic review aims to synthesize and critically evaluate prediction models for in-hospital delirium using routinely collected EHR data. Although the search strategy was intentionally broad to capture a range of acute mental status outcomes, all eligible studies identified at full-text review operationalized deterioration as delirium. This finding underscores the central role of delirium as the dominant and currently most clinically actionable manifestation of acute in-hospital mental status change within the prediction modeling literature. Accordingly, this review focuses on delirium prediction models, with particular emphasis on methodological quality, validation practices, risk of bias, and implications for the development of clinically actionable decision support tools.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Reporting Standards</title><p>This systematic review was conducted and reported in accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) guidelines [<xref ref-type="bibr" rid="ref9">9</xref>]. Given the focus on prediction model development and validation, data extraction and synthesis were additionally informed by relevant items from the TRIPOD (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis) and TRIPOD&#x2013;Artificial Intelligence (TRIPOD-AI) reporting frameworks to support structured evaluation of model characteristics, validation strategies, and reporting quality [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. This review was not prospectively registered in PROSPERO because protocol registration was not completed before screening had begun; this is acknowledged as a limitation.</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>Studies were eligible for inclusion if they met the following criteria: (1) involved adult patients (aged &#x2265;18 years) admitted to hospital settings, including ICUs, general medical or surgical wards, or mixed ICU&#x2013;ward cohorts; (2) developed, validated, or updated multivariable prediction models using routinely collected EHR or administrative data; (3) predicted delirium occurring during the index hospital admission; and (4) reported sufficient methodological or performance information to allow data extraction.</p><p>Eligible outcomes included delirium occurring during the index hospital admission, identified using validated clinical assessment tools (eg, CAM and CAM-ICU), diagnostic codes, or structured chart review. Although the search strategy was designed to capture a broader range of acute mental status deterioration outcomes, all studies meeting inclusion criteria at full-text review predicted delirium. Therefore, delirium was treated as the primary outcome for this review.</p><p>Studies using any statistical or machine learning&#x2013;based modeling approach were eligible, including traditional regression-based models and advanced machine learning methods. Both retrospective and prospective study designs were included. Studies were excluded if they (1) focused exclusively on pediatric populations; (2) were conducted in outpatient, community, or psychiatric clinic settings without hospital admission; (3) relied solely on nonroutine data sources such as imaging, genomics, or specialized psychometric instruments not typically available in EHR systems; (4) predicted outcomes defined at the population level or occurring outside the index hospital admission (eg, long-term suicide risk or postdischarge psychiatric readmission); or (5) did not report delirium or an equivalent acute in-hospital mental status outcome.</p></sec><sec id="s2-3"><title>Information Sources and Search Strategy</title><p>A comprehensive literature search was conducted to identify studies developing or validating prediction models for delirium and related acute mental status outcomes occurring during hospitalization. The search strategy was designed a priori to be intentionally broad to maximize sensitivity and capture models targeting a range of clinically relevant mental status changes. However, all studies meeting inclusion criteria at full-text review focused exclusively on delirium. This reflects the predominance of delirium as the most consistently defined and operationalized outcome in this field, and the present review therefore focuses specifically on delirium prediction models.</p><p>The following electronic databases were searched from inception to November 11, 2025: PubMed or MEDLINE, Embase, PsycINFO, and Web of Science. The search combined terms related to prediction modeling and machine learning (eg, prediction, risk model, machine learning, and AI), routinely collected health care data (eg, EHRs, administrative data, and claims data), acute mental status outcomes (eg, delirium, acute confusion, agitation, mental status change, and psychotropic medication use), and hospital settings (eg, inpatient, ward, and ICU).</p><p>Database-specific search strategies were developed using a combination of controlled vocabulary (eg, MeSH) and free-text terms. Full search strategies for each database are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. No restrictions were placed on geographic location or health care setting. Only studies published in English were included. Reference lists of included studies and relevant review papers were manually screened to identify additional eligible studies. All retrieved records were imported into reference management software, and duplicate records were removed before screening.</p></sec><sec id="s2-4"><title>Study Selection</title><p>After removal of duplicate records, titles and abstracts were screened to identify potentially eligible studies. Full-text papers were retrieved for all records deemed potentially relevant and assessed against the predefined eligibility criteria. Study selection was performed independently by 2 reviewers (CSL and GWC). Discrepancies at either the title or abstract or full-text screening stage were resolved through discussion and consensus, with consultation of a third reviewer (MHT) when necessary. The overall study selection process is summarized using a PRISMA flow diagram.</p></sec><sec id="s2-5"><title>Data Extraction</title><p>Data were extracted independently by 2 reviewers (CSL and GWC) using a standardized data extraction form developed a priori. The extraction framework was informed by established guidance for prediction model studies, including the CHARMS (Checklist for Critical Appraisal and Data Extraction for Systematic Reviews of Prediction Modelling Studies), and reporting domains from the TRIPOD and TRIPOD-AI statements [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>For each included study, extracted information included study metadata (first author, year, country, clinical setting, and study design), population characteristics (sample size and cohort definition), outcome definition and ascertainment method (eg, CAM-based assessment, <italic>ICD</italic> [<italic>International Classification of Diseases</italic>] codes, and chart review), prediction time horizon, modeling approach, predictor variables, feature selection methods, and model presentation.</p><p>Model development and validation characteristics were extracted in detail, including type of validation (internal, temporal, or external), validation datasets, and reported performance measures. Extracted performance metrics included measures of discrimination (eg, area under the receiver operating characteristic curve [AUROC]), sensitivity, specificity, and predictive values where available. Calibration measures and decision-analytic measures (eg, decision curve analysis) were recorded when reported. Information on missing data handling was extracted for each study, including use of complete-case analysis, single or multiple imputation, or algorithm-intrinsic handling of missingness. Where missing data handling was not explicitly described, this was recorded as not reported.</p></sec><sec id="s2-6"><title>Risk of Bias and Quality Assessment</title><p>Risk of bias and applicability of included studies were assessed using the PROBAST (Prediction Model Risk of Bias Assessment Tool), which evaluates 4 domains: participants, predictors, outcomes, and analysis [<xref ref-type="bibr" rid="ref13">13</xref>]. Two reviewers (CSL and GWC) independently completed PROBAST assessments using the tool&#x2019;s signaling questions. Each domain was rated as low, high, or unclear risk of bias. Discrepancies were resolved through discussion and consensus, with involvement of a third reviewer (MHT) when necessary. PROBAST assessments were used to support qualitative interpretation of methodological strengths and limitations rather than to exclude studies or weight quantitative comparisons.</p></sec><sec id="s2-7"><title>Data Synthesis and Analysis</title><p>Given substantial heterogeneity across studies with respect to clinical setting, population characteristics, outcome definitions, prediction horizons, modeling approaches, and validation strategies, a quantitative meta-analysis of model performance was not performed. Findings were therefore synthesized narratively using a structured framework focused on 4 questions: What clinical prediction task was being addressed? What routinely collected data were used? How well models performed under internal and external validation? How close the models were to clinically reliable implementation.</p><p>Prediction model characteristics and performance metrics were summarized descriptively. Performance differences were interpreted in relation to clinical setting, prediction horizon, outcome ascertainment, validation strategy, calibration reporting, and risk of bias rather than by algorithm type alone. This approach was chosen to distinguish technical feasibility from clinical readiness.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>The database search identified a total of 1086 records, including 274 from PubMed, 373 from Embase, 284 from Web of Science, and 155 from PsycINFO. After removal of duplicate and overlapping records, 673 unique records remained for title and abstract screening, of which 603 were excluded. Seventy reports were retrieved for full-text assessment. Following full-text review, 41 reports were excluded. The most common reasons for exclusion were the absence of a multivariable prediction model, outcomes not occurring during the index hospital admission, wrong publication type, wrong outcome definition, or an ineligible study population. A total of 29 studies met the inclusion criteria and were included in the final qualitative synthesis. The study selection process is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram of study selection [<xref ref-type="bibr" rid="ref9">9</xref>].</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91618_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The 29 included studies did not represent a single uniform prediction problem. Instead, they formed a heterogeneous evidence base spanning early admission screening, perioperative risk prediction, short-term ICU warning systems, and external validation or prospective evaluation of existing tools. Country, center structure, and study design are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. The narrative synthesis highlights the patterns most relevant to interpretation.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Study characteristics (country, center, and study design).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Country</td><td align="left" valign="bottom">Centers</td><td align="left" valign="bottom">Study design</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">Netherlands</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Switzerland</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Lebanon/United States</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective case&#x2013;control</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">South Korea</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Austria</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Prospective cohort</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Austria</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Mixed retrospective&#x2013;prospective</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Austria</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Prospective cohort</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">South Korea</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">China</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Canada</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Japan</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">South Korea</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Mixed retrospective&#x2013;prospective</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Prospective cohort</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Switzerland</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Prospective cohort</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Mixed retrospective&#x2013;prospective</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Mixed (multicenter+external validation)</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Germany</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Retrospective cohort</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Germany</td><td align="left" valign="top">Multicenter</td><td align="left" valign="top">Mixed retrospective&#x2013;prospective</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Single-center</td><td align="left" valign="top">Retrospective cohort</td></tr></tbody></table></table-wrap><p>Most studies used retrospective cohort designs (20/29, 69%) [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. Of 29 studies, prospective cohorts were less common (4/29, 14%) [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], 4 (14%) studies combined retrospective development with prospective validation or evaluation [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>], and 1 (3%) study used a retrospective case-control design [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>This design profile is important because much of the literature remains closer to model development than to implementation science. The strongest clinical evidence generally came from studies that moved beyond internal validation into external, temporal, or prospective evaluation.</p><p>Clinical setting, population focus, and sample size are summarized in <xref ref-type="table" rid="table2">Table 2</xref>. Clinical settings were unevenly represented. Of 29 studies, general ward populations accounted for 13 (45%) studies, mixed ward-ICU cohorts accounted for 8 (28%), ICU-only cohorts accounted for 7 (24%), and the emergency department accounted for 1 (3%). Fifteen (52%) studies were single-center, while 14 (48%) studies used multicenter data or combined multicenter development with external validation.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Clinical setting, population focus, and sample size.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Clinical setting</td><td align="left" valign="bottom">Population focus</td><td align="left" valign="bottom">Total, N</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Older adult inpatients (&#x2265;60 years)</td><td align="left" valign="top">1168 patients</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">Emergency department</td><td align="left" valign="top">Older adult inpatients (&#x2265;65 years)</td><td align="left" valign="top">44,578 patients</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Mixed ICU<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> and ward</td><td align="left" valign="top">Adult surgical inpatients</td><td align="left" valign="top">24,885 encounters</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Adult medical inpatients (COVID-19)</td><td align="left" valign="top">2907 patients</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult rehabilitation inpatients</td><td align="left" valign="top">8774 stays</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">104,303 patients</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">13,395 patients</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">41,826 patients</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult surgical inpatients</td><td align="left" valign="top">51,457 patients</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Adult inpatients (ICU and ward)</td><td align="left" valign="top">40,208 admissions</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Older adult surgical inpatients (&#x2265;50 years)</td><td align="left" valign="top">14,334 encounters</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">12,409 patients</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">4663 patients</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult trauma surgery inpatients</td><td align="left" valign="top">93 patients</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult surgical inpatients</td><td align="left" valign="top">738 patients</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Older adult orthopedic surgery inpatients</td><td align="left" valign="top">3980 patients</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult cardiac surgery inpatients</td><td align="left" valign="top">507 patients</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">34,035 patients</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">38,426 patients</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult surgical inpatients</td><td align="left" valign="top">11,863 patients</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">3284 patients</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Older adult inpatients (ED<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> admission)</td><td align="left" valign="top">28,531 patients</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Older adult inpatients (&#x2265;50 years)</td><td align="left" valign="top">8055 patients</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Older adult surgical inpatients (&#x2265;60 years)</td><td align="left" valign="top">866 patients</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Older adult inpatients</td><td align="left" valign="top">27,871 patients</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">ICU</td><td align="left" valign="top">Adult ICU patients</td><td align="left" valign="top">22,840 patients</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">NR<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Mixed ICU and ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">NR</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">General ward</td><td align="left" valign="top">Adult general inpatients</td><td align="left" valign="top">18,223</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ICU: intensive care unit.</p></fn><fn id="table2fn2"><p><sup>b</sup>ED: emergency department.</p></fn><fn id="table2fn3"><p><sup>c</sup>NR: not reported.</p></fn></table-wrap-foot></table-wrap><p>Across all studies, adult inpatients constituted the primary population of interest, but the intended use cases differed. Some models were designed for broad hospital screening, some for older adult or surgical pathways, and others for high-frequency monitoring in intensive care. Sample sizes ranged from fewer than 100 patients in a small prospective deployment cohort [<xref ref-type="bibr" rid="ref27">27</xref>] to more than 100,000 patients in large multidatabase studies [<xref ref-type="bibr" rid="ref19">19</xref>], making direct comparison of performance estimates difficult.</p><p>Outcome definitions, assessment methods, and prevalence are summarized in <xref ref-type="table" rid="table3">Table 3</xref>. Delirium was the outcome across all included studies, but the clinical meaning of the outcome varied. Of the 29 studies, 27 (93%) modeled incident delirium, 1 (3%) modeled prevalent delirium in the emergency department [<xref ref-type="bibr" rid="ref15">15</xref>], and 1 (3%) considered any delirium regardless of timing [<xref ref-type="bibr" rid="ref38">38</xref>]. Outcome labels included in-hospital delirium, postoperative delirium, ICU delirium, and recurrent or short-term delirium risk.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Outcome definition, assessment method, and prevalence.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Outcome label</td><td align="left" valign="bottom">Assessment method</td><td align="left" valign="bottom">Outcome prevalence</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">Chart review or adjudication</td><td align="left" valign="top">75/1345 (5.6%)</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">CAM<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>-based screening</td><td align="left" valign="top">1701/44,578 (3.8%)</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">Nurse routine screening (CAM-based)</td><td align="left" valign="top">1327/24,885 (5.3%)</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">EHR<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>-derived (codes or NLP<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>)</td><td align="left" valign="top">488/2907 (16.8%)</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">EHR-derived+chart review</td><td align="left" valign="top">125/8774 (1.4%)</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">CAM-ICU<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">Reported without extractable event count</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">ICU delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">12,871/56,297 windows (23%)</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">Nurse routine screening (CAM-based)</td><td align="left" valign="top">3499/64,038 visits (5.5%)</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">ICD<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> codes</td><td align="left" valign="top">1608/51,457 (3.1%)</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">ICU delirium or In-hospital delirium</td><td align="left" valign="top"><italic>ICD</italic> codes+chart review</td><td align="left" valign="top">NR<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">CAM-based+<italic>ICD</italic> codes</td><td align="left" valign="top">7198/39,968 raw cohort (18.0%)</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">ICU delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">Reported by dataset; event counts NR</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top"><italic>ICD</italic> codes+EHR text review</td><td align="left" valign="top">81/5530 (1.5%)</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">Clinical judgment/chart review</td><td align="left" valign="top">NR clinical cohort; 347/5347 external test set</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">DOS<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">103/738 (14%)</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">DSM<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup>-based diagnosis+EHR-derived</td><td align="left" valign="top">196/3980 (4.9%)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">141/507 (28%)</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">Positive CAM assessment rate reported; event count NR</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">ICU delirium</td><td align="left" valign="top">ICDSC<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">Episode prevalence reported; event count NR</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">CAM-based screening</td><td align="left" valign="top">592/6497 derivation (9.1%); 427/5366 validation (8.0%)</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">ICU delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">688/3284 (21%)</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">DOS/CAM-ICU</td><td align="left" valign="top">8057/28,351 (28.4%)</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">CAM-based screening</td><td align="left" valign="top">1107/8055 (13.7%)</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Postoperative delirium</td><td align="left" valign="top">DOS+<italic>ICD</italic> codes</td><td align="left" valign="top">100/866 (11.5%)</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top">DSM-based diagnosis+EHR-derived</td><td align="left" valign="top">2343/27,625 retrospective (8%); 43/246 prospective incident (19%)</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">CAM-ICU</td><td align="left" valign="top">NR</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top"><italic>ICD</italic> codes</td><td align="left" valign="top">NR</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">In-hospital delirium</td><td align="left" valign="top"><italic>ICD</italic> codes</td><td align="left" valign="top">NR</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Incident delirium</td><td align="left" valign="top">Nurse routine screening (CAM-based)</td><td align="left" valign="top">878/18,223 (4.8%)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>CAM: Confusion Assessment Method. </p></fn><fn id="table3fn2"><p><sup>b</sup>EHR: electronic health record. </p></fn><fn id="table3fn3"><p><sup>c</sup>NLP: natural language processing.</p></fn><fn id="table3fn4"><p><sup>d</sup>ICU: intensive care unit.</p></fn><fn id="table3fn5"><p><sup>e</sup><italic>ICD</italic>: <italic>International Classification of Diseases</italic>.</p></fn><fn id="table3fn6"><p><sup>f</sup>NR: not reported.</p></fn><fn id="table3fn7"><p><sup>g</sup>DOS: Delirium Observation Screening Scale.</p></fn><fn id="table3fn8"><p><sup>h</sup>DSM: Diagnostic and Statistical Manual of Mental Disorders.</p></fn><fn id="table3fn9"><p><sup>i</sup>ICDSC: Intensive Care Delirium Screening Checklist.</p></fn></table-wrap-foot></table-wrap><p>Outcome ascertainment was a major source of heterogeneity. Studies used structured tools such as CAM, CAM-ICU, the Delirium Observation Screening Scale, or the Intensive Care Delirium Screening Checklist, as well as <italic>ICD</italic> codes, EHR-derived definitions, NLP-enhanced ascertainment, and manual chart review or adjudication. These approaches identify overlapping but not identical clinical events, which limits the interpretability of pooled performance comparisons.</p><p>Outcome prevalence also varied substantially. Among the 20 studies with clearly extractable prevalence percentages, 6 (30%) reported prevalence below 5%, 10 (50%) reported prevalence between 5% and 20%, and 4 (20%) reported prevalence above 20%. Low prevalence was typical in general ward and broad inpatient cohorts, whereas higher prevalence was more common in ICU, surgical, and selected high-risk cohorts.</p><p>This variation has direct implications for clinical interpretation. In low-prevalence settings, even a model with good AUROC may have low positive predictive value (PPV) and may generate many false-positive alerts. In higher-prevalence settings, the same threshold can produce a very different balance between missed cases and unnecessary intervention.</p><p>Prediction horizon and TRIPOD classification are summarized in <xref ref-type="table" rid="table4">Table 4</xref>. Prediction horizons further separated the studies into different clinical tasks. Some models estimated risk at admission or during the first 24&#x2010;72 hours, some predicted postoperative delirium over a fixed perioperative period, and others updated risk dynamically using rolling ICU or hospital windows. These are not interchangeable tasks: short-term dynamic prediction benefits from more proximal clinical signals, whereas admission-time models must support earlier but less certain prevention decisions.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Prediction horizon and TRIPOD<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> classification.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Prediction horizon</td><td align="left" valign="bottom">TRIPOD type</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 4</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">During ED<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> stay</td><td align="left" valign="top">TRIPOD 4</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 1a</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">During ICU<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup> stay</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Dynamic short-term window (&#x2264;24 hours)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">During ICU stay/During hospitalization</td><td align="left" valign="top">TRIPOD 2a</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;24&#x2010;72 hours)</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;24&#x2010;72 hours)</td><td align="left" valign="top">TRIPOD 3</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Dynamic short-term window (&#x2264;24 hours)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Dynamic short-term window (&#x2264;24 hours)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">During ICU stay</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;24&#x2010;72 hours)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 4</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Postoperative period (&#x2264;7 days)</td><td align="left" valign="top">TRIPOD 4</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Dynamic short-term window (&#x2264;48 hours)</td><td align="left" valign="top">TRIPOD 1b</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2b</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">During hospitalization</td><td align="left" valign="top">TRIPOD 2a</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>TRIPOD: Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis.</p></fn><fn id="table4fn2"><p><sup>b</sup>ED: emergency department.</p></fn><fn id="table4fn3"><p><sup>c</sup>ICU: intensive care unit.</p></fn></table-wrap-foot></table-wrap><p>Accordingly, apparent differences in discrimination should not be interpreted as simple evidence that one model family is superior. They may instead reflect differences in timing, population acuity, outcome prevalence, and availability of predictors close to delirium onset.</p><p>TRIPOD classifications also reflected a field still weighted toward development and validation rather than deployment. Type 1b and 2b studies predominated, while fewer studies focused on external validation, model updating, temporal validation, or prospective implementation.</p><p>Taken together, the study characteristics show that the literature is clinically broad but methodologically fragmented. The central synthesis question is therefore not only whether EHR-based delirium prediction is feasible but which models have been tested under conditions close enough to their intended clinical use.</p></sec><sec id="s3-3"><title>Model Development and Predictors</title><p>Model categories are summarized in <xref ref-type="table" rid="table5">Table 5</xref>, with detailed model development and predictor characteristics provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref42">42</xref>]. Machine learning models were the largest single group (10/29, 34%), followed by statistical or rule-based approaches (8/29, 28%), deep learning models (5/29, 17%), and hybrid approaches combining statistical, machine learning, or deep learning components (6/29, 21%). Tree-based ensemble methods such as random forests and gradient boosting were common, while deep learning was concentrated in ICU or large-scale EHR datasets.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Model category across included studies.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Model category</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">Existing rule-based statistical model (external validation only)</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">Not applicable (model validation study; no development)</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Classical machine learning and neural network models with clinical score comparator</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Penalized logistic regression with simple clinical comparators</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Multivariable logistic regression</td></tr><tr><td align="left" valign="top">Contreras et al (2025)[<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Deep learning models including transformers and large language models</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Tree-based machine learning and recurrent neural networks</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Tree-based machine learning (Random Forest)</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Classical machine learning and statistical models</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Classical and ensemble machine learning models</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Classical machine learning and neural networks</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Classical machine learning and deep neural networks</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Tree-based machine learning with multiple outcome models</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Tree-based machine learning</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Tree-based machine learning</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Gradient boosting models</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Classical machine learning models</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Hybrid machine learning and deep learning models</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Recurrent neural networks</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Gradient boosting and penalized regression</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Multivariable logistic regression</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Classical machine learning models</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Rule-based model with statistical recalibration</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">External validation of existing model</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Rule-based statistical model</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Deep learning with attention mechanisms</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Transformer-based deep learning</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Transformer-based deep learning</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Classical machine learning and statistical models</td></tr></tbody></table></table-wrap><p>Several studies compared multiple modeling paradigms within the same dataset [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. These comparisons did not show a consistent advantage for increasing model complexity. In some cohorts, tree-based or penalized regression models performed similarly to neural network approaches, and apparent internal advantages were not always preserved during external validation.</p><p>Predictor sets were built mainly from structured EHR data. Of 29 studies, demographics were included in 28 (97%), comorbidities or diagnostic history in 25 (86%) , laboratory values in 22 (76%), medications in 20 (69%), vital signs in 17 (59%), and nursing or care process variables in 7 (24%). This pattern indicates that most models relied on broadly available routine data rather than specialized research measurements.</p><p>Temporal handling of predictors was a key distinction between models (<xref ref-type="table" rid="table6">Table 6</xref>). Of 29 studies, 15 (52%) used static single-time snapshots, most often at admission, preoperatively, or within an early fixed window. Of 29 studies, 14 (48%) incorporated time-varying information through repeated recalculation, aggregated temporal summaries, rolling windows, or sequential time series modeling.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Predictor window and temporal handling.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Predictor window</td><td align="left" valign="bottom">Temporal handling</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">Admission-time or early admission window (&#x2264;24 hours)</td><td align="left" valign="top">Static with repeated recalculation</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">ED<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> presentation window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Preoperative window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Preadmission+ early admission window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Admission-time window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Early ICU<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> window (&#x2264;24 hours)</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Mixed static+short-term dynamic ICU window</td><td align="left" valign="top">Hybrid static+temporal modeling</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Dynamic ICU window (until outcome)</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Preoperative window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Dynamic ICU window or dynamic hospital window</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Preadmission window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Early ICU window (&#x2264;24 hours)</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Admission-time+early admission window</td><td align="left" valign="top">Static with repeated recalculation</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Preadmission+early admission window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;48 hours)</td><td align="left" valign="top">Static with repeated recalculation</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Preoperative window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Preoperative+perioperative+early ICU window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Short-term dynamic window (&#x2264;24 hours)</td><td align="left" valign="top">Hybrid static+temporal modeling</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Mixed static+dynamic ICU window</td><td align="left" valign="top">Sliding or rolling time windows</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Admission-time+perioperative window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Early ICU window (&#x2264;24 hours)</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">ED presentation+early admission window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Admission-time or early admission window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Preoperative window</td><td align="left" valign="top">Static (single-time snapshot)</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;24 hours)</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Short-term dynamic window (&#x2264;24 hours)</td><td align="left" valign="top">Sequential time series modeling</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Admission-time+dynamic hospital window</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Dynamic hospital window (entire stay)</td><td align="left" valign="top">Aggregated temporal summaries</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Early admission window (&#x2264;24 hours)</td><td align="left" valign="top">Static (single-time snapshot)</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>ED: emergency department.</p></fn><fn id="table6fn2"><p><sup>b</sup>ICU: intensive care unit.</p></fn></table-wrap-foot></table-wrap><p>Of 29 studies, only 8 (28%) used NLP or unstructured free-text data [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. In several of these, NLP supported outcome ascertainment rather than contributing predictor features. Thus, despite the clinical importance of narrative documentation for cognitive and behavioral change, most models remained dependent on structured EHR fields.</p><p>Reporting of missing data handling, class imbalance, and interpretability was uneven (<xref ref-type="table" rid="table7">Table 7</xref>). Missing data handling was not reported in 24% (7/29) of the studies, and class imbalance handling was absent or not reported in 55% (16/29) of the studies. Interpretability was addressed most often through post hoc feature attribution (14/29, 48%) or intrinsically interpretable models (7/29, 24%), but these explanations were rarely linked to clinical workflow decisions.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Missing data handling and model interpretability.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Missing data handling</td><td align="left" valign="bottom">Interpretability approach</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">No imputation or model design&#x2013;based</td><td align="left" valign="top">None or not reported</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">Hybrid (intrinsic+post hoc)</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Simple imputation+complete-case</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Exclusion-based (no imputation)</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Simple imputation+missingness indicators</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Time series imputation</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Model-intrinsic handling</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Imputation (method not specified)</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Advanced imputation (DL<sup><xref ref-type="table-fn" rid="table7fn1">a</xref></sup>-based)</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Exclusion-based (no imputation)</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">None or not reported</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Limited or unclear</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Model-intrinsic handling</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Simple imputation+exclusions</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Simple imputation + missingness indicators</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Advanced imputation (ML<sup><xref ref-type="table-fn" rid="table7fn2">b</xref></sup>-based)</td><td align="left" valign="top">Hybrid (intrinsic+post hoc)</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Advanced imputation (ML-based)</td><td align="left" valign="top">Hybrid (intrinsic+post hoc)</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Simple imputation</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Intrinsic interpretability</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Time series imputation</td><td align="left" valign="top">Attention-based or ante hoc</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Post hoc feature attribution</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Limited or unclear</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Simple imputation+missingness indicators</td><td align="left" valign="top">Post hoc feature attribution</td></tr></tbody></table><table-wrap-foot><fn id="table7fn1"><p><sup>a</sup>DL: deep learning. </p></fn><fn id="table7fn2"><p><sup>b</sup>ML: machine learning. </p></fn></table-wrap-foot></table-wrap><p>Overall, model development methods show that EHR-based delirium prediction is technically feasible, but the clinical value of additional data complexity remains uncertain without stronger validation, calibration, and implementation testing.</p></sec><sec id="s3-4"><title>Model Performance</title><p>Model discrimination is summarized in <xref ref-type="table" rid="table8">Table 8</xref>, with additional model performance metrics, including sensitivity, specificity, PPV, negative predictive value (NPV), and threshold definitions provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref42">42</xref>]. Reported discrimination suggested promising technical performance but should be interpreted cautiously. Of 29 studies, internal AUROC was reported in 24 (83%) studies, with values ranging from 0.77 [<xref ref-type="bibr" rid="ref24">24</xref>] to 0.97 [<xref ref-type="bibr" rid="ref23">23</xref>]. Confidence intervals were inconsistently reported, and the internal estimates came from heterogeneous designs, including split-sample validation, cross-validation, temporal evaluation, and prospective cohorts.</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Model discrimination (internal and external AUROC<sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup>).</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Internal AUROC</td><td align="left" valign="bottom">External AUROC</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">Not reported</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">Not applicable</td><td align="left" valign="top">Kennedy: 0.777; Zucchelli: 0.701; MDP<sup><xref ref-type="table-fn" rid="table8fn2">b</xref></sup>: 0.898; REDEEM<sup><xref ref-type="table-fn" rid="table8fn3">c</xref></sup>: 0.921</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">XGBoost:<sup><xref ref-type="table-fn" rid="table8fn4">d</xref></sup> 0.851; Neural network: 0.841</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">0.75 (95% CI 0.71&#x2010;0.79)</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">0.917</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">0.848 (95% CI 0.818&#x2010;0.878)</td><td align="left" valign="top">0.824 (95% CI 0.818&#x2010;0.830)</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">0.87 (95% CI 0.86&#x2010;0.87; CatBoost on evaluation set)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">0.909 (95% CI 0.898&#x2010;0.921)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">&#x2212;0.85 to 0.86 (best models: Random Forest and GAM<sup><xref ref-type="table-fn" rid="table8fn5">e</xref></sup>)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">ICU<sup><xref ref-type="table-fn" rid="table8fn6">f</xref></sup> CatBoost: 0.974; Ward CatBoost: 0.910</td><td align="left" valign="top">Not reported</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">0.77&#x2010;0.79</td><td align="left" valign="top">0.64&#x2010;0.75</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">0.919 (XGBoost, best-performing model)</td><td align="left" valign="top">0.721 (Random Forest, best-performing model)</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">0.855 (prospective evaluation)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">0.924&#x2010;0.931 (retrained models on external cohort)</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">0.883 (95% CI 0.852&#x2010;0.915)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">0.80 (95% CI 0.77&#x2010;0.84)</td><td align="left" valign="top">0.82 (95% CI 0.80&#x2010;0.83)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">0.92 (full feature set); 0.86 (selected feature set)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">0.952 (combined model, 6-hour prediction window)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">0.909 (0&#x2010;12 hours); 0.895 (12&#x2010;24 hours)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">0.85 (cross-validation, derivation cohort)</td><td align="left" valign="top">0.86&#x2010;0.90 (XGBoost); 0.86&#x2010;0.89 (LASSO<sup><xref ref-type="table-fn" rid="table8fn7">g</xref></sup>); 0.84&#x2010;0.88 (logistic regression)</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">0.89 (training), 0.90 (test set)</td><td align="left" valign="top">0.72</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">0.839 (GBM<sup><xref ref-type="table-fn" rid="table8fn8">h</xref></sup>, best-performing model)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">0.80 (modified MDP model)</td><td align="left" valign="top">Not performed</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">0.77 (95% CI 0.72&#x2010;0.82)</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">0.81 (retrospective C-statistic; 95% CI 0.80&#x2010;0.82)</td><td align="left" valign="top">0.69 (prospective C-statistic; 95% CI 0.61&#x2010;0.77)</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">0.69&#x2010;0.81 (scenario-dependent)</td><td align="left" valign="top">Not applicable</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">0.82 (admission-time model)</td><td align="left" valign="top">Up to 0.95 (discharge model; site-averaged)</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">0.81&#x2010;0.85 (hospital-dependent; admission or discharge models)</td><td align="left" valign="top">Approximately 8 percentage points decrease when applied cross-hospital</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">0.855 (GBM, test set)</td><td align="left" valign="top">Not performed</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup>AUROC: area under the receiver operating characteristic curve.</p></fn><fn id="table8fn2"><p><sup>b</sup>MDP: Mayo Delirium Prediction.</p></fn><fn id="table8fn3"><p><sup>c</sup>REDEEM: Risk Estimate of Delirium in Elderly Emergency Medicine.</p></fn><fn id="table8fn4"><p><sup>d</sup>XGBoost: Extreme Gradient Boosting.</p></fn><fn id="table8fn5"><p><sup>e</sup>GAM: Generalized Additive Model.</p></fn><fn id="table8fn6"><p><sup>f</sup>ICU: intensive care unit.</p></fn><fn id="table8fn7"><p><sup>g</sup>LASSO: Least Absolute Shrinkage and Selection Operator.</p></fn><fn id="table8fn8"><p><sup>h</sup>GBM: Gradient Boosting Machine.</p></fn></table-wrap-foot></table-wrap><p>Of 29 studies, external AUROC was reported in 12 (41%) studies [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. Values ranged from 0.69 in prospective validation [<xref ref-type="bibr" rid="ref38">38</xref>] to approximately 0.95 in large multisite evaluations [<xref ref-type="bibr" rid="ref40">40</xref>]. However, external validation datasets differed substantially in clinical setting, prevalence, and outcome definition, limiting direct ranking of models.</p><p>Precision-recall performance and threshold reporting are summarized in <xref ref-type="table" rid="table9">Table 9</xref>. Precision-recall performance was much less frequently reported than AUROC. Of 29 studies, internal precision&#x2013;recall area under the curve (PR-AUC) was reported in 6 (21%) studies [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], and external PR-AUC in 2 (7%) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref33">33</xref>] studies. This is an important gap because many delirium prediction settings are low-prevalence tasks where AUROC can appear favorable despite limited PPV.</p><table-wrap id="t9" position="float"><label>Table 9.</label><caption><p>Precision-recall performance and threshold definition.</p></caption><table id="table9" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Internal PR-AUC<sup><xref ref-type="table-fn" rid="table9fn1">a</xref></sup></td><td align="left" valign="bottom">External PR-AUC</td><td align="left" valign="bottom">Threshold defined</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">NR<sup><xref ref-type="table-fn" rid="table9fn2">b</xref></sup></td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (&#x2265;14.1% risk cutoff)</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (predefined cutoffs per tool)</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (model-specific cutoffs)</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (Youden-optimized cut points; eg, 0.12 and 0.15)</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Not reported</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">0.192</td><td align="left" valign="top">0.118</td><td align="left" valign="top">Yes (example threshold 0.20; performance varies by hospital; additional thresholds in supplement)</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">0.62 (95% CI 0.59&#x2010;0.64)</td><td align="left" valign="top">Not performed</td><td align="left" valign="top">Yes (Youden index)</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">0.604</td><td align="left" valign="top">Not performed</td><td align="left" valign="top">Yes (operating points chosen to maximize F_ and MCC<sup><xref ref-type="table-fn" rid="table9fn3">c</xref></sup>)</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (Youden&#x2019;s J statistic)</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (threshold lowered from 0.50 to 0.40)</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (fixed threshold=0.50)</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (percentile-based thresholds: top 5% very high risk; next 10% high risk)</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (top 15% risk classified as high or very high)</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (85th or 95th percentile risk cutoffs)</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (Youden index; threshold=0.085)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">0.80 (full); 0.73 (selected)</td><td align="left" valign="top">Not performed</td><td align="left" valign="top">Not reported</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (fixed-recall analysis; example recall=0.80)</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">0.786 (0&#x2010;12 hours); 0.745 (12&#x2010;24 hours)</td><td align="left" valign="top">Not performed</td><td align="left" valign="top">Yes (threshold=0.37)</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Reported (AUPRC<sup><xref ref-type="table-fn" rid="table9fn4">d</xref></sup>; value not specified)</td><td align="left" valign="top">Reported (AUPRC; value not specified)</td><td align="left" valign="top">Not explicitly fixed</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (Youden-based cutoffs C1/ C2)</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (model-specific thresholds)</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (&#x2264;5%, 6%&#x2010;29%, &#x2265;30% risk strata)</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (predefined PIPRA<sup><xref ref-type="table-fn" rid="table9fn5">e</xref></sup> risk categories)</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (risk-strata cut points)</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">0.28&#x2010;0.45</td><td align="left" valign="top">Not applicable</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (score-based alert categories)</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (hospital-specific alert thresholds)</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">NR</td><td align="left" valign="top">NR</td><td align="left" valign="top">Yes (thresholds set at 90% sensitivity and 90% specificity)</td></tr></tbody></table><table-wrap-foot><fn id="table9fn1"><p><sup>a</sup>PR-AUC: precision&#x2013;recall area under the curve.</p></fn><fn id="table9fn2"><p><sup>b</sup>NR: not reported.</p></fn><fn id="table9fn3"><p><sup>c</sup>MCC: Matthews Correlation Coefficient.</p></fn><fn id="table9fn4"><p><sup>d</sup>AUPRC: Area under the precision-recall curve.</p></fn><fn id="table9fn5"><p><sup>e</sup>PIPRA: Pre-Interventional Preventive Risk Assessment.</p></fn></table-wrap-foot></table-wrap><p>Threshold-based metrics were difficult to compare. Although sensitivity, specificity, PPV, and NPV were often reported, thresholds were selected using different strategies, including data-driven optimization, fixed probability thresholds, and percentile-based risk strata. Few studies justified thresholds in relation to clinical resources, alert burden, or the intended intervention.</p><p>The performance synthesis therefore supports a cautious conclusion: routinely collected EHR data can discriminate delirium risk, but discrimination alone does not establish clinical usefulness. For deployment, external calibration, threshold consequences, and decision-analytic benefit are as important as AUROC.</p></sec><sec id="s3-5"><title>Validation, Calibration, and Implementation Characteristics</title><p>Validation strategies are summarized in <xref ref-type="table" rid="table10">Table 10</xref> and show a clear gap between model development and transportability testing. Of 29 studies, 9 (31%) reported internal validation only, 8 (28%) combined internal and external validation, 2 (7%) focused on external validation only, and 2 (7%) reported temporal validation only. Of 29 studies, 3 (10%) were prospective evaluations without a distinct internal or external validation phase; 2 (7%) combined external validation with prospective evaluation; 2 (7%) combined internal, external, and prospective evaluation; and 1 (3%) reported apparent performance only.</p><table-wrap id="t10" position="float"><label>Table 10.</label><caption><p>Validation strategies.</p></caption><table id="table10" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Validation scope</td><td align="left" valign="bottom">Internal validation approach</td><td align="left" valign="bottom">External validation (type)</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">External only</td><td align="left" valign="top">None or not applicable</td><td align="left" valign="top">Geographic (different hospital)</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">External only</td><td align="left" valign="top">None or not applicable</td><td align="left" valign="top">Geographic (independent ED<sup><xref ref-type="table-fn" rid="table10fn1">a</xref></sup> cohort)</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">Geographic (multiple hospitals)</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Apparent only</td><td align="left" valign="top">Apparent performance only</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Bootstrap</td><td align="left" valign="top">Geographic</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Cross-validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">Geographic (cross-setting)</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">Geographic (multihospital)</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Temporal internal validation</td><td align="left" valign="top">Geographic</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Prospective only</td><td align="left" valign="top">None or not applicable</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Internal+external+prospective</td><td align="left" valign="top">Cross-validation</td><td align="left" valign="top">Geographic+temporal</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Prospective only</td><td align="left" valign="top">None or not applicable</td><td align="left" valign="top">Prospective clinical cohort</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">Geographic (independent hospital)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Liu et al(2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Combined internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Temporal only</td><td align="left" valign="top">Cross-validation</td><td align="left" valign="top">Temporal</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">Geographic</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Cross-validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Prospective only</td><td align="left" valign="top">Prospective internal validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">External+prospective</td><td align="left" valign="top">None or not applicable</td><td align="left" valign="top">Prospective clinical cohort</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">External+prospective</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">Prospective clinical cohort</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Internal only</td><td align="left" valign="top">Cross-validation</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Internal+external</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">Geographic (multisite)</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Internal+external+prospective</td><td align="left" valign="top">Split-sample</td><td align="left" valign="top">Geographic (live or workflow)</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Temporal only</td><td align="left" valign="top">Temporal internal validation</td><td align="left" valign="top">None</td></tr></tbody></table><table-wrap-foot><fn id="table10fn1"><p><sup>a</sup>ED: emergency department.</p></fn></table-wrap-foot></table-wrap><p>Overall, 16 studies reported some form of external or prospective validation [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. Most external validation tested transfer across hospitals, health systems, time periods, or critical care databases. Cross-setting validation, such as applying an ICU-derived model to ward patients or vice versa, was uncommon.</p><p>Among the 7 studies with directly comparable internal and external AUROC values, 5 (71%) showed lower discrimination after external validation (<xref ref-type="fig" rid="figure2">Figure 2</xref>). The magnitude of decline varied, indicating that performance transportability was influenced by differences in population, data capture, outcome ascertainment, and workflow context.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Comparison of internal and external AUROC values among studies reporting directly comparable discrimination results. AUROC: area under the receiver operating characteristic curve [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>].</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91618_fig02.png"/></fig><p>In this subset, mean internal AUROC was 0.845 and mean external AUROC was 0.772, corresponding to a mean change of &#x2212;0.073. This descriptive comparison should not be interpreted as a pooled effect estimate, but it illustrates the risk of relying on internal performance when judging deployment readiness.</p><p>The external validation evidence was therefore mixed: several models remained discriminative outside their development data, but validation was often conducted in settings similar to the development environment. Evidence for robust transport across substantially different institutions, care pathways, and outcome assessment practices remains limited.</p><p>Calibration methods and decision curve analysis are summarized in <xref ref-type="table" rid="table11">Table 11</xref>. Of 29 studies, calibration was assessed in 15 (52%) studies [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. Methods included calibration plots, Brier scores, Hosmer-Lemeshow tests, calibration slope or calibration-in-the-large, Platt scaling, isotonic regression, and expected calibration error. The diversity of methods, combined with incomplete reporting, made calibration difficult to compare across studies.</p><table-wrap id="t11" position="float"><label>Table 11.</label><caption><p>Calibration methods and decision curve analysis.</p></caption><table id="table11" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Calibration method</td><td align="left" valign="bottom">Decision curve analysis</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">Calibration plots, Brier score, Platt scaling, and Spiegelhalter <italic>z</italic> test</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Calibration plots</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Hosmer-Lemeshow test; calibration plots</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">Not applicable</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Platt scaling; calibration plots</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">Calibration curves</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">Brier score</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Calibration plots (risk strata with confidence intervals)</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Calibration plots</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Calibration plots; Brier score (scaled)</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">Not applicable</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">Expected calibration error</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">Isotonic regression; Brier score</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">Calibration slope, calibration intercept, and Brier score</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Calibration plots; Brier score</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Calibration-in-the-large, calibration slope, and calibration plots</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">Not reported</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Isotonic regression; calibration plots</td><td align="left" valign="top">Yes</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Platt scaling; calibration plots</td><td align="left" valign="top">No</td></tr></tbody></table></table-wrap><p>Calibration reporting was often less mature than discrimination reporting. Several studies relied mainly on visual assessment, and recalibration after external validation was rare. Of 29 studies, decision curve analysis was reported in only 3 (10%) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref41">41</xref>], leaving limited evidence about whether model-guided decisions would improve net clinical benefit.</p><p>Prospective evaluation and implementation characteristics are summarized in <xref ref-type="table" rid="table12">Table 12</xref>. Of 29 studies, 8 (28%) included prospective evaluation [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>], and 9 (31%) reported some form of workflow integration or implementation testing [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref36">36</xref>-<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. These ranged from silent prospective validation to live EHR alerts, but few assessed downstream effects on clinician behavior, prevention delivery, alert burden, or patient outcomes.</p><table-wrap id="t12" position="float"><label>Table 12.</label><caption><p>Prospective evaluation and implementation.</p></caption><table id="table12" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study ID</td><td align="left" valign="bottom">Prospective evaluation</td><td align="left" valign="bottom">Implementation tested</td></tr></thead><tbody><tr><td align="left" valign="top">Ali et al (2023) [<xref ref-type="bibr" rid="ref14">14</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No (compared against standard VMS<sup><xref ref-type="table-fn" rid="table12fn1">a</xref></sup> questions [nonintegrated comparison])</td></tr><tr><td align="left" valign="top">Bartolacci et al (2025) [<xref ref-type="bibr" rid="ref15">15</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Bishara et al (2022) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Castro et al (2021) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Ceppi et al (2023) [<xref ref-type="bibr" rid="ref18">18</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Contreras et al (2025) [<xref ref-type="bibr" rid="ref19">19</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Contreras et al (2023) [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Corradi et al (2018) [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Davoudi et al (2017) [<xref ref-type="bibr" rid="ref22">22</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Heikal et al (2024) [<xref ref-type="bibr" rid="ref23">23</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Holler et al (2025) [<xref ref-type="bibr" rid="ref24">24</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Hur et al (2021) [<xref ref-type="bibr" rid="ref25">25</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Jauk et al (2020) [<xref ref-type="bibr" rid="ref26">26</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (fully embedded in hospital information system)</td></tr><tr><td align="left" valign="top">Jauk et al (2022) [<xref ref-type="bibr" rid="ref27">27</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (integrated into HIS<sup><xref ref-type="table-fn" rid="table12fn2">b</xref></sup>)</td></tr><tr><td align="left" valign="top">Jauk et al (2024) [<xref ref-type="bibr" rid="ref28">28</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (real-time predictions integrated into HIS; blinded to staff during study)</td></tr><tr><td align="left" valign="top">Jung et al (2022) [<xref ref-type="bibr" rid="ref29">29</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No (web-based tool only; no real-world impact evaluation)</td></tr><tr><td align="left" valign="top">Li et al (2024) [<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No (future integration proposed only)</td></tr><tr><td align="left" valign="top">Liu et al (2022) [<xref ref-type="bibr" rid="ref31">31</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Lucini et al (2023) [<xref ref-type="bibr" rid="ref32">32</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Matsumoto et al (2023) [<xref ref-type="bibr" rid="ref33">33</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Moon et al (2018) [<xref ref-type="bibr" rid="ref34">34</xref>]</td><td align="left" valign="top">Yes (after implementation)</td><td align="left" valign="top">Yes (live EHR<sup><xref ref-type="table-fn" rid="table12fn3">c</xref></sup> Kardex alert)</td></tr><tr><td align="left" valign="top">Mueller et al (2023) [<xref ref-type="bibr" rid="ref35">35</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Pagali et al (2022) [<xref ref-type="bibr" rid="ref36">36</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Partial (EHR-integrated data capture; no automated alerts)</td></tr><tr><td align="left" valign="top">Reeve et al (2025) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (embedded in routine clinical workflow)</td></tr><tr><td align="left" valign="top">Rudolph et al (2016) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (EMR<sup><xref ref-type="table-fn" rid="table12fn4">d</xref></sup>-integrated; real-time execution approximately 8 seconds)</td></tr><tr><td align="left" valign="top">Sheikhalishahi et al (2023) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr><tr><td align="left" valign="top">Sun et al (2021) [<xref ref-type="bibr" rid="ref40">40</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">Yes (live EHR integration in 2 hospitals)</td></tr><tr><td align="left" valign="top">Sun et al (2022) [<xref ref-type="bibr" rid="ref41">41</xref>]</td><td align="left" valign="top">Yes</td><td align="left" valign="top">Yes (production EHR integration)</td></tr><tr><td align="left" valign="top">Wong et al (2018) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">No</td><td align="left" valign="top">No</td></tr></tbody></table><table-wrap-foot><fn id="table12fn1"><p><sup>a</sup>VMS: Dutch safety management system (Veiligheidsmanagementsysteem).</p></fn><fn id="table12fn2"><p><sup>b</sup>HIS: hospital information system. </p></fn><fn id="table12fn3"><p><sup>c</sup>EHR: electronic health record. </p></fn><fn id="table12fn4"><p><sup>d</sup>EMR: electronic medical record.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Risk of Bias and Applicability</title><p>Risk of bias was assessed using PROBAST, with domain-level judgments summarized in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref42">42</xref>] and overall proportions illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>. Of 29 studies, overall risk of bias was judged low in 8 (28%), unclear in 10 (34%), and high in 11 (38%). The main pattern was not a lack of clinical relevance but limited methodological assurance.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>PROBAST (Prediction Model Risk of Bias Assessment Tool) stacked bar. ROB: risk of bias.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91618_fig03.png"/></fig><p>Low-risk studies generally had clearer participant selection, predictor timing, outcome ascertainment, and model evaluation [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. These studies provide the most reliable evidence that routinely collected data can support delirium prediction.</p><p>Studies with unclear risk of bias were usually limited by incomplete reporting rather than obvious methodological failure [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. Common sources of uncertainty included missing data handling, calibration assessment, feature selection, and validation procedures.</p><p>High risk of bias was identified in 11 studies, most often because of analysis-domain limitations [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. Recurrent issues included apparent-only performance reporting, case-control designs, inadequate handling or reporting of missing data, limited calibration assessment, and incomplete reporting of feature selection.</p><p>Within the 11 high-risk studies, inadequate handling or reporting of missing data was identified in 6 studies [<xref ref-type="bibr" rid="ref26">26</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref41">41</xref>], limited or absent calibration assessment in 4 studies [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref38">38</xref>], apparent-only performance without validation in 1 study [<xref ref-type="bibr" rid="ref18">18</xref>], case-control design in 2 studies [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], and incomplete feature selection reporting in 2 studies [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Several studies had more than 1 limitation.</p><p>These risk-of-bias findings help explain why high AUROC values should not be interpreted as readiness for practice. Models can appear accurate in development datasets while still being vulnerable to overfitting, miscalibration, missing data artifacts, or poor transportability.</p><p>Applicability concerns were more limited than risk-of-bias concerns. Most studies evaluated adult inpatient populations and used routinely collected EHR predictors, supporting broad relevance to hospital practice. However, applicability was still context-dependent for ICU-only, perioperative, emergency department, rehabilitation, COVID-19, and single health system models.</p><p>The overall quality assessment therefore supports a nuanced interpretation: the field is clinically relevant and technically active, but the evidence base is not yet consistently strong enough to support unqualified clinical deployment. No study was excluded on the basis of PROBAST assessment. Instead, risk-of-bias judgments were used to interpret how much confidence should be placed in the reported performance and implementation claims.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This systematic review included 29 studies developing, validating, or evaluating prediction models for in-hospital delirium using routinely collected EHR data. The principal finding is that EHR-based delirium prediction is feasible across several hospital settings, but the current evidence is fragmented across different clinical prediction tasks. The literature supports the existence of measurable risk signals in routine data; it does not yet establish that any model class is consistently ready for routine clinical deployment.</p><p>Four findings are especially important for readers. First, most models were developed retrospectively, and only a minority underwent prospective evaluation or workflow testing. Second, model performance was commonly summarized by AUROC, while calibration, precision-recall metrics, and decision-analytic evaluation were less consistently reported. Third, increased algorithmic complexity did not consistently translate into better or more transportable performance. Fourth, risk of bias was mainly driven by analysis-domain limitations rather than by lack of clinical relevance.</p><p>These findings indicate that the next stage of the field should be less focused on producing additional internally validated models and more focused on defining clinical use cases, testing transportability, calibrating models for local populations, and evaluating whether model-guided care changes decisions or outcomes.</p></sec><sec id="s4-2"><title>Interpretation in Context of Existing Literature</title><p>The heterogeneity observed in this review is consistent with previous systematic reviews of delirium prediction and broader clinical prediction modeling research [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. Models differed not only in algorithm type but also in clinical setting, prediction timing, outcome ascertainment, and validation design. These differences mean that a single pooled estimate of performance would be difficult to interpret and could obscure clinically meaningful distinctions between prediction tasks.</p><p>The predominance of machine learning approaches also mirrors wider trends in hospital and critical care prediction research [<xref ref-type="bibr" rid="ref43">43</xref>]. However, this review suggests that algorithmic sophistication is not the main bottleneck. In several studies, simpler statistical or tree-based approaches performed similarly to deep learning models, especially when evaluated under comparable conditions. Data quality, predictor timing, outcome definition, and validation context appeared at least as important as model family.</p><p>The risk-of-bias patterns also align with metaresearch showing that many published prediction models have limitations in the analysis domain, including missing data handling, overfitting, and incomplete calibration assessment [<xref ref-type="bibr" rid="ref44">44</xref>]. Evidence that high-risk models often perform less well in external validation [<xref ref-type="bibr" rid="ref45">45</xref>] is directly relevant here, because several delirium models showed lower discrimination when tested beyond their development data.</p></sec><sec id="s4-3"><title>Clinical Implications</title><p>Clinically, delirium prediction models are attractive because they could help target prevention, screening, and staffing resources to patients most likely to benefit. The reviewed studies show that routine EHR data contain useful risk information in ward, ICU, perioperative, and emergency care contexts. This supports continued development of delirium prediction as a component of clinical decision support.</p><p>The implementation evidence, however, remains incomplete. Few studies assessed whether predictions changed clinician behavior, reduced delirium incidence, improved patient outcomes, or avoided alert fatigue. Without these evaluations, models with favorable discrimination may still have limited value in practice, particularly if thresholds are poorly calibrated to local prevalence and available resources.</p></sec><sec id="s4-4"><title>Methodological Implications</title><sec id="s4-4-1"><title>Model Complexity and Performance</title><p>Across studies that directly compared modeling approaches, there was no consistent evidence that more complex models outperformed simpler alternatives. Tree-based machine learning, penalized regression, rule-based tools, and neural network approaches all achieved overlapping discrimination ranges. In several cases, models with favorable internal performance did not retain a clear advantage during external validation.</p><p>This finding does not imply that complex models are unnecessary, particularly for dynamic ICU prediction or high-dimensional time series data. Rather, it suggests that model choice should follow the clinical task, data structure, interpretability requirements, and implementation constraints. For many hospital use cases, a well-calibrated and externally validated simpler model may be more useful than a complex model with opaque behavior and limited transportability evidence. The relationship between model family and reported AUROC is shown descriptively for internal and external performance in <xref ref-type="fig" rid="figure4">Figures 4A and 4B</xref>, respectively.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Reported AUROC by final model family across included studies: (A) internal AUROC; (B) external AUROC. AUROC: area under the receiver operating characteristic curve; ML: machine learning [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref42">42</xref>].</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91618_fig04.png"/></fig></sec><sec id="s4-4-2"><title>Prediction Horizon and Task Comparability</title><p>Prediction horizon was one of the most important sources of heterogeneity. Admission-time models, perioperative models, rolling ICU models, and full-stay prediction models answer different clinical questions. They differ in how early an intervention can be triggered, how close predictors are to delirium onset, and how much uncertainty remains at the time of prediction.</p><p>Consequently, comparing AUROC values across prediction horizons can be misleading. A dynamic model predicting delirium in the next 12 hours may appear stronger partly because it uses proximal physiological information, whereas an admission-time model may be clinically valuable precisely because it operates before deterioration is obvious. Future studies should therefore define the intended prediction moment and intervention pathway before evaluating performance.</p></sec><sec id="s4-4-3"><title>Validation and Generalizability</title><p>External validation remains the key step separating promising models from generalizable tools. Although some studies tested models outside the development dataset, validation often occurred in similar clinical settings or related health systems. Cross-setting validation and temporal validation were less common, despite being highly relevant for EHR models whose predictors and labels can change with local documentation practices.</p><p>The observed decline from internal to external AUROC in most directly comparable studies reinforces the need for conservative interpretation. Models intended for deployment should be tested across institutions, time periods, and patient groups that reflect their proposed use, and they should be recalibrated when transported to new settings.</p></sec><sec id="s4-4-4"><title>Calibration and Reliability</title><p>Calibration is central to clinical reliability but was inconsistently assessed. Good discrimination indicates that a model can rank patients by risk; it does not show that predicted probabilities are accurate. For delirium prevention, inaccurate probabilities may lead to undertreatment of truly high-risk patients or excessive alerts for patients unlikely to develop delirium.</p><p>The limited use of recalibration and decision curve analysis is therefore a major evidence gap. Before deployment, models should report calibration-in-the-large, calibration slope, calibration plots, and clinically meaningful threshold analyses. Decision curve analysis or equivalent usefulness-based evaluation can help determine whether the model adds value beyond usual care or simpler screening rules.</p></sec><sec id="s4-4-5"><title>Class Imbalance and Performance Metrics</title><p>Outcome prevalence varied widely, and several cohorts had low delirium prevalence. In such settings, AUROC can overstate practical usefulness because it is insensitive to the number of false positives generated at a chosen threshold. PPV and PR-AUC are especially important when the intended intervention is resource-intensive or when repeated alerts could reduce clinician trust.</p><p>The limited reporting of PR-AUC and threshold rationale therefore weakens the clinical interpretability of many studies. Future work should present threshold-specific consequences, including the number of patients flagged, false positives, false negatives, and expected resource implications at clinically plausible operating points.</p></sec><sec id="s4-4-6"><title>Threshold Selection and Implementation</title><p>Threshold selection should be treated as a clinical design decision rather than a statistical afterthought. Data-driven thresholds such as Youden index may maximize a performance statistic in a development dataset, but they may not match local prevention capacity or acceptable alert burden. Fixed thresholds and percentile-based risk groups can be easier to implement, but they also require calibration to local prevalence and workflow.</p><p>For clinical deployment, threshold selection should be linked to the intended action: enhanced screening, multicomponent prevention, geriatric consultation, medication review, or ICU-specific intervention. The acceptable balance between sensitivity and specificity will differ across these use cases.</p></sec><sec id="s4-4-7"><title>Use of Unstructured Data and NLP</title><p>The limited use of unstructured data is notable because delirium symptoms are often documented in narrative nursing, medical, and allied health notes. NLP may therefore improve both outcome ascertainment and predictor representation. However, free-text models raise additional challenges, including annotation burden, governance, changing documentation practices, and transportability across institutions. Future NLP-enhanced models should distinguish clearly between using language data to define the outcome and using language data as predictors. These uses have different risks for information leakage, temporal validity, and clinical implementation.</p></sec></sec><sec id="s4-5"><title>Implementation Considerations</title><p>Several findings have direct implications for deployment. A model should not be implemented solely because it has a high AUROC. It should have an explicitly defined clinical role, evidence of calibration in the target population, threshold analyses tied to available resources, and prospective evaluation showing that predictions can be acted on without excessive alert burden.</p><p>Implementation studies should therefore evaluate not only model performance but also workflow fit, clinician response, alert fatigue, equity, prevention delivery, and patient outcomes. This is particularly important for delirium, where prediction is useful only if it leads to timely and feasible prevention or treatment.</p></sec><sec id="s4-6"><title>Limitations</title><p>Several limitations should be considered when interpreting the findings of this review. First, the quality of the included evidence base was variable, with a substantial proportion of studies judged to be at high risk of bias, primarily due to analytical limitations. Second, external validation and prospective evaluation were inconsistently performed, limiting confidence in the generalizability of reported performance.</p><p>This review also has inherent limitations. Only English-language studies were included, which may introduce language bias. The review was not prospectively registered before screening began, which may increase the risk of reporting bias. In addition, substantial heterogeneity in study design, outcome definitions, and reporting precluded formal meta-analysis. Finally, risk-of-bias assessment relied on the completeness of reporting in the original studies, which may have resulted in conservative or unclear judgments in some cases.</p></sec><sec id="s4-7"><title>Future Research Directions</title><p>Future research should move from model development toward clinically anchored validation and evaluation. The immediate priority is external and temporal validation across heterogeneous populations, with calibration assessed in ways that can support local recalibration rather than only discrimination ranking.</p><p>Implementation research should then test whether predictions change care in practice. Studies should specify the intended intervention pathway, alert threshold, resource assumptions, and monitoring plan before deployment, and should measure clinician response, alert burden, prevention delivery, equity, resource use, and patient outcomes.</p><p>Reporting should make threshold consequences easy to interpret. Future studies should present the number of patients flagged, false positives, false negatives, and expected workload at clinically plausible operating points so that health systems can judge whether a model is compatible with local capacity.</p><p>Future models may benefit from combining structured EHR variables with unstructured clinical notes, because cognitive and behavioral changes are often documented narratively. NLP-enhanced models should distinguish outcome ascertainment from predictor use, and maintain temporal separation between predictors and outcomes to avoid information leakage.</p><p>Expansion beyond delirium to broader acute mental status deterioration may be valuable only when outcomes are clearly defined, temporally valid, and clinically actionable. Across all future work, transparent reporting and alignment with TRIPOD, TRIPOD-AI, CHARMS, and PROBAST will be essential to move from technically promising models toward reliable decision-support tools.</p></sec><sec id="s4-8"><title>Conclusions</title><p>Routinely collected EHR data can support delirium prediction across hospital settings, but favorable discrimination alone does not establish clinical readiness. The evidence remains heterogeneous in outcome definition, prediction timing, prevalence, validation strategy, calibration reporting, and implementation maturity; risk of bias, particularly in the analysis domain, also remains common. More complex algorithms did not consistently improve performance. Future work should prioritize externally validated, well-calibrated, and clinically interpretable models evaluated prospectively within real workflows, with explicit attention to threshold consequences, alert burden, and patient benefit.</p></sec></sec></body><back><ack><p>This study was conducted as part of the authors&#x2019; academic research activities. The authors thank colleagues and peer reviewers who provided informal feedback during the development of the review protocol and data extraction framework. The authors used OpenAI ChatGPT to support language editing and formatting checks. All AI-assisted text and materials were reviewed, revised, and verified by the authors, who take full responsibility for the final content. No AI tool was used as an author.</p></ack><notes><sec><title>Funding</title><p>This research received no specific grant from any funding agency in the public, commercial, or not-for-profit sectors. Article-processing charges are covered under the UCL&#x2013;JMIR institutional agreement via Jisc.</p></sec><sec><title>Data Availability</title><p>This study is a systematic review and did not generate new primary data. All data analyzed in this review were derived from published studies and are available within the paper and its multimedia appendices. Extracted data and risk-of-bias assessments are available from the corresponding author upon reasonable request. <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> provides the full database search strategies. <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provides detailed model development and predictor characteristics across included studies. <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> provides model performance metrics, including discrimination and classification measures. <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> provides the PROBAST domain-level risk-of-bias and applicability assessments.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUROC</term><def><p>Area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb2">CAM</term><def><p>Confusion Assessment Method</p></def></def-item><def-item><term id="abb3">CHARMS</term><def><p>Checklist for Critical Appraisal and Data Extraction for Systematic Reviews of Prediction Modeling Studies</p></def></def-item><def-item><term id="abb4">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb5"><italic>ICD</italic></term><def><p><italic>International Classification of Diseases</italic></p></def></def-item><def-item><term id="abb6">ICU</term><def><p>intensive care unit</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb9">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb10">PR-AUC</term><def><p>precision&#x2013;recall area under the curve</p></def></def-item><def-item><term id="abb11">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb12">PROBAST</term><def><p>Prediction Model Risk of Bias Assessment Tool</p></def></def-item><def-item><term id="abb13">TRIPOD</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis</p></def></def-item><def-item><term id="abb14">TRIPOD-AI</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis&#x2013;Artificial Intelligence</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>LaHue</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Douglas</surname><given-names>VC</given-names> </name></person-group><article-title>Approach to altered mental status and inpatient delirium</article-title><source>Neurol Clin</source><year>2022</year><month>02</month><volume>40</volume><issue>1</issue><fpage>45</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1016/j.ncl.2021.08.004</pub-id><pub-id pub-id-type="medline">34798974</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zoremba</surname><given-names>N</given-names> </name><name name-style="western"><surname>Coburn</surname><given-names>M</given-names> </name></person-group><article-title>Acute confusional states in hospital</article-title><source>Dtsch Arztebl Int</source><year>2019</year><month>02</month><day>15</day><volume>116</volume><issue>7</issue><fpage>101</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.3238/arztebl.2019.0101</pub-id><pub-id pub-id-type="medline">30905333</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Waszynski</surname><given-names>C</given-names> </name></person-group><article-title>The confusion assessment method (CAM)</article-title><source>Semantic Scholar</source><year>2007</year><access-date>2026-01-13</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.semanticscholar.org/paper/The-Confusion-Assessment-Method-(CAM)-Waszynski/ad2f4dd6f43a70ddc4be594207b061cce10f6892">https://www.semanticscholar.org/paper/The-Confusion-Assessment-Method-(CAM)-Waszynski/ad2f4dd6f43a70ddc4be594207b061cce10f6892</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pagali</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Manning</surname><given-names>DM</given-names> </name></person-group><article-title>Predicting when a patient would be &#x201C;out of the furrow&#x201D;&#x2014;a perspective on delirium prediction</article-title><source>Mayo Clin Proc</source><year>2019</year><month>10</month><volume>94</volume><issue>10</issue><fpage>2145</fpage><lpage>2146</lpage><pub-id pub-id-type="doi">10.1016/j.mayocp.2019.08.001</pub-id><pub-id pub-id-type="medline">31585588</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lopes</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Pagali</surname><given-names>SR</given-names> </name><etal/></person-group><article-title>Ascertainment of delirium status using natural language processing from electronic health records</article-title><source>J Gerontol A Biol Sci Med Sci</source><year>2022</year><month>03</month><day>3</day><volume>77</volume><issue>3</issue><fpage>524</fpage><lpage>530</lpage><pub-id pub-id-type="doi">10.1093/gerona/glaa275</pub-id><pub-id pub-id-type="medline">35239951</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lindroth</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bratzke</surname><given-names>L</given-names> </name><name name-style="western"><surname>Purvis</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Systematic review of prediction models for delirium in the older adult inpatient</article-title><source>BMJ Open</source><year>2018</year><month>04</month><day>28</day><volume>8</volume><issue>4</issue><fpage>e019223</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2017-019223</pub-id><pub-id pub-id-type="medline">29705752</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ruppert</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Lipori</surname><given-names>J</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>S</given-names> </name><etal/></person-group><article-title>ICU delirium-prediction models: a systematic review</article-title><source>Crit Care Explor</source><year>2020</year><month>12</month><volume>2</volume><issue>12</issue><fpage>e0296</fpage><pub-id pub-id-type="doi">10.1097/CCE.0000000000000296</pub-id><pub-id pub-id-type="medline">33354672</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Pei</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Machine learning-based prediction models for delirium: a systematic review and meta-analysis</article-title><source>J Am Med Dir Assoc</source><year>2022</year><month>10</month><volume>23</volume><issue>10</issue><fpage>1655</fpage><lpage>1668</lpage><pub-id pub-id-type="doi">10.1016/j.jamda.2022.06.020</pub-id><pub-id pub-id-type="medline">35922015</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Reitsma</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name></person-group><article-title>Transparent reporting of a multivariable prediction model for individual prognosis or diagnosis (TRIPOD): the TRIPOD statement</article-title><source>BMJ</source><year>2015</year><month>01</month><day>7</day><volume>350</volume><fpage>g7594</fpage><pub-id pub-id-type="doi">10.1136/bmj.g7594</pub-id><pub-id pub-id-type="medline">25569120</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title><source>BMJ</source><year>2024</year><month>04</month><day>18</day><volume>385</volume><fpage>q902</fpage><pub-id pub-id-type="doi">10.1136/bmj.q902</pub-id><pub-id pub-id-type="medline">38636956</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>de Groot</surname><given-names>JAH</given-names> </name><name name-style="western"><surname>Bouwmeester</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Critical appraisal and data extraction for systematic reviews of prediction modelling studies: the CHARMS checklist</article-title><source>PLoS Med</source><year>2014</year><month>10</month><volume>11</volume><issue>10</issue><fpage>e1001744</fpage><pub-id pub-id-type="doi">10.1371/journal.pmed.1001744</pub-id><pub-id pub-id-type="medline">25314315</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wolff</surname><given-names>RF</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>Riley</surname><given-names>RD</given-names> </name><etal/></person-group><article-title>PROBAST: a tool to assess the risk of bias and applicability of prediction model studies</article-title><source>Ann Intern Med</source><year>2019</year><month>01</month><day>1</day><volume>170</volume><issue>1</issue><fpage>51</fpage><lpage>58</lpage><pub-id pub-id-type="doi">10.7326/M18-1376</pub-id><pub-id pub-id-type="medline">30596875</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ali</surname><given-names>MIM</given-names> </name><name name-style="western"><surname>Kalkman</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Wijers</surname><given-names>CHW</given-names> </name><name name-style="western"><surname>Fleuren</surname><given-names>HWHA</given-names> </name><name name-style="western"><surname>Kramers</surname><given-names>C</given-names> </name><name name-style="western"><surname>de Wit</surname><given-names>HAJM</given-names> </name></person-group><article-title>External validity of an automated delirium prediction model (DEMO) and comparison to the manual VMS-questions: a retrospective cohort study</article-title><source>Int J Clin Pharm</source><year>2023</year><month>10</month><volume>45</volume><issue>5</issue><fpage>1128</fpage><lpage>1135</lpage><pub-id pub-id-type="doi">10.1007/s11096-023-01641-6</pub-id><pub-id pub-id-type="medline">37713029</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bartolacci</surname><given-names>M</given-names> </name><name name-style="western"><surname>Carpenter</surname><given-names>KP</given-names> </name><name name-style="western"><surname>Jeffery</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Mullan</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Carpenter</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Bellolio</surname><given-names>F</given-names> </name></person-group><article-title>Validation of 4 risk stratification tools for delirium in the emergency department</article-title><source>JAMA Netw Open</source><year>2025</year><month>11</month><day>3</day><volume>8</volume><issue>11</issue><fpage>e2540920</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.40920</pub-id><pub-id pub-id-type="medline">41182767</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bishara</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chiu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Whitlock</surname><given-names>EL</given-names> </name><etal/></person-group><article-title>Postoperative delirium prediction using machine learning models and preoperative electronic health record data</article-title><source>BMC Anesthesiol</source><year>2022</year><month>01</month><day>3</day><volume>22</volume><issue>1</issue><fpage>8</fpage><pub-id pub-id-type="doi">10.1186/s12871-021-01543-y</pub-id><pub-id pub-id-type="medline">34979919</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Sacks</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Perlis</surname><given-names>RH</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>TH</given-names> </name></person-group><article-title>Development and external validation of a delirium prediction model for hospitalized patients with coronavirus disease 2019</article-title><source>J Acad Consult Liaison Psychiatry</source><year>2021</year><volume>62</volume><issue>3</issue><fpage>298</fpage><lpage>308</lpage><pub-id pub-id-type="doi">10.1016/j.jaclp.2020.12.005</pub-id><pub-id pub-id-type="medline">33688635</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ceppi</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Rauch</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Sp&#x00F6;ndlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Meier</surname><given-names>CR</given-names> </name><name name-style="western"><surname>S&#x00E1;ndor</surname><given-names>PS</given-names> </name></person-group><article-title>Assessing the risk of developing delirium on admission to inpatient rehabilitation: a clinical prediction model</article-title><source>J Am Med Dir Assoc</source><year>2023</year><month>12</month><volume>24</volume><issue>12</issue><fpage>1931</fpage><lpage>1935</lpage><pub-id pub-id-type="doi">10.1016/j.jamda.2023.07.003</pub-id><pub-id pub-id-type="medline">37573886</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Contreras</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A large language model for delirium prediction in the intensive care unit using structured electronic health records</article-title><source>Sci Rep</source><year>2025</year><month>11</month><day>6</day><volume>15</volume><issue>1</issue><fpage>38890</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-22634-7</pub-id><pub-id pub-id-type="medline">41198740</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Contreras</surname><given-names>M</given-names> </name><name name-style="western"><surname>Silva</surname><given-names>B</given-names> </name><name name-style="western"><surname>Shickel</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Dynamic delirium prediction in the intensive care unit using machine learning on electronic health records</article-title><source>IEEE EMBS Int Conf Biomed Health Inform</source><year>2023</year><month>10</month><volume>2023</volume><pub-id pub-id-type="doi">10.1109/bhi58575.2023.10313445</pub-id><pub-id pub-id-type="medline">38585187</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Corradi</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mather</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Waszynski</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Dicks</surname><given-names>RS</given-names> </name></person-group><article-title>Prediction of incident delirium using a Random Forest classifier</article-title><source>J Med Syst</source><year>2018</year><month>11</month><day>14</day><volume>42</volume><issue>12</issue><fpage>261</fpage><pub-id pub-id-type="doi">10.1007/s10916-018-1109-0</pub-id><pub-id pub-id-type="medline">30430256</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Davoudi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ozrazgat-Baslanti</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ebadi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bursian</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Bihorac</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rashidi</surname><given-names>P</given-names> </name></person-group><article-title>Delirium prediction using machine learning models on predictive electronic health records data</article-title><year>2017</year><conf-name>2017 IEEE 17th International Conference on Bioinformatics and Bioengineering (BIBE)</conf-name><conf-date>Oct 23-25, 2017</conf-date><conf-loc>Washington, DC, USA</conf-loc><fpage>568</fpage><lpage>573</lpage><pub-id pub-id-type="doi">10.1109/BIBE.2017.00014</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heikal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Saad</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ghanime</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>Using machine learning and electronic health records to identify neuropsychiatric risk scores for delirium in ICU and general hospital settings</article-title><source>Neuropsychiatr Dis Treat</source><year>2024</year><volume>20</volume><fpage>1861</fpage><lpage>1876</lpage><pub-id pub-id-type="doi">10.2147/NDT.S479756</pub-id><pub-id pub-id-type="medline">39372875</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holler</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ludema</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ben Miled</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Development and Validation of a routine electronic health record-based delirium prediction model for surgical patients without dementia: retrospective case-control study</article-title><source>JMIR Perioper Med</source><year>2025</year><month>01</month><day>9</day><volume>8</volume><fpage>e59422</fpage><pub-id pub-id-type="doi">10.2196/59422</pub-id><pub-id pub-id-type="medline">39786865</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hur</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ha</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cha</surname><given-names>WC</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>CR</given-names> </name></person-group><article-title>A machine learning-based algorithm for the Prediction of Intensive Care Unit Delirium (PRIDE): retrospective study</article-title><source>JMIR Med Inform</source><year>2021</year><month>07</month><day>26</day><volume>9</volume><issue>7</issue><fpage>e23401</fpage><pub-id pub-id-type="doi">10.2196/23401</pub-id><pub-id pub-id-type="medline">34309567</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jauk</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kramer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gro&#x00DF;auer</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Risk prediction of delirium in hospitalized patients using machine learning: an implementation and prospective evaluation study</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>07</month><day>1</day><volume>27</volume><issue>9</issue><fpage>1383</fpage><lpage>1392</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa113</pub-id><pub-id pub-id-type="medline">32968811</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jauk</surname><given-names>S</given-names> </name><name name-style="western"><surname>Veeranki</surname><given-names>SPK</given-names> </name><name name-style="western"><surname>Kramer</surname><given-names>D</given-names> </name><etal/></person-group><article-title>External validation of a machine learning based delirium prediction software in clinical routine</article-title><source>Stud Health Technol Inform</source><year>2022</year><month>05</month><day>16</day><volume>293</volume><fpage>93</fpage><lpage>100</lpage><pub-id pub-id-type="doi">10.3233/SHTI220353</pub-id><pub-id pub-id-type="medline">35592966</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jauk</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kramer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sumerauer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Veeranki</surname><given-names>SPK</given-names> </name><name name-style="western"><surname>Schrempf</surname><given-names>M</given-names> </name><name name-style="western"><surname>Puchwein</surname><given-names>P</given-names> </name></person-group><article-title>Machine learning-based delirium prediction in surgical in-patients: a prospective validation study</article-title><source>JAMIA Open</source><year>2024</year><month>10</month><volume>7</volume><issue>3</issue><fpage>ooae091</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooae091</pub-id><pub-id pub-id-type="medline">39297150</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jung</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A machine-learning model to predict postoperative delirium following knee arthroplasty using electronic health records</article-title><source>BMC Psychiatry</source><year>2022</year><month>06</month><day>27</day><volume>22</volume><issue>1</issue><fpage>436</fpage><pub-id pub-id-type="doi">10.1186/s12888-022-04067-y</pub-id><pub-id pub-id-type="medline">35761274</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A machine learning-based prediction model for postoperative delirium in cardiac valve surgery using electronic health records</article-title><source>BMC Cardiovasc Disord</source><year>2024</year><month>01</month><day>18</day><volume>24</volume><issue>1</issue><fpage>56</fpage><pub-id pub-id-type="doi">10.1186/s12872-024-03723-3</pub-id><pub-id pub-id-type="medline">38238677</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schlesinger</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>AB</given-names> </name><etal/></person-group><article-title>New onset delirium prediction using machine learning and long short-term memory (LSTM) in electronic health record</article-title><source>J Am Med Inform Assoc</source><year>2022</year><month>12</month><day>13</day><volume>30</volume><issue>1</issue><fpage>120</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocac210</pub-id><pub-id pub-id-type="medline">36303456</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lucini</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Stelfox</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name></person-group><article-title>Deep learning-based recurrent delirium prediction in critically ill patients</article-title><source>Crit CARE Med</source><year>2023</year><month>04</month><day>1</day><volume>51</volume><issue>4</issue><fpage>492</fpage><lpage>502</lpage><pub-id pub-id-type="doi">10.1097/CCM.0000000000005789</pub-id><pub-id pub-id-type="medline">36790184</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matsumoto</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nohara</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sakaguchi</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Temporal generalizability of machine learning models for predicting postoperative delirium using electronic health record data: model development and validation study</article-title><source>JMIR Perioper Med</source><year>2023</year><month>10</month><day>26</day><volume>6</volume><fpage>e50895</fpage><pub-id pub-id-type="doi">10.2196/50895</pub-id><pub-id pub-id-type="medline">37883164</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moon</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SM</given-names> </name></person-group><article-title>Development and validation of an automated delirium risk assessment system (Auto-DelRAS) implemented in the electronic health record system</article-title><source>Int J Nurs Stud</source><year>2018</year><month>01</month><volume>77</volume><fpage>46</fpage><lpage>53</lpage><pub-id pub-id-type="doi">10.1016/j.ijnurstu.2017.09.014</pub-id><pub-id pub-id-type="medline">29035732</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mueller</surname><given-names>B</given-names> </name><name name-style="western"><surname>Street</surname><given-names>WN</given-names> </name><name name-style="western"><surname>Carnahan</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating the performance of machine learning methods for risk estimation of delirium in patients hospitalized from the emergency department</article-title><source>Acta Psychiatr Scand</source><year>2023</year><month>05</month><volume>147</volume><issue>5</issue><fpage>493</fpage><lpage>505</lpage><pub-id pub-id-type="doi">10.1111/acps.13551</pub-id><pub-id pub-id-type="medline">36999191</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pagali</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Fischer</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Kashiwagi</surname><given-names>DT</given-names> </name><etal/></person-group><article-title>Validation and recalibration of modified Mayo delirium prediction tool in a hospitalized cohort</article-title><source>J Acad Consult Liaison Psychiatry</source><year>2022</year><volume>63</volume><issue>6</issue><fpage>521</fpage><lpage>528</lpage><pub-id pub-id-type="doi">10.1016/j.jaclp.2022.05.006</pub-id><pub-id pub-id-type="medline">35660677</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reeve</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Schmutz Gelsomino</surname><given-names>N</given-names> </name><name name-style="western"><surname>Venturini</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Prospective external validation of the automated PIPRA multivariable prediction model for postoperative delirium on real-world data from a consecutive cohort of non-cardiac surgery inpatients</article-title><source>BMJ Health Care Inform</source><year>2025</year><month>04</month><day>10</day><volume>32</volume><issue>1</issue><fpage>e101291</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2024-101291</pub-id><pub-id pub-id-type="medline">40216453</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rudolph</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Doherty</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>B</given-names> </name><name name-style="western"><surname>Driver</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Archambault</surname><given-names>E</given-names> </name></person-group><article-title>Validation of a delirium risk assessment using electronic medical record information</article-title><source>J Am Med Dir Assoc</source><year>2016</year><month>03</month><day>1</day><volume>17</volume><issue>3</issue><fpage>244</fpage><lpage>248</lpage><pub-id pub-id-type="doi">10.1016/j.jamda.2015.10.020</pub-id><pub-id pub-id-type="medline">26705000</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sheikhalishahi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhattacharyya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Celi</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Osmani</surname><given-names>V</given-names> </name></person-group><article-title>An interpretable deep learning model for time-series electronic health records: case study of delirium prediction in critical care</article-title><source>Artif Intell Med</source><year>2023</year><month>10</month><volume>144</volume><fpage>102659</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2023.102659</pub-id><pub-id pub-id-type="medline">37783541</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>H</given-names> </name><name name-style="western"><surname>Depraetere</surname><given-names>K</given-names> </name><name name-style="western"><surname>Meesseman</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A scalable approach for developing clinical risk prediction applications in different hospitals</article-title><source>J Biomed Inform</source><year>2021</year><month>06</month><volume>118</volume><fpage>103783</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2021.103783</pub-id><pub-id pub-id-type="medline">33887456</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>H</given-names> </name><name name-style="western"><surname>Depraetere</surname><given-names>K</given-names> </name><name name-style="western"><surname>Meesseman</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Machine learning-based prediction models for different clinical risks in different hospitals: evaluation of live performance</article-title><source>J Med Internet Res</source><year>2022</year><month>06</month><day>7</day><volume>24</volume><issue>6</issue><fpage>e34295</fpage><pub-id pub-id-type="doi">10.2196/34295</pub-id><pub-id pub-id-type="medline">35502887</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>A</given-names> </name><name name-style="western"><surname>Young</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Gonzales</surname><given-names>R</given-names> </name><name name-style="western"><surname>Douglas</surname><given-names>VC</given-names> </name><name name-style="western"><surname>Hadley</surname><given-names>D</given-names> </name></person-group><article-title>Development and validation of an electronic health record-based machine learning model to estimate delirium risk in newly hospitalized patients without known cognitive impairment</article-title><source>JAMA Netw Open</source><year>2018</year><month>08</month><day>3</day><volume>1</volume><issue>4</issue><fpage>e181018</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2018.1018</pub-id><pub-id pub-id-type="medline">30646095</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shillan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sterne</surname><given-names>JAC</given-names> </name><name name-style="western"><surname>Champneys</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gibbison</surname><given-names>B</given-names> </name></person-group><article-title>Use of machine learning to analyse routinely collected intensive care unit data: a systematic review</article-title><source>Crit Care</source><year>2019</year><month>08</month><day>22</day><volume>23</volume><issue>1</issue><fpage>284</fpage><pub-id pub-id-type="doi">10.1186/s13054-019-2564-9</pub-id><pub-id pub-id-type="medline">31439010</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Andaur Navarro</surname><given-names>CL</given-names> </name><name name-style="western"><surname>Damen</surname><given-names>JAA</given-names> </name><name name-style="western"><surname>Takada</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Risk of bias in studies on prediction models developed using supervised machine learning techniques: systematic review</article-title><source>BMJ</source><year>2021</year><month>10</month><day>20</day><volume>375</volume><fpage>n2281</fpage><pub-id pub-id-type="doi">10.1136/bmj.n2281</pub-id><pub-id pub-id-type="medline">34670780</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Venema</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wessler</surname><given-names>BS</given-names> </name><name name-style="western"><surname>Paulus</surname><given-names>JK</given-names> </name><etal/></person-group><article-title>Large-scale validation of the prediction model risk of bias assessment Tool (PROBAST) using a short form: high risk of bias models show poorer discrimination</article-title><source>J Clin Epidemiol</source><year>2021</year><month>10</month><volume>138</volume><fpage>32</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2021.06.017</pub-id><pub-id pub-id-type="medline">34175377</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Full database search strategies for all electronic databases.</p><media xlink:href="medinform_v14i1e91618_app1.xlsx" xlink:title="XLSX File, 8 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Detailed model development and predictor characteristics across included studies.</p><media xlink:href="medinform_v14i1e91618_app2.xlsx" xlink:title="XLSX File, 13 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Model performance metrics, including discrimination and classification measures.</p><media xlink:href="medinform_v14i1e91618_app3.xlsx" xlink:title="XLSX File, 12 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>PROBAST (Prediction Model Risk of Bias Assessment Tool) domain-level risk of bias and applicability assessments.</p><media xlink:href="medinform_v14i1e91618_app4.xlsx" xlink:title="XLSX File, 13 KB"/></supplementary-material><supplementary-material id="app5"><label>Checklist 1</label><p>PRISMA checklist.</p><media xlink:href="medinform_v14i1e91618_app5.pdf" xlink:title="PDF File, 168 KB"/></supplementary-material></app-group></back></article>