<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e84260</article-id><article-id pub-id-type="doi">10.2196/84260</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Prediction of Postoperative Vomiting Within 24 Hours Using Machine Learning With Large Language Model&#x2013;Enhanced Interpretability: Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Huan-Jun</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Lee</surname><given-names>Wei-Po</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gau</surname><given-names>Tz-Ping</given-names></name><degrees>MD, MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cheng</surname><given-names>Kuang-I</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Wei</surname><given-names>Cheng-Ru</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Information Management, National Sun Yat-sen University</institution><addr-line>No. 70, Lianhai Rd., Gushan District, Kaohsiung 80424</addr-line><addr-line>Kaohsiung</addr-line><country>Taiwan</country></aff><aff id="aff2"><institution>Department of Anesthesiology, Kaohsiung Medical University Hospital</institution><addr-line>Kaohsiung</addr-line><country>Taiwan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Salehnasab</surname><given-names>Cirruse</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yusuff</surname><given-names>Taofeek</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Peijnenburg</surname><given-names>Willie</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Cheng-Ru Wei, MSc, Department of Information Management, National Sun Yat-sen University, No. 70, Lianhai Rd., Gushan District, Kaohsiung 80424, Taiwan, Kaohsiung, Taiwan, 886 983337217; <email>chengruwei.acad@gmail.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e84260</elocation-id><history><date date-type="received"><day>18</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>19</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>07</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Huan-Jun Wang, Wei-Po Lee, Tz-Ping Gau, Kuang-I Cheng, Cheng-Ru Wei. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 31.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e84260"/><abstract><sec><title>Background</title><p>Postoperative nausea and vomiting are common complications after anesthesia. However, vomiting represents a clinically distinct and objectively measurable endpoint.</p></sec><sec><title>Objective</title><p>This study aimed to develop and internally validate predictive models for postoperative vomiting within 24 hours using structured perioperative data and unstructured clinical text, while introducing a structured framework that separates feature construction from interpretability using large language models (LLMs).</p></sec><sec sec-type="methods"><title>Methods</title><p>We analyzed 33,460 anesthesia records from a single center (2019&#x2010;2022). Two temporally defined prediction tasks were constructed to reflect real-world clinical decision-making and prevent information leakage: a preoperative model using variables available before anesthesia induction, and a perioperative model using variables available up to the end of surgery. Structured data were modeled using machine learning algorithms (logistic regression, Extreme Gradient Boosting, Light Gradient Boosting Machine [LightGBM]). Unstructured clinical text was incorporated through a deterministic, concept-driven preprocessing pipeline, where LLMs were used solely for normalization (temperature=0) without feature generation, followed by rule-based concept mapping and feature encoding. Post hoc interpretability was further supported using an LLM-based Question Answering Chain module. Model performance was evaluated using receiver operating characteristic-area under the curve (AUC), precision-recall AUC, calibration metrics, and threshold-based operating characteristics. Classification thresholds were selected using the Youden J statistic, and all metrics were reported with 95% CIs derived from bootstrap resampling. Decision curve analysis was performed to assess clinical utility.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 33,460 surgical procedures were included, of which 3607 (10.8%) experienced postoperative vomiting within 24 hours. In the preoperative task, LightGBM achieved an AUC of 0.729 (95% CI 0.706&#x2010;0.749), compared with 0.610 (95% CI 0.588&#x2010;0.632) for the Apfel score. In the end-of-surgery task, LightGBM achieved an AUC of 0.735 (95% CI 0.714&#x2010;0.757). At the Youden-optimal threshold, the negative predictive value exceeded 0.95 across all models. Decision curve analysis demonstrated positive net benefit across clinically relevant threshold probabilities. Incorporating text-derived features provided modest improvements, while LLM-based explanation modules generated structured, natural-language explanations intended to enhance interpretability without substantially improving predictive performance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Machine learning models can effectively predict postoperative vomiting within 24 hours using perioperative data. The proposed framework demonstrates that LLMs can be integrated in a controlled and reproducible manner&#x2014;restricted to deterministic normalization and post hoc reasoning&#x2014;to generate natural-language explanations intended to enhance the interpretability of model predictions, without introducing information leakage or altering predictive modeling. As no formal clinician-based evaluation was conducted, this interpretability benefit cannot yet be objectively confirmed, and the generated explanations should be regarded as a useful interpretability aid to be validated in future clinician-centered studies. External, multicenter validation is required before broader clinical applicability can be assumed.</p></sec></abstract><kwd-group><kwd>postoperative vomiting</kwd><kwd>machine learning</kwd><kwd>clinical prediction model</kwd><kwd>perioperative risk prediction</kwd><kwd>electronic health records</kwd><kwd>interpretability</kwd><kwd>SHAP</kwd><kwd>large language models</kwd><kwd>clinical decision support</kwd><kwd>temporal validation</kwd><kwd>Shapley Additive Explanations</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>In clinical medicine, anesthesia is a fundamental component across a wide range of procedures. However, it is associated with several postoperative complications, among which postoperative nausea and vomiting (PONV) remains one of the most common and clinically significant. PONV can disrupt recovery, increase the risk of secondary complications, prolong hospitalization, and negatively impact patients&#x2019; perceived quality of life [<xref ref-type="bibr" rid="ref1">1</xref>].</p><p>Accurate identification and stratification of PONV risk factors are essential for guiding preventive strategies. The overall incidence of PONV is estimated at 27.7%, underscoring its clinical importance [<xref ref-type="bibr" rid="ref2">2</xref>]. Traditional risk assessment approaches primarily rely on static patient characteristics, such as sex, age, smoking status, and prior history of PONV or motion sickness [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. However, these factors alone may be insufficient to capture the complexity of perioperative risk.</p><p>Although PONV includes both nausea and vomiting, these outcomes differ in mechanisms and clinical implications. Therefore, this study focuses specifically on postoperative vomiting within 24 hours (POV 24h) as a clearly defined and objectively measurable endpoint. Existing prediction tools, such as Apfel&#x2019;s simplified risk score, report moderate accuracy (55%&#x2010;80%) [<xref ref-type="bibr" rid="ref5">5</xref>], but are largely based on fixed patient characteristics and do not account for the temporal availability of perioperative information [<xref ref-type="bibr" rid="ref6">6</xref>]. As a result, their applicability in real-time clinical decision-making remains limited.</p><p>Despite these advances, current prediction models face several key methodological limitations. First, most models do not explicitly distinguish between variables available at different clinical time points, potentially introducing information leakage by incorporating variables not available at the intended prediction time point, thereby limiting real-world clinical deployment. Second, perioperative factors&#x2014;such as intraoperative physiological changes and anesthetic exposure&#x2014;are often underrepresented despite their clinical relevance. Third, unstructured clinical text, including medical history and procedural notes, is rarely incorporated into predictive modeling due to integration challenges.</p><p>To address these gaps, we propose a temporally structured prediction framework that separates preoperative and perioperative modeling tasks. In addition, we explore the use of large language models (LLMs) in a structured framework that separates feature construction from interpretability to transform unstructured clinical text into structured representations and to support post hoc clinical reasoning, with the primary goal of generating structured explanations and supporting transparent post hoc review, rather than substantially improving predictive performance.</p><p>To evaluate this framework, we conducted a large-scale study using 33,460 anesthesia records, incorporating both structured (eg, physiological parameters, surgical, and anesthetic information) and unstructured data (eg, medical history and procedural notes). The objective of this study was to develop and internally validate predictive models for POV 24h. Two models were constructed based on the availability of clinical data: a preoperative model using variables available before anesthesia induction, and a perioperative model using variables available up to the end of surgery. Furthermore, we investigated whether LLM-based processing of unstructured clinical text could be performed while maintaining methodological rigor and supporting clinically grounded reasoning.</p></sec><sec id="s1-2"><title>Related Works</title><p>To overcome the limitations of conventional models, recent studies have increasingly turned to machine learning (ML) and deep learning (DL) techniques for PONV prediction [<xref ref-type="bibr" rid="ref7">7</xref>]. These methods are well-suited for analyzing high-dimensional clinical datasets and capturing complex, nonlinear relationships that often elude traditional statistical models [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Although most prior studies focus on the composite outcome of PONV, vomiting can be assessed as a distinct and clinically measurable postoperative outcome within a defined time frame [<xref ref-type="bibr" rid="ref10">10</xref>]. Therefore, this study specifically focuses on POV 24h.</p><p>Among these approaches, gradient boosting models like LightGBM (Light Gradient Boosting Machine) have been widely used to identify key PONV predictors and assist anesthesiologists in postoperative risk evaluation [<xref ref-type="bibr" rid="ref11">11</xref>]. Common predictors highlighted across ML studies include sex, smoking status, type of surgical procedure, and use of opioids [<xref ref-type="bibr" rid="ref12">12</xref>]. DL further advances prediction by modeling intricate interactions among features, with comparative analyses showing that algorithms such as logistic regression, support vector classifiers (SVCs), and AdaBoost frequently outperform traditional scoring systems [<xref ref-type="bibr" rid="ref13">13</xref>]. Notably, TabNet, a DL architecture tailored for tabular data, leverages instance-level feature selection and attention mechanisms to improve efficiency and model transparency [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>Despite these advances, the opaque nature of many ML and DL models presents a barrier to clinical adoption. To address this, explainable AI (XAI) techniques such as Shapley Additive Explanations (SHAP) have been introduced to illuminate the decision-making process of complex models. SHAP enables clinicians to assess the influence of individual features on prediction outcomes [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. For instance, Jiang et al [<xref ref-type="bibr" rid="ref18">18</xref>] demonstrated that incorporating traditional Chinese medicine (TCM) metrics alongside glycosylated hemoglobin levels substantially enhanced the interpretability of ML models in diabetic neuropathy. Similarly, in PONV research, SHAP has revealed previously unrecognized high-dimensional risk factors linked to delayed but clinically significant postoperative nausea and vomiting (CIPONV) [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Preoperative assessment and optimization have long been recognized as critical components in surgical care [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. More recently, LLMs such as ChatGPT have been increasingly explored as tools to enhance the interpretability and usage of unstructured clinical text. LLMs have shown promise in preoperative evaluations and have even demonstrated alignment with expert consensus on pharmacologic strategies for PONV management [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. However, current LLM implementations tend to function independently of structured patient data and often overlook patient-specific factors such as age, sex, and clinical history [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>Past PONV research has laid a strong foundation for risk factor identification, offering essential insights for future feature engineering and data preprocessing efforts [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. Nonetheless, in today&#x2019;s fast-paced clinical environments, real-time decision-making is increasingly entrusted to frontline providers. Moreover, ML and DL systems should serve as enhancements to, and not replacements for, human clinical decision-making, especially in real-time environments where clinicians must rapidly interpret patient data [<xref ref-type="bibr" rid="ref29">29</xref>].</p><p>Recent studies emphasize that beyond predictive performance, model interpretability and calibration are critical for clinical deployment. For instance, explainable tree-based models such as ExtraTrees have shown that even models with moderate discrimination require careful evaluation of calibration and feature contributions before clinical adoption, highlighting the importance of aligning model outputs with real-world clinical distributions and decision thresholds [<xref ref-type="bibr" rid="ref30">30</xref>]. In addition, systematic reviews of clinical AI applications have identified persistent challenges in interpretability, bias, and insufficient validation, particularly in surgical and perioperative decision-making [<xref ref-type="bibr" rid="ref31">31</xref>]. These concerns further underscore the need for greater transparency and trust in AI systems, as emphasized in broader discussions on AI deployment [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>Taken together, future research should not only focus on improving predictive accuracy but also prioritize interpretable modeling, robust validation (including calibration), and transparency-aware model design, thereby fostering greater clinician confidence and facilitating real-world implementation.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We propose a clinically grounded framework for predicting POV 24h using ML, deep learning (DL), and LLMs. The framework comprises three core modules: data preprocessing, model development, and interpretability analysis (<xref ref-type="fig" rid="figure1">Figure 1</xref>). Traditional ML and DL models serve as predictive baselines, while LLM-based techniques support the structured transformation of unstructured clinical data and support post hoc interpretation within the modeling framework.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Proposed framework for predicting postoperative vomiting within 24 hours (POV 24h). The framework integrates modules for data preprocessing, model construction, and interpretability analysis. Traditional machine learning (ML) and deep learning (DL) methods serve as predictive baselines, while large language model (LLM)-based techniques support the structured representation of unstructured clinical data and generate post hoc explanations of model predictions. DL: deep learning; LLM: large-language model; ML: machine learning; POV: postoperative vomiting; QA: question answer.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig01.png"/></fig><p>The primary endpoint was POV 24h, excluding nausea-only events. For each surgical case, multiple postoperative interviews were aggregated at the surgery level. A case was labeled positive (1) if any documented vomiting event occurred within 24 hours, and negative (0) if all available interviews indicated no vomiting. Surgeries without a valid 24-hour follow-up were excluded from the main analysis.</p><p>To reflect real-world clinical decision-making, we defined two prediction tasks based on the timing of feature availability: (1) a preoperative model using variables available before anesthesia induction, and (2) a perioperative model using variables available up to the end of surgery. Only variables available prior to each prediction time point were included in model training. Postoperative variables were excluded from all prediction models. To minimize the risk of temporal leakage, all variables were classified according to temporal availability prior to model construction (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Variables unavailable at the intended prediction time point were excluded from the corresponding prediction task.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This retrospective study was approved by the Institutional Review Board of Kaohsiung Medical University Hospital (approval number: KMUHIRB-E(I)-20200040). The requirement for informed consent was waived due to the deidentified and retrospective nature of the data. The study was conducted in accordance with the principles of the Declaration of Helsinki.</p><p>The dataset was obtained from the hospital's electronic medical records system. Data collection and processing were conducted following institutional review board (IRB) approval (IRB No: KMUHIRB-E(I)-20200040).</p></sec><sec id="s2-3"><title>Structured Data Preprocessing</title><p>The dataset used in this study was obtained from a collaborating medical center and comprises both structured and unstructured data. Structured data includes physiological measurements, anesthetic information, and surgical procedure types. Unstructured data consists of narrative text, such as patient medical history and surgical records.</p><p>Structured data preprocessing was performed to ensure data quality and consistency prior to model development. An important characteristic of this dataset is that the hospital information system encodes the absence of an event or measurement as a value of 0 rather than as a conventional missing value (not a number [NaN]). For example, an intraoperative drug dosage of 0 indicates that the drug was not administered, and an airway-equipment size of 0 indicates that the corresponding procedure was not performed. Consequently, the structured feature set contained no conventional missing (NaN) values: across all 78 perioperative model features and the full study cohort (n=33,460), the NaN rate was 0%. A median- and mode-imputation step (for numeric and categorical variables, respectively) was nonetheless retained within the unified modeling pipeline as a safeguard for prospective deployment, where sporadic true missing values may occur; in the present study dataset this step was not triggered because no NaN values were present. All preprocessing procedures were refitted independently within each training fold before being applied to the corresponding validation data, ensuring that any imputation parameters would be derived exclusively from training data and preventing information leakage.</p><p>It is therefore important to distinguish these structurally coded zeros from genuine missing data. To characterize their distribution, we summarized the per-variable zero rate across the cohort. Most numeric features were near-completely recorded, whereas a small number of procedure-specific measurements (eg, airway-equipment sizes and selected intraoperative drug dosages) showed high zero rates that reflect non-performance of the corresponding procedure rather than missing information. A detailed per-variable zero-rate summary is provided as a supplementary file (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Alternative imputation strategies, including K-Nearest Neighbors (KNN) imputation, were explored during preliminary assessment but did not materially affect downstream model performance and were therefore not adopted in the primary pipeline.</p><p>Extreme values were handled using percentile-based capping to reduce the influence of outliers while preserving clinically meaningful variation. Antiemetic administration was retained as a perioperative variable rather than excluded. As this variable may reflect clinicians&#x2019; responses to perceived risk and thus introduce potential confounding, prediction tasks were explicitly defined based on temporal availability, with variables included only when available at the time of prediction. The potential confounding effect of this variable was considered during the interpretation of model performance.</p><p>The study protocol was reviewed and approved by the Institutional Review Board (IRB) of Kaohsiung Medical University Hospital (approval number: KMUHIRB-E(I)-20200040). All data used for analysis were fully deidentified in accordance with institutional data protection and ethical standards. Direct identifiers (eg, patient names, national identification numbers, and dates of birth) were completely removed. For indirect identifiers (eg, admission time, hospital stay, and diagnosis codes), data masking or generalization techniques were applied to minimize the risk of reidentification.</p></sec><sec id="s2-4"><title>Unstructured Text Data Preprocessing</title><p>LLM-based methods for unstructured clinical text processing were implemented using a fixed model version with deterministic settings (temperature=0) to ensure reproducibility. All inputs to the LLM were strictly limited to deidentified text. Direct identifiers (eg, patient names, identification numbers, and exact dates) were completely removed prior to processing. When a commercial API was used, only deidentified text was transmitted through secure, encrypted channels in accordance with institutional data governance policies. No direct identifiers or personally identifiable information were transferred to external services. Alternatively, LLM processing could be conducted within a controlled or self-hosted environment. This design supports reproducible and secure use of LLM technologies within the clinical data pipeline while maintaining compliance with data protection regulations.</p><p>Handling unstructured clinical text requires approaches beyond conventional categorical encoding. We implemented a concept-based framework in which narrative text was normalized and mapped to predefined clinical concepts using LLM-assisted semantic standardization. Ordinal encoding was applied separately to ordered categorical variables, such as American Society of Anesthesiologists (ASA) physical status, by assigning ranked numerical values, enabling machine learning models to capture ordinal relationships effectively. However, more complex narrative text requires semantic interpretation that cannot be fully represented through predefined ordinal or categorical structures alone.</p><p>To extract clinically meaningful signals from medical history, surgical notes, and diagnostic descriptions, we implemented a keyword-based feature construction approach guided by predefined clinical concepts derived from prior literature and domain knowledge. The LLM was used solely to standardize semantically related expressions (eg, synonyms and linguistic variations) into consistent representations aligned with predefined clinical concepts. The identified concepts were subsequently mapped to structured features using a predefined scoring scheme, ensuring consistent and reproducible feature construction. The overall preprocessing pipeline for unstructured clinical text is illustrated in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Clinical data preprocessing framework for unstructured text.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig02.png"/></fig><p>Beyond keyword-based feature construction, we further incorporated Question Answering (QA) Chain, an LLM-based framework built on LangChain, as a post hoc explanation module to support clinical interpretation of model outputs. QAChain operates after model prediction by leveraging structured inputs, including model outputs and derived clinical features, to generate context-aware explanations. With a deterministic setting (temperature=0), the model produces consistent, structured explanations aligned with clinical reasoning and grounded in patient context.</p><p>For example, it may generate: &#x201C;The patient received high-dose opioids and underwent prolonged general anesthesia, both of which contribute significantly to PONV risk.&#x201D; Such outputs provide interpretable narratives that connect model predictions with clinically relevant factors. The QAChain-based explanation process is illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>. The full prompt template, system instructions, and configuration details are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> to ensure reproducibility and transparency.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>QAChain-based post hoc explanation framework. The QAChain module generates structured, natural-language explanations based on model predictions and input features. This module is used solely for interpretability purposes and does not participate in model training, feature construction, or prediction. DL: deep learning; ML: machine learning. QAChain: Question Answering Chain.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig03.png"/></fig></sec><sec id="s2-5"><title>Model Development</title><p>We developed predictive models for POV 24h using a temporally structured dual-task framework aligned with clinical decision-making. Two independent prediction tasks were constructed to prevent information leakage: (1) a preoperative model using variables available before anesthesia induction, and (2) a perioperative model using variables available up to the end of surgery. Only features available prior to each prediction time point were included in the corresponding model.</p><p>For structured data, three ML algorithms were implemented: logistic regression, LightGBM, and XGBoost (Extreme Gradient Boosting). Logistic regression served as an interpretable baseline model, while gradient boosting models (LightGBM and XGBoost) were used as primary predictive approaches due to their strong performance on tabular clinical data and their ability to capture nonlinear relationships and feature interactions.</p><p>Unstructured clinical text features derived from the preprocessing pipeline were incorporated into the model as structured inputs using a predefined keyword-based scoring system. This approach enabled integration of clinically relevant information from narrative data while maintaining methodological rigor and preventing information leakage.</p><p>Model training and evaluation were conducted under a temporal split design to simulate real-world deployment, in which models trained on earlier data were evaluated on unseen future cases. Model performance was assessed using discrimination (receiver operating characteristic&#x2013;area under the curve and precision-recall-area under the curve), calibration (Brier score and calibration plots), and clinically relevant operating metrics. Classification thresholds were selected using Youden&#x2019;s J statistic, and decision curve analysis was performed to evaluate clinical utility. All performance metrics were reported with 95% CIs derived from bootstrap resampling.</p></sec><sec id="s2-6"><title>Interpretability Analysis</title><p>To enhance transparency and support clinical interpretability, we incorporated both model-based and LLM-assisted explanation approaches. First, model-level interpretability was assessed using SHAP. SHAP values were used to quantify the contribution of individual features to model predictions and to derive global feature importance rankings based on the mean absolute SHAP values across samples. These analyses enabled the identification of key predictors associated with POV 24h.</p><p>Second, the keyword-based scoring system provided interpretable representations of unstructured clinical text, enabling examination of text-derived risk factors within the model.</p><p>Finally, we implemented QAChain, an LLM-based post hoc explanation module, to generate context-aware natural-language interpretations of model outputs. QAChain was configured with deterministic settings (temperature=0) to ensure reproducibility. Importantly, this module was not used for predictive modeling, but solely to provide human-readable explanations aligned with clinical reasoning.</p><p>Together, these approaches provide complementary perspectives on model behavior by combining quantitative feature attribution with qualitative explanation, thereby supporting clinical understanding and potential real-world application.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Dataset</title><p>The study population consisted of patients who underwent anesthesia for surgical procedures at Kaohsiung Medical University Hospital between July 2019 and September 2022. The initial dataset included 79,919 anesthesia records comprising 127 variables and 95,304 postoperative interview records comprising 81 variables. These datasets were merged at the surgery level using a unique surgery identifier, resulting in a combined dataset with 208 variables.</p><p>Cohort construction was performed based on predefined inclusion and exclusion criteria to ensure data quality and clinical consistency. Patients younger than 18 years, duplicate records, and cases with invalid or inconsistent timestamps were excluded. Patients with an ASA physical status of 5 or 6 were also excluded due to their extreme clinical risk profiles. In addition, cases without valid postoperative follow-up within 24 hours were excluded from the primary analysis. The cohort selection process is illustrated in <xref ref-type="fig" rid="figure4">Figure 4</xref>. (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>)</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Cohort selection flow diagram.ASA: American Society of Anesthesiologists; POV: postoperative vomiting;</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig04.png"/></fig><p>Prophylactic antiemetic use was retained as a variable and further examined through sensitivity and subgroup analyses. After cohort construction, the final study population consisted of 33,460 surgical cases. Among these, 3607 cases (10.8%) experienced postoperative vomiting within 24 hours (POV 24h), while 29,853 cases (89.2%) did not.</p><p>The primary endpoint was defined as POV 24h. For each surgical case, postoperative interview records across multiple time windows were aggregated at the surgery level. A case was labeled as positive if vomiting occurred at any time within the 24-hour postoperative period.</p><p>Given the class imbalance in POV 24h outcomes, model evaluation focused on metrics that reflect both discrimination and clinical relevance, including receiver operating characteristic &#x2013; area under the curve, precision-recall area under the curve, and operating characteristics such as sensitivity and positive predictive value.</p></sec><sec id="s3-2"><title>Model Performance</title><p>We evaluated three ML models (logistic regression, LightGBM, and XGBoost) for POV 24h. Logistic regression served as an interpretable baseline, while LightGBM and XGBoost were used to capture nonlinear relationships and feature interactions in tabular clinical data. The Apfel score was included as a clinical reference baseline.</p><p>All models were evaluated using a temporal split protocol, in which models were trained on earlier cases and tested on subsequent, unseen cases. Two prediction settings were defined according to clinical workflow: a preoperative model (using variables available before anesthesia induction) and a perioperative model (using variables available up to the end of surgery).</p><p><xref ref-type="table" rid="table1">Table 1</xref> summarizes model performance using only structured variables. Across both prediction settings, discrimination was consistent across model families. Gradient boosting models generally achieved slightly higher receiver operating characteristic-area under the curve and precision-recall-area under the curve than logistic regression, while logistic regression showed lower Brier scores, indicating better calibrated probability estimates. The receiver operating characteristic curves for the preoperative and perioperative prediction models are shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>, demonstrating similar discrimination performance across the two settings, with only a marginal improvement when perioperative variables were included.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Model performance using structured variables only Bootstrap-based 95% CIs (BCa, 1000 resamples) are reported for machine learning models.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" colspan="2">Prediction task and model</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table1fn1">a,b</xref></sup> (95% CI)</td><td align="left" valign="bottom">PR-AUC<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub><italic>1</italic></sub>-score (95% CI)</td><td align="left" valign="bottom">Brier score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Preoperative</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Apfel score</td><td align="left" valign="top">0.606</td><td align="left" valign="top">0.122</td><td align="left" valign="top">0.040</td><td align="left" valign="top">0.097</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.717 (0.695 to 0.739)</td><td align="left" valign="top">0.205 (0.176 to 0.230)</td><td align="left" valign="top">0.217 (0.180 to 0.251)</td><td align="left" valign="top">0.090 (0.086 to 0.094)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">0.722 (0.703 to 0.746)</td><td align="left" valign="top">0.201 (0.174 to 0.226)</td><td align="left" valign="top">0.278 (0.254 to 0.302)</td><td align="left" valign="top">0.173 (0.168 to 0.176)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">0.720 (0.700 to 0.742)</td><td align="left" valign="top">0.200 (0.176 to 0.223)</td><td align="left" valign="top">0.278 (0.256 to 0.300)</td><td align="left" valign="top">0.172 (0.168 to 0.175)</td></tr><tr><td align="left" valign="top" colspan="6">Perioperative</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.720 (0.700 to 0.741)</td><td align="left" valign="top">0.210 (0.177 to 0.234)</td><td align="left" valign="top">0.164 (0.128 to 0.198)</td><td align="left" valign="top">0.083 (0.079 to 0.088)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM</td><td align="left" valign="top">0.730 (0.707 to 0.751)</td><td align="left" valign="top">0.212 (0.184 to 0.238)</td><td align="left" valign="top">0.285 (0.262 to 0.307)</td><td align="left" valign="top">0.170 (0.166 to 0.174)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost</td><td align="left" valign="top">0.730 (0.708 to 0.752)</td><td align="left" valign="top">0.213 (0.184 to 0.238)</td><td align="left" valign="top">0.289 (0.262 to 0.313)</td><td align="left" valign="top">0.168 (0.164 to 0.172)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>ROC: receiver operating characteristic.</p></fn><fn id="table1fn2"><p><sup>b</sup>AUC: area under the curve.</p></fn><fn id="table1fn3"><p><sup>c</sup>PR: precision-recall.</p></fn><fn id="table1fn4"><p><sup>d</sup>LightGBM: Light Gradient Boosting Machine. </p></fn><fn id="table1fn5"><p><sup>e</sup>XGBoost: Extreme Gradient Boosting.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>ROC curve comparison for preoperative and perioperative POV prediction models. AUC: area under the curve; POV: postoperative vomiting; ROC: receiver operating characteristic.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig05.png"/></fig><p>The transition from preoperative to perioperative prediction yielded only marginal improvements in discrimination metrics. Notably, changes in threshold-dependent metrics such as <italic>F</italic><sub>1</sub>-score were inconsistent across models, reflecting differences in operating characteristics rather than overall predictive capability. This suggests that a substantial proportion of predictive information is already available prior to anesthesia induction, with intraoperative variables providing limited incremental value in overall model discrimination.</p><p>The relatively small performance differences between model types further indicate that, under the current feature set, model selection plays a secondary role compared to the availability and timing of clinical information. The Apfel score, a rule-based clinical model based on four binary predictors, showed lower discrimination and <italic>F</italic><sub>1</sub>-scores than ML models.</p><p>Calibration curves for both preoperative and perioperative models are presented in <xref ref-type="fig" rid="figure6">Figure 6</xref>. Overall, predicted probabilities showed reasonable agreement with observed event rates, although some deviation from perfect calibration was observed at higher predicted risk levels. This is consistent with the Brier score results reported in <xref ref-type="table" rid="table1">Table 1</xref>, which show that logistic regression demonstrated slightly better calibration performance. Clinical performance metrics at the Youden-optimal threshold are summarized in <xref ref-type="table" rid="table2">Table 2</xref>. The between-model differences in calibration carry practical implications for the intended clinical use. Because the framework is intended to support selective antiemetic prophylaxis&#x2014;where action depends on whether a patient&#x2019;s estimated risk exceeds an institution-specific threshold&#x2014;well-calibrated probabilities are as important as discrimination. Logistic regression produced the lowest Brier scores and the closest agreement between predicted and observed risk, whereas the gradient boosting models, despite marginally higher discrimination, tended to overestimate risk at the highest predicted-probability levels. This suggests that, when absolute risk estimates rather than rank ordering alone are used to guide prophylaxis, either the better-calibrated logistic regression model or a post hoc recalibration of the gradient boosting models (eg, Platt scaling or isotonic regression) would be preferable.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Calibration curves for preoperative and perioperative POV prediction models. POV: postoperative vomiting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig06.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Clinical performance of machine learning models at the Youden-optimal threshold.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Threshold</td><td align="left" valign="bottom">Sensitivity</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">PPV<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom">Youden J</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Preoperative</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.17</td><td align="left" valign="top">0.600</td><td align="left" valign="top">0.733</td><td align="left" valign="top">0.179</td><td align="left" valign="top">0.950</td><td align="left" valign="top">0.333</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.48</td><td align="left" valign="top">0.613</td><td align="left" valign="top">0.741</td><td align="left" valign="top">0.188</td><td align="left" valign="top">0.952</td><td align="left" valign="top">0.355</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.44</td><td align="left" valign="top">0.671</td><td align="left" valign="top">0.678</td><td align="left" valign="top">0.168</td><td align="left" valign="top">0.955</td><td align="left" valign="top">0.348</td></tr><tr><td align="left" valign="top" colspan="7">Perioperative</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.635</td><td align="left" valign="top">0.712</td><td align="left" valign="top">0.177</td><td align="left" valign="top">0.952</td><td align="left" valign="top">0.347</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM</td><td align="left" valign="top">0.50</td><td align="left" valign="top">0.593</td><td align="left" valign="top">0.762</td><td align="left" valign="top">0.195</td><td align="left" valign="top">0.951</td><td align="left" valign="top">0.355</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost</td><td align="left" valign="top">0.44</td><td align="left" valign="top">0.664</td><td align="left" valign="top">0.683</td><td align="left" valign="top">0.169</td><td align="left" valign="top">0.954</td><td align="left" valign="top">0.347</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PPV: positive predictive value. </p></fn><fn id="table2fn2"><p><sup>b</sup>NPV: negative predictive value.</p></fn><fn id="table2fn3"><p><sup>c</sup>LightGBM: Light Gradient Boosting Machine.</p></fn><fn id="table2fn4"><p><sup>d</sup>XGBoost: Extreme Gradient Boosting.</p></fn></table-wrap-foot></table-wrap><p>We further evaluated the contribution of unstructured clinical text by incorporating deterministic text-derived scores as structured model inputs. <xref ref-type="table" rid="table3">Table 3</xref> presents model performance after including text-derived features. Across models, the inclusion of text-derived features resulted in modest changes in discrimination metrics, with slight improvements observed in several model and task settings. The magnitude of improvement was limited, indicating that structured perioperative variables account for the majority of the predictive signal.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Model performance with text-derived features Bootstrap-based 95% CIs (bias-corrected and accelerated [BCa], 1000 resamples) are reported for machine learning models.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" colspan="2">Prediction task and model</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup><sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">PR-AUC<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Brier score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Preoperative</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.719 (0.699 to 0.742)</td><td align="left" valign="top">0.212 (0.183 to 0.238)</td><td align="left" valign="top">0.209 (0.173 to 0.245)</td><td align="left" valign="top">0.087 (0.082 to 0.092)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">0.722 (0.701 to 0.745)</td><td align="left" valign="top">0.210 (0.181 to 0.236)</td><td align="left" valign="top">0.284 (0.262 to 0.309)</td><td align="left" valign="top">0.172 (0.168 to 0.176)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="top">0.725 (0.701 to 0.745)</td><td align="left" valign="top">0.209 (0.184 to 0.235)</td><td align="left" valign="top">0.285 (0.264 to 0.309)</td><td align="left" valign="top">0.171 (0.168 to 0.175)</td></tr><tr><td align="left" valign="top" colspan="6">Perioperative</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.724 (0.704 to 0.746)</td><td align="left" valign="top">0.215 (0.188 to 0.244)</td><td align="left" valign="top">0.171 (0.137 to 0.206)</td><td align="left" valign="top">0.083 (0.078 to 0.087)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LightGBM</td><td align="left" valign="top">0.731 (0.709 to 0.750)</td><td align="left" valign="top">0.220 (0.190 to 0.244)</td><td align="left" valign="top">0.287 (0.261 to 0.308)</td><td align="left" valign="top">0.169 (0.166 to 0.173)</td></tr><tr><td align="left" valign="top" colspan="2"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost</td><td align="left" valign="top">0.732 (0.710 to 0.752)</td><td align="left" valign="top">0.222 (0.192 to 0.248)</td><td align="left" valign="top">0.287 (0.262 to 0.312)</td><td align="left" valign="top">0.168 (0.164 to 0.172)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ROC: receiver operating characteristic.</p></fn><fn id="table3fn2"><p><sup>b</sup>AUC: area under the curve.</p></fn><fn id="table3fn3"><p><sup>c</sup>PR: precision-recall.</p></fn><fn id="table3fn4"><p><sup>d</sup>LightGBM: Light Gradient Boosting Machine.</p></fn><fn id="table3fn5"><p><sup>e</sup>XGBoost: Extreme Gradient Boosting.</p></fn></table-wrap-foot></table-wrap><p>Text-derived features appear to provide complementary information, enriching the representation of clinical context without substantially altering overall model performance.</p></sec><sec id="s3-3"><title>Decision Curve Analysis</title><p>Decision curve analysis was performed to evaluate the predictive models&#x2019; clinical utility across a range of threshold probabilities (<xref ref-type="fig" rid="figure7">Figure 7</xref>). For both preoperative and perioperative prediction tasks, the machine learning models demonstrated positive net benefit across clinically relevant threshold probabilities, supporting the potential value of model-guided selective prophylaxis strategies compared with treat-all or treat-none approaches.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Decision curve analysis preoperative and perioperative postoperative vomiting (POV) prediction models. (A) Preoperative prediction models. (B) End-of-surgery prediction models. Net benefit is plotted against threshold probability for logistic regression, LightGBM, and XGBoost models, along with treat-all and treat-none reference strategies. LightGBM: Light Gradient Boosting Machine; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig07.png"/></fig></sec><sec id="s3-4"><title>Interpretability Analysis</title><p>To enhance transparency and support clinical interpretation, we adopted a multi-level interpretability approach integrating model-based feature attribution, text-derived feature representation, and LLM-assisted explanation. SHAP was used to quantify feature contributions at the model level, the deterministic keyword-based scoring system provided traceability for unstructured clinical text, and QAChain generated context-aware, human-readable explanations of model predictions. Together, these approaches offer complementary insights into model behavior and support clinically meaningful interpretation.</p></sec><sec id="s3-5"><title>SHAP Analysis</title><p>To interpret the model&#x2019;s internal decision-making mechanism, SHAP analysis was performed on the XGBoost model, which achieved the best overall predictive performance. <xref ref-type="fig" rid="figure8">Figure 8</xref> presents the combined SHAP summary and dependence plots for the preoperative model. The summary plot ranks features according to their global importance based on mean absolute SHAP values, while the dependence plots further illustrate the direction and magnitude of each feature&#x2019;s effect on the predicted PONV risk.</p><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Combined SHAP summary and dependence plots for the XGBoost preoperative model. The summary bar plot displays global feature importance based on mean absolute SHAP values, while the dependence plots illustrate the direction and magnitude of each feature&#x2019;s effect on prediction outcomes. Female sex, red blood cells (RBC), and general anesthesia are associated with increased predicted postoperative nausea and vomiting (PONV) risk, whereas total intravenous anesthesia (TIVA) shows a strong protective effect. Age demonstrates an inverse relationship with risk. Other variables, such as American Society of Anesthesiologists (ASA) status and surgical specialties, exhibit more complex patterns. GA: general anesthesia: PONV: postoperative nausea and vomiting: RBC: red blood cells: SHAP: Shapley Additive Explanations: TIVA: total intravenous anesthesia: WBC: white blood cells; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig08.png"/></fig><p>The most influential features include female sex, total intravenous anesthesia (TIVA), red blood cells, general anesthesia, age, ASA status, and selected surgical specialties. Among these, several variables exhibit clear, clinically meaningful directional patterns. Female sex shows a strong positive association with predicted PONV risk, with consistently higher SHAP values observed in female patients. In contrast, TIVA exhibits a substantial protective effect, with markedly negative SHAP values indicating reduced predicted risk. Age demonstrates an inverse relationship, where younger patients are associated with higher predicted risk, as reflected by decreasing SHAP values with increasing age.</p><p>Continuous variables, such as red blood cells, show a positive trend, indicating that higher values are associated with a higher predicted risk. Similarly, general anesthesia is associated with higher prediction outputs compared to nongeneral anesthesia approaches. Other variables, including ASA status and surgical specialties, exhibit more complex or nonlinear patterns, suggesting potential interactions or confounding effects within the model. These features contribute to prediction but require cautious interpretation.</p><p>Overall, the SHAP analysis indicates that demographic characteristics, anesthetic strategies, and perioperative physiological variables jointly drive model predictions, with several features demonstrating consistent and interpretable directional effects aligned with clinical knowledge. Similar patterns were observed in the perioperative model (<xref ref-type="fig" rid="figure9">Figure 9</xref>), with additional contributions from intraoperative variables such as anesthesia duration and opioid use.</p><fig position="float" id="figure9"><label>Figure 9.</label><caption><p>SHAP summary and dependence plots for the perioperative XGBoost model. The summary plot (top) presents the global importance of features ranked by mean absolute SHAP values, while the dependence plots (bottom) illustrate the relationship between feature values and their contributions to model predictions. Key predictors include sex, anesthesia duration, TIVA, opioid use (fentanyl), age, and laboratory variables. Text-derived features, including concept-based indicators (eg, cx_text_smoker) and procedure-specific features (eg, port-a-catheter implantation), also contribute to model predictions, although their influence is smaller compared to primary clinical variables. BUN: blood urea nitrogen; PCA: patient-controlled analgesia; RBC: red blood cells; SHAP: Shapley Additive Explanations; TIVA: total intravenous anesthesia; WBC: white blood cells; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig09.png"/></fig><p>A structured summary of the top features, including their direction and clinical interpretation, is provided in <xref ref-type="table" rid="table4">Table 4</xref>.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Summary of top SHAP features with direction and clinical interpretation.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Feature</td><td align="left" valign="bottom">Direction</td><td align="left" valign="bottom">Clinical interpretation</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Preoperative model</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female sex</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Strong established risk factor for PONV<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TIVA<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Protective compared to general anesthesia</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RBC<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">May reflect physiological status; associated with increased risk</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GA<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Known to increase PONV risk vs non-GA</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age</td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Younger patients have higher risk</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Colorectal surgery</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Procedure-related risk variation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ASA<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">Nonlinear</td><td align="left" valign="top">Reflects overall health; nonlinear effect</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Orthopedic surgery</td><td align="left" valign="top">Nonlinear</td><td align="left" valign="top">Procedure-specific variation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Oral_Fr</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Airway-related procedural factor</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>WBC<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Weak or indirect association</td></tr><tr><td align="left" valign="top" colspan="3">Perioperative model</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female sex</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Strong risk factor</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Anesthesia duration</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Longer exposure increases risk</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TIVA<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Protective anesthetic strategy</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Fentanyl (intraop)</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Opioid-related risk increase</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age</td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Younger patients at higher risk</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RBC<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Physiological association</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ASA</td><td align="left" valign="top">Nonlinear</td><td align="left" valign="top">Reflects baseline health</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Smoking (text-derived)</td><td align="left" valign="top">&#x2193;</td><td align="left" valign="top">Smoking paradox (known effect)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Surgery type (colorectal)</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Procedure-related risk</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Total_text_score</td><td align="left" valign="top">&#x2191;</td><td align="left" valign="top">Aggregated clinical context</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>PONV: postoperative nausea and vomiting.</p></fn><fn id="table4fn2"><p><sup>b</sup>TIVA: total intravenous anesthesia. </p></fn><fn id="table4fn3"><p><sup>c</sup>RBC: red blood cell.</p></fn><fn id="table4fn4"><p><sup>d</sup>GA: general anesthesia.</p></fn><fn id="table4fn5"><p><sup>e</sup>ASA: American Society of Anesthesiologists.  </p></fn><fn id="table4fn6"><p><sup>f</sup>WBC: white blood cell.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Text-Derived Feature Analysis</title><p>To connect the preprocessing pipeline with model interpretability, we examined how text-derived features constructed from LLM-assisted normalization and concept-based mapping contribute to model predictions.</p><p>The LLM was applied to selected unstructured clinical text fields, including medical history descriptions, surgical notes, and diagnostic narratives. These fields contain heterogeneous linguistic expressions, such as synonyms, abbreviations, and mixed-language representations, which cannot be reliably standardized using simple rule-based methods. Using a deterministic configuration (temperature=0), the LLM was employed to normalize semantically equivalent expressions into consistent representations aligned with predefined clinical concepts (eg, smoking status, prior PONV, and laparoscopic procedures). Importantly, the LLM was not used to generate new features or discover additional concepts, and no outcome-related information was used in this process. Only text fields available prior to the prediction time point were included, ensuring temporal consistency and preventing information leakage.</p><p>Clinically relevant concepts were predefined based on established literature and domain knowledge. Following normalization, clinical narratives were mapped to structured features using deterministic and rule-based procedures. Three types of text-derived features were constructed: (1) concept-based binary indicators (cx_text_*), representing the presence or absence of clinically meaningful conditions; (2) aggregated scoring features (eg, history_score, surgery_score, diagnosis_score, and total_text_score), derived from predefined weighting schemes; and (3) count-based features based on keyword frequency. While all final features are constructed deterministically, the LLM enhances the consistency and robustness of concept extraction from unstructured text.</p><p>Within the XGBoost model, SHAP analysis (<xref ref-type="fig" rid="figure9">Figure 9</xref>) shows that text-derived contributions are concentrated in a small number of concept-based features. In particular, cx_text_smoker and cx_text_ponv_history show substantially higher contributions than most other text-derived features, indicating that clinically meaningful concept indicators account for the majority of the text-derived signal in the model. In comparison, composite scoring variables (eg, history_score, surgery_score, diagnosis_score) show relatively limited contribution, while total_text_score provides a moderate but secondary contribution.</p><p>In addition to concept-based features, certain procedure-specific text-derived features (eg, port-a-catheter implantation) also appear among the top contributors, reflecting localized clinical patterns captured from surgical narratives that are not fully represented by predefined concept categories.</p><p>To further examine how the model uses different representations of clinical text, we grouped text-derived features into concept-based, aggregated-score, and count-based categories and visualized their SHAP contributions (<xref ref-type="fig" rid="figure9">Figure 9</xref>). This analysis shows that the model primarily relies on concept-based indicators, with aggregated- and count-based representations playing a secondary role.</p><p>Overall, these findings indicate that LLM-assisted semantic normalization enables consistent extraction of clinically meaningful concepts, which the model subsequently leverages as structured features. This design supports integrating unstructured clinical text into predictive modeling while preserving interpretability and reproducibility. Additional supporting analyses of text-derived features, including concept hit rates, postoperative vomiting rate comparisons, signal strength, the coverage funnel, and concept-level statistics, are provided in (<xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>).</p></sec><sec id="s3-7"><title>QAChain Reasoning Analysis</title><p>To evaluate the interpretability and reliability of the proposed framework, we conducted a comprehensive analysis of QAChain, an LLM-based post hoc explanation module that generates context-aware, human-readable interpretations of model predictions.</p><p>At the case level, QAChain consistently produced clinically meaningful explanations aligned with patient-specific risk profiles. For high-risk cases, explanations frequently emphasized well-established risk factors such as opioid exposure (eg, intraoperative fentanyl), prolonged anesthesia duration, and general anesthesia. For instance, explanations often highlighted that high-dose opioid administration and extended anesthesia duration jointly contribute to increased PONV risk. In contrast, for low-risk cases, QAChain tended to emphasize protective or mitigating factors, such as the use of TIVA and shorter procedural duration. For intermediate or borderline cases, QAChain generated balanced explanations that incorporated both risk-enhancing and risk-reducing factors, reflecting nuanced and context-sensitive reasoning.</p><p>To quantitatively assess whether QAChain explanations align with model behavior, we evaluated their alignment with SHAP-based feature attributions, which serve as a proxy for model-derived feature importance. Specifically, for each case, the top-k features ranked by SHAP values were compared with the features explicitly referenced in the corresponding QAChain explanation. Alignment was defined as the proportion of SHAP top-k features that were mentioned in the generated explanation.</p><p>As illustrated in <xref ref-type="fig" rid="figure10">Figure 10</xref>, QAChain explanations show high consistency with model-derived feature importance at the surface level. Most clinically relevant features&#x2014;such as intraoperative fentanyl, TIVA, anesthesia duration, patient age, and female sex&#x2014;are frequently reflected in QAChain explanations, achieving near-perfect alignment rates (&#x2248;99%&#x2010;100%). For example, high-frequency features such as female sex (n=2863) and anesthesia duration (n=2023) exhibit alignment rates above 99%, indicating that QAChain reliably captures dominant predictive signals even in large-scale clinical data.</p><fig position="float" id="figure10"><label>Figure 10.</label><caption><p>Feature-level alignment between QAChain explanations and SHAP attributions (perioperative XGBoost model). Each bar represents the proportion of cases in which a SHAP top-k feature is explicitly referenced in the corresponding QAChain-generated explanation. Numbers above bars indicate feature occurrence frequency. Most clinically relevant features align nearly perfectly, demonstrating that QAChain explanations are strongly grounded in model-derived feature importance. ASA: American Society of Anesthesiologists; SHAP: Shapley Additive Explanations; TIVA: total intravenous anesthesia; RBC: red blood cell; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e84260_fig10.png"/></fig><p>However, this alignment metric captures lexical feature mention rather than direct causal faithfulness. Therefore, the results should be interpreted as surface-form concordance between QAChain explanations and SHAP-derived feature importance, rather than definitive evidence of full mechanistic fidelity. Features with lower alignment, such as oral fentanyl use, likely reflect semantic abstraction, where QAChain generalizes specific variables into broader clinical concepts (eg, opioid exposure) rather than consistently preserving fine-grained distinctions. Additionally, variables such as missing values (&#x201C;nan&#x201D;) and categorical surgical department labels are rarely mentioned, suggesting that QAChain selectively prioritizes clinically interpretable and semantically meaningful features in its natural-language explanations.</p><p>Overall, these findings indicate that QAChain produces stable, clinically coherent explanations that are meaningfully related to model attributions, while allowing semantic abstraction and human-readable reasoning.</p><p>To ensure reproducibility, QAChain was configured with deterministic settings (temperature=0), resulting in highly consistent outputs across repeated runs with identical inputs. Furthermore, QAChain explanations were generated from structured, engineered features derived from the modeling pipeline, without direct access to raw unstructured text or outcome variables. This design reduces the risk of information leakage and helps ensure that explanations remain aligned with the model&#x2019;s input space.</p><p>Collectively, these findings indicate that QAChain produces stable explanations that are internally consistent with model-derived signals and are intended to enhance the interpretability of model predictions. Because no formal clinician-based evaluation was performed, this interpretability benefit cannot be objectively confirmed at this stage; the explanations should therefore be regarded as a useful interpretability reference to be validated in future clinician-centered studies.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>This study developed and internally validated a temporally structured clinical decision-support framework for POV 24h using ML and LLM-assisted interpretability methods. The results show that model performance was comparable across algorithms, with gradient boosting models demonstrating slightly higher discrimination and logistic regression showing better calibration. Most predictive information was already available before anesthesia induction, suggesting that clinically actionable risk stratification may be feasible during preoperative assessment and could support earlier prophylaxis planning and perioperative risk management. The limited incremental gain from perioperative variables suggests that a substantial proportion of clinically relevant risk information is already available before surgery. In addition, text-derived features contributed limited but consistent improvements, suggesting a complementary role. Finally, integrating SHAP-based attribution and LLM-based explanation enabled a multi-level interpretability framework that supports a clinically meaningful understanding of model predictions.</p><p>Building upon these findings, this study provides a deeper analysis of a temporally structured framework for POV 24h, integrating structured clinical variables, text-derived features, and multi-level interpretability methods.</p><p>The inclusion of text-derived features yielded consistent but limited performance gains, suggesting that unstructured clinical text provides complementary rather than dominant predictive signals. Importantly, the proposed LLM-assisted preprocessing approach enables consistent extraction of clinically meaningful concepts from heterogeneous narrative data while preserving interpretability and reproducibility.</p><p>From an interpretability perspective, combining SHAP-based feature attribution with QAChain-generated explanations provides a multi-level view of model behavior. SHAP offers quantitative insight into feature contributions, while QAChain translates these signals into human-readable clinical narratives. The alignment analysis suggests that QAChain explanations are meaningfully related to dominant model signals, though the evaluation is based on lexical features and does not establish full causal faithfulness. Instead, QAChain can be interpreted as a translation layer that bridges structured feature attribution and clinically interpretable reasoning. We note that this characterizes the explanations generated by the framework; the extent to which they enhance clinician interpretability or usability remains to be confirmed through future clinician-based evaluation.</p><p>These findings position the proposed framework as a practical approach for enhancing the interpretability of clinical machine learning systems through multi-level explanation, without relying on a single explanation modality. Because a formal clinician-based evaluation was not conducted, the magnitude of this interpretability benefit could not be objectively quantified in the present study and remains to be confirmed through future clinician-centered evaluation.</p><p>Importantly, the role of LLMs in this framework should be interpreted as a deliberate design choice rather than a strategy aimed at maximizing predictive performance. While the inclusion of text-derived features resulted in only modest improvements, this reflects the dominance of structured perioperative variables in this prediction task rather than a limitation of the approach.</p><p>Instead, LLMs were intentionally constrained to deterministic semantic normalization and post hoc explanation rather than being used as autonomous predictive generators, thereby minimizing the risk of introducing additional predictive signal or potential information leakage. Under this design, QAChain functions as a translation layer that converts model-derived feature attributions into clinically interpretable narratives, rather than acting as an independent predictive component.</p><p>This design aligns with emerging directions in clinical AI, where interpretability, transparency, and alignment with clinical reasoning are increasingly recognized as essential for real-world deployment. From this perspective, the limited performance gain associated with LLM integration should be interpreted as evidence that explanations intended to enhance interpretability can be generated without compromising model validity, with the magnitude of this interpretability benefit remaining to be confirmed through future clinician-based evaluation.</p><p>Several limitations should be noted. First, the alignment analysis between QAChain explanations and SHAP features is based on surface-form matching, which may not fully capture semantic equivalence or causal reasoning. Second, QAChain explanations are generated using a fixed prompt and deterministic settings, which may limit variability but do not guarantee true interpretive fidelity. Third, this study is based on retrospective data from a single medical center, which limits external generalizability. Multicenter validation of POV prediction models is inherently challenging due to the heterogeneity of institutional anesthesia practices, antiemetic protocols, and postoperative interview procedures across sites, which can substantially alter both feature distributions and outcome definitions. Furthermore, standardized multi-center data sharing in perioperative research remains limited by privacy regulations and the absence of unified electronic health record schemas. These structural barriers make single-center development studies a common and accepted starting point in this domain. External validation in independent, multi-center cohorts nevertheless remains an essential prerequisite before broader clinical applicability of the proposed models can be assumed, and we caution against generalizing the present single-center findings to other settings without such validation. In addition, the clinical usefulness of QAChain-generated explanations was not formally evaluated by clinicians in this study. Conducting a prospective clinician assessment requires a dedicated deployment environment and institutional review board approval beyond the scope of the current retrospective study. The alignment analysis between QAChain explanations and SHAP attributions provides an objective proxy for consistency, but this does not substitute for human-centered evaluation. Clinician-based usability studies are therefore explicitly identified as a priority for future research.</p><sec id="s4-1"><title>Conclusion</title><p>This study presents a temporally structured framework for POV 24h that integrates structured clinical variables, text-derived features, and multilevel interpretability methods. By explicitly separating preoperative and perioperative prediction tasks, the framework reflects real-world clinical decision-making and avoids information leakage.</p><p>The results show that predictive performance is largely driven by structured clinical variables available before anesthesia induction, with intraoperative variables and text-derived features providing complementary but limited additional contributions. These findings highlight the importance of early-stage risk stratification and suggest that most predictive information is already available before surgery.</p><p>From an interpretability perspective, the proposed framework combines SHAP-based feature attribution with QAChain-generated explanations to provide both quantitative and human-readable insights into model behavior. Rather than serving as a standalone explanation method, QAChain functions as a translation layer that converts structured feature contributions into clinically interpretable narratives. The observed alignment between QAChain explanations and model-derived feature importance indicates meaningful consistency with dominant predictive signals, while allowing for semantic abstraction and contextual reasoning. This consistency supports the intended interpretability contribution of the explanations, although it does not by itself constitute an objective, clinician-based validation of interpretability or usability.</p><p>Overall, this work demonstrates a practical, transparent approach to integrating machine learning and large language models into clinical prediction tasks. By bridging structured modeling and natural-language explanation, the proposed framework provides multi-level explanations intended to enhance interpretability and support more transparent clinical decision-making. Because clinician-based evaluation was beyond the scope of this study, this interpretability benefit should be regarded as a promising reference rather than an objectively validated outcome, and warrants confirmation in future clinician-centered studies. Before broader clinical adoption can be considered, the models also require prospective, multi-center external validation; the present results should therefore be regarded as evidence of internal feasibility rather than established clinical applicability.</p></sec></sec></body><back><notes><sec><title>Funding</title><p>This work was supported in part by the National Science and Technology Council (NSTC), Taiwan, under grant number 113-2221-E-110-076-MY2, and the NSYSU-KMU Joint Research Project #NSYSUKMU 110-P001.</p></sec></notes><fn-group><fn fn-type="con"><p>Writing&#x2013;original draft, validation, methodology, investigation: HJW</p><p>Writing&#x2013;review &#x0026; editing, writing&#x2013;original draft, validation, methodology, investigation, supervision, conceptualization, resources, project administration, funding acquisition: WPL</p><p>Methodology, data curation, investigation: TPG</p><p>Resources, supervision, validation, writing&#x2013;review &#x0026; editing: KIC</p><p>Writing&#x2013;review &#x0026; editing, validation, formal analysis, methodology, project administration: CRW</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ASA</term><def><p>American Society of Anesthesiologists</p></def></def-item><def-item><term id="abb2">CIPONV</term><def><p>clinically significant postoperative nausea and vomiting</p></def></def-item><def-item><term id="abb3">DL</term><def><p>deep learning</p></def></def-item><def-item><term id="abb4">GA</term><def><p>general anesthesia</p></def></def-item><def-item><term id="abb5">KNN</term><def><p>k-nearest neighbor</p></def></def-item><def-item><term id="abb6">LightGBM</term><def><p>Light Gradient Boosting Machine</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb9">NaN</term><def><p>not a number</p></def></def-item><def-item><term id="abb10">PONV</term><def><p>postoperative nausea and vomiting:</p></def></def-item><def-item><term id="abb11">POV 24h</term><def><p>predicting postoperative vomiting within 24 hours</p></def></def-item><def-item><term id="abb12">QA</term><def><p>Question Answering</p></def></def-item><def-item><term id="abb13">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item><def-item><term id="abb14">TCM</term><def><p>traditional Chinese medicine</p></def></def-item><def-item><term id="abb15">TIVA</term><def><p>total intravenous anesthesia</p></def></def-item><def-item><term id="abb16">WBC</term><def><p>white blood cells</p></def></def-item><def-item><term id="abb17">XAI</term><def><p>explainable AI</p></def></def-item><def-item><term id="abb18">XGBoost</term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>C</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>H</given-names> </name></person-group><article-title>A prediction model for postoperative nausea and vomiting after laparoscopic surgery for gynecologic cancers</article-title><source>Clin Ther</source><year>2025</year><month>02</month><volume>47</volume><issue>2</issue><fpage>143</fpage><lpage>147</lpage><pub-id pub-id-type="doi">10.1016/j.clinthera.2024.11.018</pub-id><pub-id pub-id-type="medline">39645472</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amirshahi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Behnamfar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Badakhsh</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Prevalence of postoperative nausea and vomiting: a systematic review and meta-analysis</article-title><source>Saudi J Anaesth</source><year>2020</year><volume>14</volume><issue>1</issue><fpage>48</fpage><lpage>56</lpage><pub-id pub-id-type="doi">10.4103/sja.SJA_401_19</pub-id><pub-id pub-id-type="medline">31998020</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lerman</surname><given-names>J</given-names> </name></person-group><article-title>Surgical and patient factors involved in postoperative nausea and vomiting</article-title><source>Br J Anaesth</source><year>1992</year><volume>69</volume><issue>7 Suppl 1</issue><fpage>24S</fpage><lpage>32S</lpage><pub-id pub-id-type="doi">10.1093/bja/69.supplement_1.24s</pub-id><pub-id pub-id-type="medline">1486011</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Belani</surname><given-names>KG</given-names> </name><name name-style="western"><surname>Bergese</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Fourth consensus guidelines for the management of postoperative nausea and vomiting</article-title><source>Anesth Analg</source><year>2020</year><month>08</month><volume>131</volume><issue>2</issue><fpage>411</fpage><lpage>448</lpage><pub-id pub-id-type="doi">10.1213/ANE.0000000000004833</pub-id><pub-id pub-id-type="medline">32467512</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>TJ</given-names> </name></person-group><article-title>Risk factors for postoperative nausea and vomiting</article-title><source>Anesth Analg</source><year>2006</year><month>06</month><volume>102</volume><issue>6</issue><fpage>1884</fpage><lpage>1898</lpage><pub-id pub-id-type="doi">10.1213/01.ANE.0000219597.16143.4D</pub-id><pub-id pub-id-type="medline">16717343</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Palazzo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Evans</surname><given-names>R</given-names> </name></person-group><article-title>Logistic regression analysis of fixed patient factors for postoperative sickness: a model for risk assessment</article-title><source>Br J Anaesth</source><year>1993</year><month>02</month><volume>70</volume><issue>2</issue><fpage>135</fpage><lpage>140</lpage><pub-id pub-id-type="doi">10.1093/bja/70.2.135</pub-id><pub-id pub-id-type="medline">8435254</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Glebov</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lazebnik</surname><given-names>T</given-names> </name><name name-style="western"><surname>Katsin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Orkin</surname><given-names>B</given-names> </name><name name-style="western"><surname>Berkenstadt</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bunimovich-Mendrazitsky</surname><given-names>S</given-names> </name></person-group><article-title>Predicting postoperative nausea and vomiting using machine learning: a model development and validation study</article-title><source>BMC Anesthesiol</source><year>2025</year><month>03</month><day>20</day><volume>25</volume><issue>1</issue><fpage>135</fpage><pub-id pub-id-type="doi">10.1186/s12871-025-02987-2</pub-id><pub-id pub-id-type="medline">40114048</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mizuguchi</surname><given-names>T</given-names> </name><name name-style="western"><surname>Sawamura</surname><given-names>S</given-names> </name></person-group><article-title>Machine learning-based causal models for predicting the response of individual patients to dexamethasone treatment as prophylactic antiemetic</article-title><source>Sci Rep</source><year>2023</year><month>05</month><day>9</day><volume>13</volume><issue>1</issue><fpage>37161041</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-34505-0</pub-id><pub-id pub-id-type="medline">37161041</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thottakkara</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ozrazgat-Baslanti</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hupf</surname><given-names>BB</given-names> </name><etal/></person-group><article-title>Application of machine learning techniques to high-dimensional clinical data to forecast postoperative complications</article-title><source>PLoS One</source><year>2016</year><volume>11</volume><issue>5</issue><fpage>e0155705</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0155705</pub-id><pub-id pub-id-type="medline">27232332</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Diemunsch</surname><given-names>P</given-names> </name><name name-style="western"><surname>Habib</surname><given-names>AS</given-names> </name><etal/></person-group><article-title>Consensus guidelines for the management of postoperative nausea and vomiting</article-title><source>Anesth Analg</source><year>2014</year><month>01</month><volume>118</volume><issue>1</issue><fpage>85</fpage><lpage>113</lpage><pub-id pub-id-type="doi">10.1213/ANE.0000000000000002</pub-id><pub-id pub-id-type="medline">24356162</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoshijima</surname><given-names>H</given-names> </name><name name-style="western"><surname>Miyazaki</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mitsui</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Omachi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yamauchi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mizuta</surname><given-names>K</given-names> </name></person-group><article-title>Machine learning-based identification of the risk factors for postoperative nausea and vomiting in adults</article-title><source>PLoS One</source><year>2024</year><volume>19</volume><issue>8</issue><fpage>e0308755</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0308755</pub-id><pub-id pub-id-type="medline">39146357</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Cheon</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>MG</given-names> </name><etal/></person-group><article-title>Postoperative nausea and vomiting prediction: machine learning insights from a comprehensive analysis of perioperative data</article-title><source>Bioengineering (Basel)</source><year>2023</year><month>10</month><day>1</day><volume>10</volume><issue>10</issue><fpage>1152</fpage><pub-id pub-id-type="doi">10.3390/bioengineering10101152</pub-id><pub-id pub-id-type="medline">37892882</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name></person-group><article-title>Predicting early postoperative PONV using multiple machine-learning- and deep-learning-algorithms</article-title><source>BMC Med Res Methodol</source><year>2023</year><month>05</month><day>31</day><volume>23</volume><issue>1</issue><fpage>133</fpage><pub-id pub-id-type="doi">10.1186/s12874-023-01955-z</pub-id><pub-id pub-id-type="medline">37259031</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arik</surname><given-names>S&#x00D6;</given-names> </name><name name-style="western"><surname>Pfister</surname><given-names>T</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Arik</surname><given-names>S&#x00D6;</given-names> </name><name name-style="western"><surname>Pfister</surname><given-names>T</given-names> </name></person-group><article-title>TabNet: Attentive Interpretable Tabular Learning</article-title><source>AAAI</source><year>2021</year><volume>35</volume><issue>8</issue><fpage>6679</fpage><lpage>6687</lpage><pub-id pub-id-type="doi">10.1609/aaai.v35i8.16826</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salih</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Raisi&#x2010;Estabragh</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Galazzo</surname><given-names>IB</given-names> </name><etal/></person-group><article-title>A perspective on explainable artificial intelligence methods: SHAP and LIME</article-title><source>Adv Intell Syst</source><year>2025</year><month>01</month><access-date>2026-07-24</access-date><volume>7</volume><issue>1</issue><comment><ext-link ext-link-type="uri" xlink:href="https://advanced.onlinelibrary.wiley.com/toc/26404567/7/1">https://advanced.onlinelibrary.wiley.com/toc/26404567/7/1</ext-link></comment><pub-id pub-id-type="doi">10.1002/aisy.202400304</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paneru</surname><given-names>B</given-names> </name><name name-style="western"><surname>Paneru</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sapkota</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Poudyal</surname><given-names>R</given-names> </name></person-group><article-title>Enhancing healthcare with AI: sustainable AI and IoT-powered ecosystem for patient aid and interpretability analysis using SHAP</article-title><source>Meas: Sens</source><year>2024</year><month>12</month><volume>36</volume><fpage>101305</fpage><pub-id pub-id-type="doi">10.1016/j.measen.2024.101305</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Interpretability of clinical decision support systems based on artificial intelligence from technological and medical perspective: a systematic review</article-title><source>J Healthc Eng</source><year>2023</year><volume>2023</volume><issue>1</issue><fpage>9919269</fpage><pub-id pub-id-type="doi">10.1155/2023/9919269</pub-id><pub-id pub-id-type="medline">36776958</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Multi-feature, Chinese-Western medicine-integrated prediction model for diabetic peripheral neuropathy based on machine learning and SHAP</article-title><source>Diabetes Metab Res Rev</source><year>2024</year><month>05</month><volume>40</volume><issue>4</issue><fpage>e3801</fpage><pub-id pub-id-type="doi">10.1002/dmrr.3801</pub-id><pub-id pub-id-type="medline">38616511</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name></person-group><article-title>A machine learning-based prediction model for delayed clinically important postoperative nausea and vomiting in high-risk patients undergoing laparoscopic gastrointestinal surgery</article-title><source>Am J Surg</source><year>2024</year><month>11</month><volume>237</volume><fpage>115912</fpage><pub-id pub-id-type="doi">10.1016/j.amjsurg.2024.115912</pub-id><pub-id pub-id-type="medline">39182286</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aronson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Murray</surname><given-names>S</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Roadmap for transforming preoperative assessment to preoperative optimization</article-title><source>Anesth Analg</source><year>2020</year><month>04</month><volume>130</volume><issue>4</issue><fpage>811</fpage><lpage>819</lpage><pub-id pub-id-type="doi">10.1213/ANE.0000000000004571</pub-id><pub-id pub-id-type="medline">31990733</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rashid-Kolvear</surname><given-names>M</given-names> </name><name name-style="western"><surname>Waseem</surname><given-names>R</given-names> </name><name name-style="western"><surname>Englesakis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>F</given-names> </name></person-group><article-title>Virtual preoperative assessment in surgical patients: a systematic review and meta-analysis</article-title><source>J Clin Anesth</source><year>2021</year><month>12</month><volume>75</volume><fpage>110540</fpage><pub-id pub-id-type="doi">10.1016/j.jclinane.2021.110540</pub-id><pub-id pub-id-type="medline">34649158</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdel Malek</surname><given-names>M</given-names> </name><name name-style="western"><surname>van Velzen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dahan</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Generation of preoperative anaesthetic plans by ChatGPT-4.0: a mixed-method study</article-title><source>Br J Anaesth</source><year>2025</year><month>05</month><volume>134</volume><issue>5</issue><fpage>1333</fpage><lpage>1340</lpage><pub-id pub-id-type="doi">10.1016/j.bja.2024.08.038</pub-id><pub-id pub-id-type="medline">39547871</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval augmented generation for 10 large language models and its generalizability in assessing medical fitness</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>5</day><volume>8</volume><issue>1</issue><fpage>187</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01519-z</pub-id><pub-id pub-id-type="medline">40185842</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sedaghat</surname><given-names>S</given-names> </name></person-group><article-title>Early applications of ChatGPT in medical practice, education and research</article-title><source>Clin Med (Lond)</source><year>2023</year><month>05</month><volume>23</volume><issue>3</issue><fpage>278</fpage><lpage>279</lpage><pub-id pub-id-type="doi">10.7861/clinmed.2023-0078</pub-id><pub-id pub-id-type="medline">37085182</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dave</surname><given-names>T</given-names> </name><name name-style="western"><surname>Athaluri</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name></person-group><article-title>ChatGPT in medicine: an overview of its applications, advantages, limitations, future prospects, and ethical considerations</article-title><source>Front Artif Intell</source><year>2023</year><volume>6</volume><fpage>1169595</fpage><pub-id pub-id-type="doi">10.3389/frai.2023.1169595</pub-id><pub-id pub-id-type="medline">37215063</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Patterson</surname><given-names>BL</given-names> </name><etal/></person-group><article-title>Using AI-generated suggestions from ChatGPT to optimize clinical decision support</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>06</month><day>20</day><volume>30</volume><issue>7</issue><fpage>1237</fpage><lpage>1245</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad072</pub-id><pub-id pub-id-type="medline">37087108</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name></person-group><article-title>PONV Management in adult patients: evidence-based summary</article-title><source>J Perianesth Nurs</source><year>2024</year><month>12</month><volume>39</volume><issue>6</issue><fpage>1095</fpage><lpage>1103</lpage><pub-id pub-id-type="doi">10.1016/j.jopan.2024.01.027</pub-id><pub-id pub-id-type="medline">38935008</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shim</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Ryu</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>EA</given-names> </name><etal/></person-group><article-title>Machine learning for prediction of postoperative nausea and vomiting in patients with intravenous patient-controlled analgesia</article-title><source>PLoS One</source><year>2022</year><volume>17</volume><issue>12</issue><fpage>e0277957</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0277957</pub-id><pub-id pub-id-type="medline">36548346</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oei</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Bakkes</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mischi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bouwman</surname><given-names>RA</given-names> </name><name name-style="western"><surname>van Sloun</surname><given-names>RJG</given-names> </name><name name-style="western"><surname>Turco</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence in clinical decision support and the prediction of adverse events</article-title><source>Front Digit Health</source><year>2025</year><volume>7</volume><fpage>1403047</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2025.1403047</pub-id><pub-id pub-id-type="medline">40520218</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ghaderzadeh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rafie</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Salehnasab</surname><given-names>C</given-names> </name></person-group><article-title>Explainable extratreeclassifier model for early detection of type 2 diabetes: evidence from the PERSIAN Dena Cohort</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>12</month><day>31</day><volume>26</volume><issue>1</issue><fpage>36</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-03333-9</pub-id><pub-id pub-id-type="medline">41469992</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ruiz</surname><given-names>NI</given-names> </name><name name-style="western"><surname>Cardona Salazar</surname><given-names>I</given-names> </name><name name-style="western"><surname>Naranjo Palacio</surname><given-names>LX</given-names> </name><name name-style="western"><surname>Agudelo Agudelo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ledesma Parra</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Flores Rodriguez</surname><given-names>JC</given-names> </name></person-group><article-title>Accuracy and reliability of artificial intelligence in surgical decision-making: a literature review</article-title><source>Cureus</source><year>2025</year><month>10</month><volume>17</volume><issue>10</issue><fpage>e95337</fpage><pub-id pub-id-type="doi">10.7759/cureus.95337</pub-id><pub-id pub-id-type="medline">41287740</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>P</given-names> </name><name name-style="western"><surname>Biessmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Teubner</surname><given-names>T</given-names> </name></person-group><article-title>Transparency and trust in artificial intelligence systems</article-title><source>J Decis Syst</source><year>2020</year><month>10</month><day>1</day><volume>29</volume><issue>4</issue><fpage>260</fpage><lpage>278</lpage><pub-id pub-id-type="doi">10.1080/12460125.2020.1819094</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walmsley</surname><given-names>J</given-names> </name></person-group><article-title>Artificial intelligence and the value of transparency</article-title><source>AI &#x0026; Soc</source><year>2021</year><month>06</month><volume>36</volume><issue>2</issue><fpage>585</fpage><lpage>595</lpage><pub-id pub-id-type="doi">10.1007/s00146-020-01066-z</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Feature timing classification and temporal availability mapping for model development.</p><media xlink:href="medinform_v14i1e84260_app1.docx" xlink:title="DOCX File, 43 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Per-variable zero-rate and true missingness summary of the structured feature set.</p><media xlink:href="medinform_v14i1e84260_app2.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Configuration, prompt template, system instructions, and governance details of the QAChain large language model explanation module.</p><media xlink:href="medinform_v14i1e84260_app3.docx" xlink:title="DOCX File, 31 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Baseline characteristics of the study population stratified by postoperative vomiting within 24 hours.</p><media xlink:href="medinform_v14i1e84260_app4.docx" xlink:title="DOCX File, 34 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Supporting analysis of text-derived features, including concept hit rates, postoperative vomiting rate comparisons, signal strength, the coverage funnel, and concept-level statistics.</p><media xlink:href="medinform_v14i1e84260_app5.docx" xlink:title="DOCX File, 404 KB"/></supplementary-material></app-group></back></article>