<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e98207</article-id><article-id pub-id-type="doi">10.2196/98207</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Model&#x2013;Based Clinical Decision Support for Antibiotic Selection and Dose Recommendation in Hospitalized Patients With Pneumonia: Multicenter Retrospective Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhang</surname><given-names>Yang</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Li</surname><given-names>Li</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tan</surname><given-names>Chunting</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ji</surname><given-names>Mengyuan</given-names></name><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tian</surname><given-names>Xican</given-names></name><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mu</surname><given-names>Xiangdong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Jun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Gu</surname><given-names>Yu</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Liu</surname><given-names>Honglei</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>School of Biomedical Engineering, Capital Medical University</institution><addr-line>No. 10 Xitoutiao, You&#x2019;anmenwai, Fengtai District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff2"><institution>Beijing Key Laboratory of Clinical Engineering Solutions for Mental Health, Capital Medical University</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Respiratory and Critical Care Medicine, School of Clinical Medicine, Tsinghua Medicine, Beijing Tsinghua Changgung Hospital, Tsinghua University</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Respiratory Medicine, Beijing Friendship Hospital, Capital Medical University</institution><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Wei</surname><given-names>Lei</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Idowu</surname><given-names>Nike</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Honglei Liu, PhD, School of Biomedical Engineering, Capital Medical University, No. 10 Xitoutiao, You&#x2019;anmenwai, Fengtai District, Beijing, 100069, China, 86 010-83911542; <email>liuhonglei@ccmu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>8</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e98207</elocation-id><history><date date-type="received"><day>14</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>12</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>14</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yang Zhang, Li Li, Chunting Tan, Mengyuan Ji, Xican Tian, Xiangdong Mu, Jun Li, Yu Gu, Honglei Liu. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 4.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e98207"/><abstract><sec><title>Background</title><p>Pneumonia is a common infectious disease, and antibiotic treatment in hospitalized patients must balance efficacy, safety, and resistance risk. However, antibiotic selection and dose adjustment still rely heavily on clinician experience. Although large language models (LLMs) are promising for clinical reasoning, their direct use for antibiotic selection and dose recommendation is limited by hallucinations and weak adherence to clinical constraints.</p></sec><sec><title>Objective</title><p>This study aimed to develop and externally validate a constrained LLM-based clinical decision support pipeline for antibiotic selection and dose recommendation in hospitalized patients with pneumonia.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a multicenter retrospective study using electronic health record narratives, antibiotic orders, and laboratory indicators of hepatic and renal function from 331 hospitalized patients with pneumonia from 2 hospitals in China. The development cohort included 233 patients, and the external validation cohort included 98 patients. The pipeline integrated dual-branch retrieval (similar-case vector retrieval plus guideline-based knowledge graph retrieval), clinician-defined rule constraints, and hybrid-context reasoning. DeepSeek-V3, GLM-4.6, and GPT-4o were evaluated using <italic>F</italic><sub>1</sub>-score and Jaccard accuracy.</p></sec><sec sec-type="results"><title>Results</title><p>On the internal test set, the full pipeline using DeepSeek-V3 achieved the best performance, with an <italic>F</italic><sub>1</sub>-score of 0.8110 (95% CI 0.7371-0.8762) and Jaccard accuracy of 0.7624 (95% CI 0.6810-0.8386) for antibiotic selection and an <italic>F</italic><sub>1</sub>-score of 0.7538 (95% CI 0.6671-0.8329) and Jaccard accuracy of 0.7076 (95% CI 0.6145-0.7938) for joint antibiotic selection plus dosing recommendation. On the external validation set, performance remained high, with an <italic>F</italic><sub>1</sub>-score of 0.8605 (95% CI 0.7891-0.9252) and Jaccard accuracy of 0.8571 (95% CI 0.7857-0.9184) for antibiotic selection, and an <italic>F</italic><sub>1</sub>-score of 0.8503 (95% CI 0.7789-0.9150) and Jaccard accuracy of 0.8469 (95% CI 0.7755-0.9133) for antibiotic selection plus dosing recommendation. The system also provided traceable evidence and rule trigger information to support clinician review.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>A constrained, retrieval-augmented LLM pipeline improved the consistency and interpretability of antibiotic selection and dose recommendation for hospitalized patients with pneumonia and provided preliminary evidence of cross-site generalizability.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>retrieval-augmented generation</kwd><kwd>pneumonia</kwd><kwd>medication recommendation</kwd><kwd>knowledge graph</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Pneumonia is a major infectious disease worldwide and remains a leading cause of morbidity and mortality. Accordingly, standardized diagnosis, treatment, and antimicrobial management are essential components of modern health care. In China, the burden of community-acquired and hospital-acquired pneumonia remains substantial, driven in part by rapid population aging and increasing numbers of immunosuppressed patients and patients with multimorbidity [<xref ref-type="bibr" rid="ref1">1</xref>]. At the same time, the evolution of bacterial resistance continues to outpace the development and clinical availability of new antimicrobials, making inappropriate antibiotic use a major global public health challenge [<xref ref-type="bibr" rid="ref2">2</xref>]. In real-world practice, inappropriate antimicrobial prescribing remains common and may compromise treatment efficacy, increase the risk of adverse drug events, accelerate the spread of antimicrobial resistance, and raise health care costs [<xref ref-type="bibr" rid="ref3">3</xref>]. For hospitalized patients with pneumonia, antibiotic selection often requires early empirical treatment before definitive microbiological results become available. At the same time, patient-specific factors such as hepatic or renal impairment, older age, and multimorbidity can substantially influence antibiotic selection and dose recommendation. These challenges highlight the need for clinical decision support (CDS) tools that can assist individualized antimicrobial therapy while remaining aligned with guideline constraints.</p><p>Before the advent of large language models (LLMs), biomedical informatics had long explored CDS approaches. Early rule-based and knowledge-based expert systems [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>] encoded medical knowledge as &#x201C;if-then&#x201D; rules with strong interpretability but relied heavily on manual maintenance and were difficult to update in response to evolving guidelines, emerging evidence, and complex clinical contexts. Subsequently, traditional natural language processing and machine learning methods were applied to tasks such as clinical text structuring, risk prediction, and adverse event detection [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. Deep learning models, including recurrent neural networks and bidirectional encoder representations from transformers, further improved clinical text representations [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. However, prediction or classification remained the dominant paradigm at this stage, often without providing an auditable chain of clinical reasoning [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. In the area of medication recommendation and individualized prescribing, previous studies have mainly focused on medication information extraction, prescription review alerts, or medication prediction [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>], with limited ability to jointly incorporate guideline constraints, patient-specific variation, and cross-modal clinical evidence. In addition, limited interpretability and traceability have remained major barriers to clinical adoption in high-risk medication decisions.</p><p>In recent years, advances in LLMs for medical context understanding and text generation have expanded biomedical informatics from &#x201C;structuring and prediction&#x201D; toward &#x201C;clinical text reasoning and explainable generation&#x201D; [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. These models have shown promise in tasks such as medical record information extraction and clinical question answering [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. For medication recommendation, LLMs can integrate illness descriptions, prior medications, and test results to generate candidate regimens and rationales [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. With retrieval-augmented generation, external sources such as guideline statements, drug labels, and local clinical pathways can be incorporated into the reasoning process, thereby improving knowledge coverage and output consistency [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref29">29</xref>]. However, in high-risk settings such as antibiotic selection and dose recommendation, important safety challenges remain. Adherence to fine-grained constraints, including dose limits, contraindications, and hepatic or renal dose adjustment, is often unstable [<xref ref-type="bibr" rid="ref26">26</xref>]. Model conclusions may also vary according to input phrasing and context organization [<xref ref-type="bibr" rid="ref23">23</xref>], and hallucinations may produce potentially unsafe recommendations [<xref ref-type="bibr" rid="ref30">30</xref>]. To support safe antimicrobial decision-making, LLM outputs should be grounded in traceable evidence, guided by explicit clinical constraints, and accompanied by reasoning that can be reviewed and verified by clinicians. These limitations highlight the need for constrained, traceable, and auditable LLM-based frameworks that can systematically incorporate external evidence and explicit clinical rules into the reasoning process.</p><p>Against this background, we developed and validated an LLM-based CDS pipeline for antibiotic selection and dose recommendation in hospitalized patients with pneumonia. Designed as an assistive tool for antimicrobial stewardship and medication review rather than a replacement for clinician judgment, the proposed framework is conceptually aligned with the information-gathering process of infectious disease consultation, in which clinicians integrate patient characteristics, laboratory findings, prior experience, and guideline recommendations to support empirical antimicrobial decisions. To support this process, the framework combines two key strategies to improve the safety and consistency of LLM-based recommendations: (1) a dual-branch retrieval mechanism that integrates similar-case vector retrieval with guideline-based knowledge graph retrieval to connect patient-specific information with external evidence and (2) clinician-defined rule constraints that explicitly encode dose adjustment, contraindications, and hepatic or renal function considerations. We further evaluated this framework on multicenter real-world data through internal testing and external validation, with comparisons across multiple LLMs (DeepSeek-V3, GLM-4.6 [Z.ai], and GPT-4o [OpenAI]). By generating recommendations together with traceable evidence and rule trigger information, the proposed system aimed to support rapid clinician verification and final decision-making while improving the interpretability, robustness, and clinical applicability of LLM-assisted antimicrobial management.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Datasets</title><p>We conducted a multicenter retrospective study using electronic health record narratives, antibiotic orders (including prescription elements such as dose and route), and biochemical laboratory reports from hospitalized patients with pneumonia at 2 hospitals in China. Beijing Tsinghua Changgung Hospital served as the development cohort (n=233; January 2020-January 2024) and was split into training and internal test sets using a 7:3 ratio. Beijing Friendship Hospital served as the external validation cohort (n=98; April 2019-September 2019).</p><p>Inclusion criteria were as follows: (1) an admission diagnosis of community-acquired or hospital-acquired pneumonia, (2) complete clinical records, and (3) at least one antibiotic order during hospitalization. The exclusion criterion was confirmed COVID-19 pneumonia.</p><p>We constructed the dataset for model input and evaluation using three data domains: (1) patient context, including admission narratives such as the chief concern, history of present illness, physical examination, and initial diagnosis; (2) medication orders, including antibiotic prescriptions and related elements such as drug name, dose, route, and dosing frequency where available; and (3) laboratory indicators, including 7 routinely used hepatic and renal function indicators closely related to antimicrobial safety (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), which were used to support dose recommendation and safety assessment.</p><p>For evaluation, raw antibiotic orders were not directly used as the gold standard but served as source information for reference label construction. Two experienced respiratory physicians reviewed the admission records and laboratory results to determine the final reference regimens. Therefore, model performance was evaluated against expert-adjudicated reference regimens rather than unreviewed raw electronic health record antibiotic orders.</p></sec><sec id="s2-2"><title>Model</title><sec id="s2-2-1"><title>Overview</title><p>This study aimed to develop and validate a retrieval-augmented medication decision pipeline for antibiotic selection and dose recommendation. The pipeline takes admission notes and 7 hepatic and renal function indicators as inputs and consists of three core components: (1) a dual-branch retrieval framework, which integrates similar-case vector retrieval with guideline-based knowledge graph retrieval; (2) clinician-defined rule constraints, in which clinician-curated rules are injected as explicit constraints; and (3) a hybrid-context reasoning module, which integrates patient-specific information, retrieved evidence, and rule constraints to generate antibiotic and dose recommendations with traceable supporting evidence for clinician review. The systematic architecture and workflow of the framework are illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Systematic architecture of the proposed retrieval-augmented medication decision pipeline. ALT: alanine aminotransferase; AST: aspartate aminotransferase; BUN: blood urea nitrogen; CoT: chain of thought.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e98207_fig01.png"/></fig></sec><sec id="s2-2-2"><title>Dual-Branch Retrieval Framework</title><sec id="s2-2-2-1"><title>Similar-Case Vector Retrieval</title><p>To enable robust retrieval from noisy clinical narratives, we adopted a summarize-embed-retrieve workflow. For each historical case, an LLM first generated a structured clinical summary from the admission note and laboratory report covering (1) baseline characteristics, such as demographics, symptoms, and vital signs; (2) examinations and tests, such as imaging findings, infection-related markers, and hepatic and renal function, with laboratory abnormalities discretized into 4 levels (normal, mild, moderate, and severe); and (3) medical history, such as comorbidities, allergies, and prior medications. These summaries were then encoded into dense vectors using the pretrained embedding model bge-m3 to construct a case memory. For a new patient, cosine similarity was calculated against the case memory, and the top 5 most similar cases together with their antibiotic regimens were retrieved as experience-based context for downstream reasoning. The top-5 setting was chosen empirically to balance similar-case coverage with contextual noise and prompt length constraints. The similar-case retrieval configuration and the prompt template used for similar-case summarization are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-2-2-2"><title>Guideline-Based Knowledge Graph Retrieval</title><p>To incorporate evidence-based knowledge and explicit clinical constraints, we used authoritative clinical guidelines, including the Chinese Guidelines for the Diagnosis and Treatment of Community-Acquired Pneumonia in Adults (2016 edition) [<xref ref-type="bibr" rid="ref31">31</xref>]. Using the LightRAG framework, we performed structured extraction of guideline text to construct a decision-oriented knowledge graph containing 1637 entities and 2077 semantic relations. The detailed parameters are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The workflow included 3 steps. First, key medical concepts were extracted and normalized, including clinical symptoms, disease classification and severity, pathogen risk, drugs and administration routes, hepatic and renal function status, contraindications and precautions, and dose adjustment criteria. Second, core relationships among these entities were extracted to support constrained reasoning; relation types included &#x201C;recommended_for,&#x201D; &#x201C;contraindicated_in,&#x201D; &#x201C;dose_adjustment_for,&#x201D; &#x201C;not_recommended_with,&#x201D; &#x201C;covers_pathogen,&#x201D; and other clinical-supporting relations. Third, the extracted information was organized into head-relation-tail triples, with guideline paragraphs or clauses retained as source metadata to enable evidence traceability.</p><p>During inference, the system retrieved knowledge snippets or subgraphs relevant to antibiotic selection, contraindications, and hepatic or renal dose adjustment based on the patient context and key laboratory indicators. These retrieved results were then provided to the downstream reasoning module as supporting evidence.</p></sec></sec></sec><sec id="s2-3"><title>Clinician-Defined Rule Constraints</title><p>To improve the safety and clinical compliance of the recommendations, we incorporated clinician-defined rule constraints curated by physicians. These rules covered care setting and severity stratification, pathogen risk and coverage strategies, restrictions and contraindications for combination therapy, and key boundary conditions related to hepatic and renal function. The rules were injected into the structured prompt as explicit constraint context to guide antibiotic selection and dose recommendation. The representative rules are provided in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-4"><title>Hybrid-Context Reasoning</title><p>We integrated three sources of information into a unified hybrid context: (1) patient context, including admission narratives and key laboratory indicators; (2) experience context, including the top 5 similar cases and their antibiotic regimens; and (3) constraint context, including clinician-defined rule constraints and guideline or knowledge graph evidence retrieved using LightRAG.</p><p>During inference, the hybrid context was organized into a structured prompt and submitted to LightRAG in hybrid mode. All model inference was conducted in Chinese. The comprehensive prompt template, including the detailed instructions for context integration and reasoning steps, is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Guided by key entities and cues in the patient context, the system retrieved local entity-level details and global relational summaries related to antibiotic selection, contraindications, and hepatic or renal dose adjustment. These retrieved results were then combined with the patient context and rule constraints for downstream reasoning. On the basis of this integrated context, the model generated antibiotic and dose recommendations through a stepwise reasoning process. The final JSON array was extracted from each model response and parsed using Python&#x2019;s json.loads() function (Python Software Foundation). If parsing failed or the output was empty, generation was repeated up to 5 times. No manual correction was performed.</p></sec><sec id="s2-5"><title>Model Implementation and Evaluation Metrics</title><p>We implemented the proposed pipeline using DeepSeek-V3, GLM-4.6, and GPT-4o. Detailed model implementation settings are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. To examine the contribution of each module, we evaluated four categories of experimental settings: (1) a <italic>base model</italic> without retrieval augmentation or clinician-defined rule constraints; (2) <italic>single-module variants</italic>, in which only one component was added; (3) <italic>dual-module variants</italic>, in which 2 components were combined; and (4) the <italic>full pipeline</italic>, which integrated similar-case vector retrieval, guideline-based knowledge graph retrieval, and clinician-defined rule constraints.</p><p>The framework was evaluated on both the internal test set and the external validation cohort. Two tasks were assessed: <italic>antibiotic selection</italic> and <italic>joint antibiotic selection plus dosing recommendation</italic>. For the antibiotic selection task, only the standardized medication name was considered during matching. For the antibiotic selection plus dosing task, medication items were matched after standardizing drug name, dose, dosing frequency, and route of administration. Dose values were unit normalized when possible, and a 5% numeric tolerance was allowed to avoid penalizing minor unit conversion or formatting differences. Performance was measured using Jaccard accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score. All metrics were calculated at the patient level based on true-positive, false-positive, and false-negative medication items and then averaged across cases. We used patient-level bootstrap resampling with 10,000 iterations to calculate 95% CIs for the performance metrics. To further assess robustness and clinical applicability, we conducted subgroup analyses in high-risk populations, including older patients and patients with hepatic or renal impairment. The detailed subgroup classification criteria are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Cross-site generalizability was examined using the external validation cohort.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This multicenter retrospective study was approved by the institutional ethics committees of Beijing Tsinghua Changgung Hospital (approval 23694-4-01) and Beijing Friendship Hospital (approval 2023-P2-031-01). The requirement for additional informed consent was waived by the ethics committees because this study used retrospective, deidentified clinical data and involved no direct patient contact or intervention. All extracted records were deidentified before analysis, and direct identifiers such as names, medical record numbers, phone numbers, and addresses were removed. The analytic data were stored in a secure research environment accessible only to authorized study personnel. No participants received compensation because this was a retrospective secondary analysis. No identifiable patient information appears in the manuscript figures, tables, or supplementary materials.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics and Study Cohorts</title><p>A total of 331 hospitalized patients with pneumonia were included in this study divided into a development cohort (n=233, 70.4% from Beijing Tsinghua Changgung Hospital) and an external validation cohort (n=98, 29.6% from Beijing Friendship Hospital). Baseline demographic characteristics, hepatic or renal impairment, and antibiotic class distributions are shown in <xref ref-type="table" rid="table1">Table 1</xref>. Compared with the external validation cohort, the development cohort included older patients and a higher proportion of male patients and patients with hepatic or renal impairment. The internal test set was broadly comparable to the overall development cohort in demographic and clinical characteristics, particularly in age and the prevalence of hepatic or renal dysfunction, although some variation was observed in antibiotic class distributions. Quinolones were the most commonly used antibiotic class in the development cohort, internal test set, and external validation cohort.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline demographic and clinical characteristics of patients in the development set, internal test set, and external validation set<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable</td><td align="left" valign="bottom">Development set (n=233)</td><td align="left" valign="bottom">Internal test set (n=70)</td><td align="left" valign="bottom">External validation set (n=98)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Demographic characteristics</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gender (man), n/N (%)</td><td align="left" valign="top">148/233 (63.5)</td><td align="left" valign="top">38/70 (54.3)</td><td align="left" valign="top">39/98 (39.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (y), median (IQR)</td><td align="left" valign="top">68.00 (59.00-75.00)</td><td align="left" valign="top">67.50 (60.00-74.00)</td><td align="left" valign="top">62.50 (54.25-73.00)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Advanced age (&#x2265;65 y), n/N (%)</td><td align="left" valign="top">149/233 (63.9)</td><td align="left" valign="top">47/70 (67.1)</td><td align="left" valign="top">43/98 (43.9)</td></tr><tr><td align="left" valign="top" colspan="4">Medical history, n/N (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hepatic or renal dysfunction</td><td align="left" valign="top">90/233 (38.6)</td><td align="left" valign="top">27/70 (38.6)</td><td align="left" valign="top">22/98 (22.4)</td></tr><tr><td align="left" valign="top" colspan="4">Antibiotic categories, n/N (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Quinolones</td><td align="left" valign="top">147/368 (39.9)</td><td align="left" valign="top">52/98 (53.1)</td><td align="left" valign="top">79/110 (71.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x03B2;-lactamase inhibitor combinations</td><td align="left" valign="top">101/368 (27.4)</td><td align="left" valign="top">22/98 (22.4)</td><td align="left" valign="top">8/110 (7.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Antifungals</td><td align="left" valign="top">34/368 (9.2)</td><td align="left" valign="top">1/98 (1)</td><td align="left" valign="top">2/110 (1.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cephalosporins</td><td align="left" valign="top">26/368 (7.1)</td><td align="left" valign="top">2/98 (2)</td><td align="left" valign="top">9/110 (8.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Carbapenems</td><td align="left" valign="top">15/368 (4.1)</td><td align="left" valign="top">4/98 (4.1)</td><td align="left" valign="top">7/110 (6.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Antivirals</td><td align="left" valign="top">13/368 (3.5)</td><td align="left" valign="top">4/98 (4.1)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Aminoglycosides</td><td align="left" valign="top">8/368 (2.2)</td><td align="left" valign="top">7/98 (7.1)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tetracyclines</td><td align="left" valign="top">8/368 (2.2)</td><td align="left" valign="top">4/98 (4.1)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sulfonamides</td><td align="left" valign="top">5/368 (1.4)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Glycopeptides and polypeptides</td><td align="left" valign="top">4/368 (1.1)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">2/110 (1.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Macrolides</td><td align="left" valign="top">3/368 (0.8)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">3/110 (2.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Oxazolidinones</td><td align="left" valign="top">4/368 (1.1)</td><td align="left" valign="top">2/98 (2)</td><td align="left" valign="top">0 (0)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Percentages for antibiotic categories were calculated as the number of occurrences of each category divided by the total antibiotic category occurrences in each cohort (development cohort: n=368; internal test set: n=98; external validation cohort: n=110), reflecting prescription and category occurrence&#x2013;level statistics.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Performance for Antibiotic Selection and Joint Antibiotic Selection Plus Dosing Recommendation</title><p>We compared different LLMs and integration strategies across 2 tasks: antibiotic selection and joint antibiotic selection plus dosing recommendation (<xref ref-type="table" rid="table2">Table 2</xref>). On the internal test set, the full pipeline with DeepSeek-V3 achieved the best performance among all evaluated LLM-based methods, with an <italic>F</italic><sub>1</sub>-score of 0.8110 (95% CI 0.7371-0.8762) and a Jaccard accuracy of 0.7624 (95% CI 0.6810-0.8386) for antibiotic selection and an <italic>F</italic><sub>1</sub>-score of 0.7538 (95% CI 0.6671-0.8329) and a Jaccard accuracy of 0.7076 (95% CI 0.6145-0.7938) for joint antibiotic selection plus dosing recommendation. The full pipeline with GLM-4.6 and GPT-4o also outperformed their corresponding base models but remained inferior to the DeepSeek-V3&#x2013;based implementation.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Performance comparison between our model and the base large language models on the internal test set.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score(95% CI)</td><td align="left" valign="bottom">Jaccard accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Our model+DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top"><italic>0.8110 (0.7371-0.8762)</italic><sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top"><italic>0.7624 (0.6810-0.8386)</italic></td><td align="left" valign="top"><italic>0.8429 (0.7690-0.9095)</italic></td><td align="left" valign="top">0.8119 (0.7357-0.8833)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top"><italic>0.7538 (0.6671-0.8329)</italic></td><td align="left" valign="top"><italic>0.7076 (0.6145-0.7938)</italic></td><td align="left" valign="top"><italic>0.7833 (0.6952-0.8643)</italic></td><td align="left" valign="top">0.7524 (0.6619-0.8334)</td></tr><tr><td align="left" valign="top" colspan="5">Our model+GLM-4.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7101 (0.6286-0.7872)</td><td align="left" valign="top">0.6467 (0.5602-0.7321)</td><td align="left" valign="top">0.7619 (0.6761-0.8429)</td><td align="left" valign="top">0.7024 (0.6167-0.7857)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.6734 (0.5814-0.7619)</td><td align="left" valign="top">0.6217 (0.5264-0.7164)</td><td align="left" valign="top">0.7238 (0.6262-0.8167)</td><td align="left" valign="top">0.6667 (0.5714-0.7595)</td></tr><tr><td align="left" valign="top" colspan="5">Our model+GPT-4o</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7714 (0.7181-0.8243)</td><td align="left" valign="top">0.6790 (0.6143-0.7476)</td><td align="left" valign="top">0.7048 (0.6405-0.7714)</td><td align="left" valign="top"><italic>0.9143 (0.8595-0.9619)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.7257 (0.6662-0.7843)</td><td align="left" valign="top">0.6279 (0.5579-0.7017)</td><td align="left" valign="top">0.6571 (0.5929-0.7238)</td><td align="left" valign="top"><italic>0.8690 (0.8024-0.9286)</italic></td></tr><tr><td align="left" valign="top" colspan="5">DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.4913 (0.4046-0.5753)</td><td align="left" valign="top">0.4110 (0.3274-0.4952)</td><td align="left" valign="top">0.4774 (0.3893-0.5619)</td><td align="left" valign="top">0.5381 (0.4429-0.6333)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.4075 (0.3220-0.4952)</td><td align="left" valign="top">0.3336 (0.2550-0.4167)</td><td align="left" valign="top">0.3964 (0.3119-0.4833)</td><td align="left" valign="top">0.4476 (0.3500-0.5476)</td></tr><tr><td align="left" valign="top" colspan="5">GLM-4.6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.3793 (0.2941-0.4644)</td><td align="left" valign="top">0.3081 (0.2286-0.3883)</td><td align="left" valign="top">0.3667 (0.2821-0.4524)</td><td align="left" valign="top">0.4238 (0.3262-0.5190)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.3317 (0.2503-0.4174)</td><td align="left" valign="top">0.2652 (0.1926-0.3450)</td><td align="left" valign="top">0.3214 (0.2405-0.4071)</td><td align="left" valign="top">0.3714 (0.2786-0.4690)</td></tr><tr><td align="left" valign="top" colspan="5">GPT-4o</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.2939 (0.2265-0.3641)</td><td align="left" valign="top">0.2129 (0.1595-0.2712)</td><td align="left" valign="top">0.2679 (0.2048-0.3346)</td><td align="left" valign="top">0.3500 (0.2643-0.4381)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.2060 (0.1436-0.2735)</td><td align="left" valign="top">0.1469 (0.0997-0.2000)</td><td align="left" valign="top">0.1845 (0.1286-0.2440)</td><td align="left" valign="top">0.2500 (0.1714-0.3381)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Italicized values indicate the highest values.</p></fn></table-wrap-foot></table-wrap><p>On the external validation cohort, the full pipeline with DeepSeek-V3 maintained a strong performance (<xref ref-type="table" rid="table3">Table 3</xref>), achieving an <italic>F</italic><sub>1</sub> score of 0.8605 (95% CI 0.7891-0.9252) and a Jaccard accuracy of 0.8571 (95% CI 0.7857-0.9184) for antibiotic selection and an <italic>F</italic><sub>1</sub>-score of 0.8503 (95% CI 0.7789-0.9150) and a Jaccard accuracy of 0.8469 (95% CI 0.7755-0.9133) for joint antibiotic selection plus dosing recommendation. In contrast, the base DeepSeek-V3 model without retrieval augmentation or clinician-defined rule constraints showed marked performance degradation, with an <italic>F</italic><sub>1</sub> score of 0.4163 (95% CI 0.3289-0.5031) and a Jaccard accuracy of 0.3801 (95% CI 0.2951-0.4660) for antibiotic selection and an <italic>F</italic><sub>1</sub> score of 0.3129 (95% CI 0.2303-0.3952) and a Jaccard accuracy of 0.2832 (95% CI 0.2049-0.3622) for joint antibiotic selection plus dosing recommendation.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Performance comparison between our model and the base DeepSeek-V3 model on the external validation set.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Jaccard accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Our model+DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top"><italic>0.8605 (0.7891-0.9252)</italic><sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="top"><italic>0.8571 (0.7857-0.9184)</italic></td><td align="left" valign="top"><italic>0.8673 (0.7959-0.9286)</italic></td><td align="left" valign="top"><italic>0.8571 (0.7857-0.9184)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top"><italic>0.8503 (0.7789-0.9150)</italic></td><td align="left" valign="top"><italic>0.8469 (0.7755-0.9133)</italic></td><td align="left" valign="top"><italic>0.8571 (0.7857-0.9184)</italic></td><td align="left" valign="top"><italic>0.8469 (0.7755-0.9133)</italic></td></tr><tr><td align="left" valign="top" colspan="5">DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.4163 (0.3289-0.5031)</td><td align="left" valign="top">0.3801 (0.2951-0.4660)</td><td align="left" valign="top">0.3878 (0.3027-0.4728)</td><td align="left" valign="top">0.4728 (0.3741-0.5697)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.3129 (0.2303-0.3952)</td><td align="left" valign="top">0.2832 (0.2049-0.3622)</td><td align="left" valign="top">0.2874 (0.2092-0.3673)</td><td align="left" valign="top">0.3622 (0.2704-0.4541)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Italicized values indicate the highest values.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Ablation Study</title><p>To assess the contribution of individual modules, we performed an ablation study using the full pipeline as the reference configuration. The configurations of all ablation variants are summarized in <xref ref-type="table" rid="table4">Table 4</xref>, and the corresponding performance results are presented in <xref ref-type="table" rid="table5">Table 5</xref>. In the antibiotic selection task, the full model achieved an <italic>F</italic><sub>1</sub>-score of 0.8110. Removing clinician-defined rule constraints reduced the <italic>F</italic><sub>1</sub>-score to 0.7429, removing similar-case vector retrieval reduced it to 0.7156, and removing guideline-based knowledge graph retrieval reduced it to 0.7843. A similar pattern was observed for joint antibiotic selection plus dosing recommendation, where the full model achieved an <italic>F</italic><sub>1</sub>-score of 0.7538 compared with 0.6474 after removing rule constraints, 0.5946 after removing vector retrieval, and 0.7119 after removing guideline-based knowledge graph retrieval.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Definitions of ablation configurations.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Ablation configuration</td><td align="left" valign="bottom">Similar-case vector retrieval</td><td align="left" valign="bottom">Guideline-based knowledge graph retrieval</td><td align="left" valign="bottom">Clinician-defined rule constraints</td></tr></thead><tbody><tr><td align="left" valign="top">Our model</td><td align="left" valign="top">+<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">+</td><td align="left" valign="top">+</td></tr><tr><td align="left" valign="top">Our model &#x2013; GraphRAG<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+</td><td align="left" valign="top">&#x2013;<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">+</td></tr><tr><td align="left" valign="top">Our model &#x2013; vector</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">+</td><td align="left" valign="top">+</td></tr><tr><td align="left" valign="top">Our model &#x2013; rule</td><td align="left" valign="top">+</td><td align="left" valign="top">+</td><td align="left" valign="top">&#x2013;</td></tr><tr><td align="left" valign="top">Our model &#x2013; GraphRAG &#x2013; vector</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">+</td></tr><tr><td align="left" valign="top">Our model &#x2013; GraphRAG &#x2013; rule</td><td align="left" valign="top">+</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">&#x2013;</td></tr><tr><td align="left" valign="top">Our model &#x2013; vector &#x2013; rule</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">+</td><td align="left" valign="top">&#x2013;</td></tr><tr><td align="left" valign="top">DeepSeek-V3</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">&#x2013;</td><td align="left" valign="top">&#x2013;</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Module included.</p></fn><fn id="table4fn2"><p><sup>b</sup>GraphRAG: Graph Retrieval-Augmented Generation.</p></fn><fn id="table4fn3"><p><sup>c</sup>Module removed.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Ablation analysis of the contributions of individual modules to model performance.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Jaccard accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="5">Our model</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top"><italic>0.8110 (0.7371-0.8762)</italic><sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top"><italic>0.7624 (0.6810-0.8386)</italic></td><td align="left" valign="top">0.8429 (0.7690-0.9095)</td><td align="left" valign="top"><italic>0.8119 (0.7357-0.8833)</italic></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top"><italic>0.7538 (0.6671-0.8329)</italic></td><td align="left" valign="top"><italic>0.7076 (0.6145-0.7938)</italic></td><td align="left" valign="top"><italic>0.7833 (0.6952-0.8643)</italic></td><td align="left" valign="top"><italic>0.7524 (0.6619-0.8334)</italic></td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; GraphRAG<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7843 (0.7152-0.8476)</td><td align="left" valign="top">0.7217 (0.6417-0.7964)</td><td align="left" valign="top"><italic>0.8452 (0.7738-0.9095)</italic></td><td align="left" valign="top">0.7738 (0.6976-0.8452)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.7119 (0.6262-0.7929)</td><td align="left" valign="top">0.6538 (0.5633-0.7405)</td><td align="left" valign="top">0.7595 (0.6690-0.8429)</td><td align="left" valign="top">0.7071 (0.6167-0.7929)</td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; rule</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7429 (0.6748-0.8077)</td><td align="left" valign="top">0.6633 (0.5857-0.7405)</td><td align="left" valign="top">0.7476 (0.6714-0.8214)</td><td align="left" valign="top">0.7952 (0.7190-0.8643)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.6474 (0.5648-0.7262)</td><td align="left" valign="top">0.5683 (0.4836-0.6529)</td><td align="left" valign="top">0.6440 (0.5583-0.7286)</td><td align="left" valign="top">0.7000 (0.6095-0.7857)</td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; vector</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7156 (0.6427-0.7857)</td><td align="left" valign="top">0.6381 (0.5595-0.7155)</td><td align="left" valign="top">0.8060 (0.7262-0.8821)</td><td align="left" valign="top">0.6905 (0.6095-0.7690)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.5946 (0.5000-0.6850)</td><td align="left" valign="top">0.5310 (0.4381-0.6238)</td><td align="left" valign="top">0.6655 (0.5619-0.7643)</td><td align="left" valign="top">0.5714 (0.4762-0.6643)</td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; GraphRAG &#x2013; vector</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.7233 (0.6548-0.7890)</td><td align="left" valign="top">0.6390 (0.5602-0.7176)</td><td align="left" valign="top">0.8095 (0.7357-0.8786)</td><td align="left" valign="top">0.6929 (0.6190-0.7667)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.6233 (0.5362-0.7081)</td><td align="left" valign="top">0.5533 (0.4621-0.6445)</td><td align="left" valign="top">0.6810 (0.5857-0.7714)</td><td align="left" valign="top">0.6024 (0.5143-0.6881)</td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; GraphRAG &#x2013; rule</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.6253 (0.5458-0.7012)</td><td align="left" valign="top">0.5367 (0.4545-0.6212)</td><td align="left" valign="top">0.5986 (0.5190-0.6786)</td><td align="left" valign="top">0.7143 (0.6262-0.7976)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.5663 (0.4829-0.6486)</td><td align="left" valign="top">0.4843 (0.4000-0.5700)</td><td align="left" valign="top">0.5438 (0.4581-0.6310)</td><td align="left" valign="top">0.6452 (0.5500-0.7381)</td></tr><tr><td align="left" valign="top" colspan="5">Our model &#x2013; vector &#x2013; rule</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.6447 (0.5589-0.7267)</td><td align="left" valign="top">0.5717 (0.4843-0.6581)</td><td align="left" valign="top">0.6298 (0.5405-0.7167)</td><td align="left" valign="top">0.7048 (0.6119-0.7976)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.5641 (0.4770-0.6529)</td><td align="left" valign="top">0.4912 (0.4031-0.5814)</td><td align="left" valign="top">0.5417 (0.4536-0.6333)</td><td align="left" valign="top">0.6262 (0.5310-0.7214)</td></tr><tr><td align="left" valign="top" colspan="5">DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication</td><td align="left" valign="top">0.4913 (0.4046-0.5753)</td><td align="left" valign="top">0.4110 (0.3274-0.4952)</td><td align="left" valign="top">0.4774 (0.3893-0.5619)</td><td align="left" valign="top">0.5381 (0.4429-0.6333)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medication+dosing</td><td align="left" valign="top">0.4075 (0.3220-0.4952)</td><td align="left" valign="top">0.3336 (0.2550-0.4167)</td><td align="left" valign="top">0.3964 (0.3119-0.4833)</td><td align="left" valign="top">0.4476 (0.3500-0.5476)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Italicized values indicate the highest values.</p></fn><fn id="table5fn2"><p><sup>b</sup>GraphRAG: Graph Retrieval-Augmented Generation.</p></fn></table-wrap-foot></table-wrap><p>Performance declined further when multiple components were removed simultaneously. For example, the variant without guideline-based knowledge graph retrieval plus rule constraints achieved an <italic>F</italic><sub>1</sub>-score of 0.6253 (95% CI 0.5458-0.7012) for antibiotic selection and 0.5663 (95% CI 0.4829-0.6486) for joint antibiotic selection plus dosing recommendation, whereas the variant without vector retrieval plus rule constraints achieved <italic>F</italic><sub>1</sub>-scores of 0.6447 (95% CI 0.5589-0.7267) and 0.5641 (95% CI 0.4770-0.6529), respectively. The base model without retrieval augmentation or rule constraints showed the lowest overall performance.</p></sec><sec id="s3-4"><title>Subgroup Analyses by Hepatic or Renal Function and Age</title><p>We further evaluated model robustness for the joint antibiotic selection plus dosing recommendation task in clinically important subgroups defined by hepatic or renal function and age using the internal test set (n=70; <xref ref-type="fig" rid="figure2">Figure 2</xref>). In the hepatic or renal function subgroups, the full pipeline using DeepSeek-V3 achieved mean <italic>F</italic><sub>1</sub>-scores of 0.7062 (95% CI 0.5823-0.8198) in patients with impairment and 0.7939 (95% CI 0.6798-0.8991) in those without impairment. In the age-based subgroups, the corresponding mean <italic>F</italic><sub>1</sub>-scores were 0.7291 (95% CI 0.6177-0.8326) in patients aged 65 years or older and 0.8043 (95% CI 0.6884-0.9058) in those younger than 65 years. Overall, the integration strategy maintained relatively stable performance across all subgroups, whereas the base LLM approach consistently underperformed across the board.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Subgroup analysis of model performance for the joint antibiotic selection plus dosing recommendation task on the internal test set by (A) hepatic and renal function and (B) age. Error bars represent 95% CIs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e98207_fig02.png"/></fig></sec><sec id="s3-5"><title>Error Analysis and Expert Evaluation</title><p>To further assess recommendation safety and reasoning quality, respiratory physicians reviewed 45 error cases and a random sample of 50 generated recommendations. Error cases were categorized as safe but suboptimal or potentially unsafe. Of the 45 reviewed error cases, 43 (95.6%) were classified as safe but suboptimal recommendations, whereas only 2 (4.4%) were considered potentially unsafe. Both unsafe cases were related to insufficient consideration of hepatic or renal function during antibiotic selection or dose recommendation.</p><p>For reasoning quality evaluation, physicians assessed 50 cases across 4 dimensions: evidence relevance, rule applicability, reasoning consistency, and hallucination risk. For evidence relevance, 88% (n=44) of the cases received the highest score, and 12% (n=6) of the cases received a partial relevance score, with no cases rated as irrelevant. All reviewed cases received acceptable ratings for rule applicability, reasoning consistency, and hallucination risk, with no cases judged to contain incorrect rule application, inconsistent reasoning, or clinically significant hallucinations.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this multicenter retrospective study, we developed and validated a constrained LLM-based CDS pipeline for antibiotic selection and dose recommendation in hospitalized patients with pneumonia. The proposed framework, which integrates dual-branch retrieval, clinician-defined rule constraints, and hybrid-context reasoning, consistently outperformed the corresponding base LLMs across both the internal test set and the external validation cohort. It also maintained a relatively stable performance in clinically important high-risk subgroups, including older patients and patients with hepatic or renal impairment. Notably, the higher performance in the external validation cohort may be partly related to differences in patient characteristics and prescribing patterns between the 2 hospitals. The external validation cohort had a more concentrated antibiotic distribution and relatively simpler medication combinations, whereas the development cohort included a broader range of antibiotic categories and more complex regimens, which may have made the recommendation task more challenging. Together, these findings suggest that combining external evidence retrieval with explicit clinical constraints can improve the consistency and robustness and provide preliminary evidence of cross-site generalizability of LLM-assisted antimicrobial decision support.</p><p>The performance gains were attributable to the complementary contributions of multiple components rather than any single module. Ablation analyses showed consistent performance declines after removal of clinician-defined rule constraints, similar-case vector retrieval, or guideline-based knowledge graph retrieval, with larger reductions observed when multiple components were removed simultaneously (<xref ref-type="table" rid="table5">Table 5</xref>). These findings indicate that the 3 information sources served distinct but complementary functions: similar-case retrieval provided patient-aligned experience-based references; guideline-based knowledge graph retrieval contributed structured evidence-based knowledge; and clinician-defined rule constraints encoded explicit boundary conditions for dose adjustment, contraindications, and hepatic or renal function. In inpatient pneumonia management, where antibiotic decisions are both time sensitive and high risk, parametric LLM knowledge alone may be insufficient. By integrating patient-specific context, external evidence, and explicit clinical constraints within a unified reasoning framework, the proposed framework improved the consistency and safety of antibiotic recommendations. This design is also conceptually aligned with the early empirical decision-making process typically performed during infectious disease consultation while remaining intended as a clinician support tool rather than a replacement for specialist consultation or clinician judgment.</p></sec><sec id="s4-2"><title>Comparison to Prior Work</title><p>Compared with traditional black-box prediction models, the proposed framework emphasizes a more interpretable, traceable, and auditable form of CDS. The system is not intended to replace clinician judgment but to function as an assistive tool for antimicrobial stewardship and medication review in hospitalized patients with pneumonia. Its practical value lies in its ability to rapidly integrate admission narratives, key hepatic and renal function indicators, similar-case experience, and guideline-based evidence to generate candidate recommendations for antibiotic selection and dose recommendation. At the same time, the system provides traceable evidence and rule trigger information, which may facilitate rapid clinician verification and improve the transparency of the decision-making process. Given that the external validation cohort differed from the development cohort in age, sex distribution, hepatic or renal impairment, and antibiotic class distribution, whereas the full pipeline still maintained strong performance, the findings further support the potential applicability of this approach in heterogeneous real-world settings.</p><p>This study has several limitations. First, the sample size was relatively limited, and the data were derived from only 2 hospitals. In particular, the external validation cohort included only 98 patients, and the distribution of antibiotic categories was imbalanced. This limits the strength of conclusions regarding cross-site generalizability. Therefore, further validation in larger, multi-regional, multilevel, and more category-balanced health care settings is needed. Second, although the reference labels were adjudicated by experienced physicians, the source information included real-world antibiotic orders, which are influenced by clinician experience, local prescribing preferences, and institutional antimicrobial stewardship policies. Accordingly, these labels should not be interpreted as an absolute gold standard, and such label-level heterogeneity may affect model generalizability. In addition, the drug library in the prompt was sorted by common clinical use, which may have introduced positional or frequency-related bias; future evaluations should test alternative drug list orderings. Third, stable structured body weight and BMI information was not available in the retrospective dataset. Consequently, the framework could not fully evaluate individualized weight-based dosing strategies for antimicrobial agents that require body weight adjustment. Future studies should incorporate structured body weight information to support more precise dose recommendation. Fourth, antibiotic options, resistance patterns, and clinical guidelines continue to evolve over time, so the knowledge graph, rule base, and retrieval resources will require ongoing maintenance and updating. In addition, the guideline-based knowledge graph was constructed from publicly available guidelines; therefore, we cannot fully exclude the possibility that some guideline knowledge was already present in the evaluated LLMs&#x2019; parametric memory. Future work should further evaluate the framework using recently updated guidelines or institution-specific antimicrobial stewardship pathways to better isolate the contribution of retrieval augmentation. Fifth, there was a temporal mismatch between the development and external validation cohorts. Although patients with confirmed COVID-19 pneumonia were excluded, residual differences in pneumonia etiology and antimicrobial prescribing behavior across time may still affect cross-cohort comparability. Finally, this study was retrospective and did not include prospective evaluation in real clinical workflows. Future work should assess the framework in human-in-the-loop practice settings and further examine its effects on prescribing quality, antimicrobial stewardship, and clinical outcomes.</p></sec><sec id="s4-3"><title>Conclusions</title><p>In conclusion, we developed and externally validated a constrained LLM-based CDS pipeline for antibiotic selection and dose recommendation in hospitalized patients with pneumonia. By integrating dual-branch retrieval, clinician-defined rule constraints, and hybrid-context reasoning, the proposed framework improved the consistency and interpretability of LLM-assisted antimicrobial decision support and provided preliminary evidence of cross-site generalizability. The system also provided traceable evidence and rule trigger information to support clinician verification, highlighting its potential value for antimicrobial stewardship and medication review. Further prospective evaluation is needed to assess its performance and utility in real-world clinical workflows.</p></sec></sec></body><back><ack><p>During manuscript preparation and revision, the authors used generative AI to assist with code drafting and language editing. GPT-5.2 was not used to make clinical judgments, generate reference labels, interpret study findings, or draw scientific conclusions. All AI-generated code and text were reviewed, edited, and validated by the authors, who take full responsibility for the accuracy, integrity, and final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Beijing Natural Science Foundation (7242264, 26QY0484, and L246059) and Capital Medical University Basic Clinical Research Cultivation Program Project (JLPYPT2025001).</p></sec><sec><title>Data Availability</title><p>The data used in this study were obtained from the electronic health records of Beijing Tsinghua Changgung Hospital and Beijing Friendship Hospital. The datasets are not publicly available because they consist of deidentified patient-level clinical data and remain subject to institutional ethics requirements and patient privacy protections. Data may be made available from the corresponding author on reasonable request and subject to approval by the participating institutions.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: HL</p><p>Data curation: LL, XT, MJ</p><p>Investigation: LL, CT, XM, JL</p><p>Software: YZ</p><p>Supervision: YG, HL</p><p>Writing&#x2014;original draft: YZ, LL</p><p>Writing&#x2014;review and editing: YG, HL</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CDS</term><def><p>clinical decision support</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>F</given-names> </name><etal/></person-group><article-title>The mortality and years of life lost for community-acquired pneumonia before and during COVID-19 pandemic in China</article-title><source>Lancet Reg Health West Pac</source><year>2023</year><volume>42</volume><fpage>100968</fpage><pub-id pub-id-type="doi">10.1016/j.lanwpc.2023.100968</pub-id><pub-id pub-id-type="medline">38022712</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>The Lancet</collab></person-group><article-title>Antimicrobial resistance: an agenda for all</article-title><source>Lancet</source><year>2024</year><month>06</month><day>1</day><volume>403</volume><issue>10442</issue><fpage>2349</fpage><pub-id pub-id-type="doi">10.1016/S0140-6736(24)01076-6</pub-id><pub-id pub-id-type="medline">38797177</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2021 Antimicrobial Resistance Collaborators</collab></person-group><article-title>Global burden of bacterial antimicrobial resistance 1990-2021: a systematic analysis with forecasts to 2050</article-title><source>Lancet</source><year>2024</year><month>09</month><day>28</day><volume>404</volume><issue>10459</issue><fpage>1199</fpage><lpage>1226</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(24)01867-1</pub-id><pub-id pub-id-type="medline">39299261</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shortliffe</surname><given-names>EH</given-names> </name></person-group><article-title>Mycin: a knowledge-based computer program applied to infectious diseases</article-title><source>Proc Annu Symp Comput Appl Med Care</source><year>1977</year><access-date>2026-07-22</access-date><volume>5</volume><fpage>66</fpage><lpage>69</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://pmc.ncbi.nlm.nih.gov/articles/PMC2464549/">https://pmc.ncbi.nlm.nih.gov/articles/PMC2464549/</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miller</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Pople</surname><given-names>HE</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Myers</surname><given-names>JD</given-names> </name></person-group><article-title>Internist-1, an experimental computer-based diagnostic consultant for general internal medicine</article-title><source>N Engl J Med</source><year>1982</year><month>08</month><day>19</day><volume>307</volume><issue>8</issue><fpage>468</fpage><lpage>476</lpage><pub-id pub-id-type="doi">10.1056/NEJM198208193070803</pub-id><pub-id pub-id-type="medline">7048091</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Stenner</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Doan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>KB</given-names> </name><name name-style="western"><surname>Waitman</surname><given-names>LR</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>JC</given-names> </name></person-group><article-title>MedEx: a medication information extraction system for clinical narratives</article-title><source>J Am Med Inform Assoc</source><year>2010</year><volume>17</volume><issue>1</issue><fpage>19</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1197/jamia.M3378</pub-id><pub-id pub-id-type="medline">20064797</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savova</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Masanz</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Ogren</surname><given-names>PV</given-names> </name><etal/></person-group><article-title>Mayo clinical Text Analysis and Knowledge Extraction System (cTAKES): architecture, component evaluation and applications</article-title><source>J Am Med Inform Assoc</source><year>2010</year><volume>17</volume><issue>5</issue><fpage>507</fpage><lpage>513</lpage><pub-id pub-id-type="doi">10.1136/jamia.2009.001560</pub-id><pub-id pub-id-type="medline">20819853</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Oren</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Scalable and accurate deep learning with electronic health records</article-title><source>NPJ Digit Med</source><year>2018</year><volume>1</volume><fpage>18</fpage><pub-id pub-id-type="doi">10.1038/s41746-018-0029-1</pub-id><pub-id pub-id-type="medline">31304302</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mienye</surname><given-names>ID</given-names> </name><name name-style="western"><surname>Swart</surname><given-names>TG</given-names> </name><name name-style="western"><surname>Obaido</surname><given-names>G</given-names> </name></person-group><article-title>Recurrent neural networks: a comprehensive review of architectures, variants, and applications</article-title><source>Information</source><year>2024</year><volume>15</volume><issue>9</issue><fpage>517</fpage><pub-id pub-id-type="doi">10.3390/info15090517</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name></person-group><article-title>Opportunities and challenges in developing deep learning models using electronic health records data: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>10</month><day>1</day><volume>25</volume><issue>10</issue><fpage>1419</fpage><lpage>1428</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocy068</pub-id><pub-id pub-id-type="medline">29893864</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><etal/></person-group><article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title><source>Bioinformatics</source><year>2020</year><month>02</month><day>15</day><volume>36</volume><issue>4</issue><fpage>1234</fpage><lpage>1240</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id><pub-id pub-id-type="medline">31501885</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Altosaar</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ranganath</surname><given-names>R</given-names> </name></person-group><article-title>ClinicalBERT: modeling clinical notes and predicting hospital readmission</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 10, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1904.05342</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rasmy</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhi</surname><given-names>D</given-names> </name></person-group><article-title>Med-BERT: pretrained contextualized embeddings on large-scale structured electronic health records for disease prediction</article-title><source>NPJ Digit Med</source><year>2021</year><month>05</month><day>20</day><volume>4</volume><issue>1</issue><fpage>86</fpage><pub-id pub-id-type="doi">10.1038/s41746-021-00455-y</pub-id><pub-id pub-id-type="medline">34017034</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Datta</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Deep learning in clinical natural language processing: a methodical review</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>03</month><day>1</day><volume>27</volume><issue>3</issue><fpage>457</fpage><lpage>470</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocz200</pub-id><pub-id pub-id-type="medline">31794016</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>London</surname><given-names>AJ</given-names> </name></person-group><article-title>Artificial intelligence and black-box medical decisions: accuracy versus explainability</article-title><source>Hastings Cent Rep</source><year>2019</year><month>01</month><volume>49</volume><issue>1</issue><fpage>15</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1002/hast.973</pub-id><pub-id pub-id-type="medline">30790315</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Corny</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rajkumar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>O</given-names> </name><etal/></person-group><article-title>A machine learning-based clinical decision support system to identify prescriptions with a high risk of medication error</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>11</month><day>1</day><volume>27</volume><issue>11</issue><fpage>1688</fpage><lpage>1694</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa154</pub-id><pub-id pub-id-type="medline">32984901</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name></person-group><article-title>GAMENet: Graph Augmented MEmory Networks for recommending medication combination</article-title><source>Proc AAAI Conf Artif Intell</source><year>2019</year><volume>33</volume><issue>1</issue><fpage>1126</fpage><lpage>1133</lpage><pub-id pub-id-type="doi">10.1609/aaai.v33i01.33011126</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DS</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Reasoning-driven large language models in medicine: opportunities, challenges, and the road ahead</article-title><source>Lancet Digit Health</source><year>2026</year><month>01</month><volume>8</volume><issue>1</issue><fpage>100931</fpage><pub-id pub-id-type="doi">10.1016/j.landig.2025.100931</pub-id><pub-id pub-id-type="medline">41620322</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Derby</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Large language models with temporal reasoning for longitudinal clinical summarization and prediction</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 30, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.18724</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Boll</surname><given-names>HO</given-names> </name><name name-style="western"><surname>Boll</surname><given-names>AO</given-names> </name><name name-style="western"><surname>Boll</surname><given-names>LP</given-names> </name><name name-style="western"><surname>Hanna</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Calixto</surname><given-names>I</given-names> </name></person-group><article-title>DistillNote: LLM-based clinical note summaries improve heart failure diagnosis</article-title><source>arXiv</source><access-date>2026-07-22</access-date><comment>Preprint posted online on  Jun 20, 2025</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/html/2506.16777v1">https://arxiv.org/html/2506.16777v1</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ntinopoulos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Rodriguez Cetina Biefer</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tudorache</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Large language models for data extraction from unstructured and semi-structured electronic health records: a multiple model performance evaluation</article-title><source>BMJ Health Care Inform</source><year>2025</year><month>01</month><day>19</day><volume>32</volume><issue>1</issue><fpage>e101139</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2024-101139</pub-id><pub-id pub-id-type="medline">39832824</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>De Vito</surname><given-names>A</given-names> </name><name name-style="western"><surname>Geremia</surname><given-names>N</given-names> </name><name name-style="western"><surname>Bavaro</surname><given-names>DF</given-names> </name><etal/></person-group><article-title>Comparing large language models for antibiotic prescribing in different clinical scenarios: which performs better?</article-title><source>Clin Microbiol Infect</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>1336</fpage><lpage>1342</lpage><pub-id pub-id-type="doi">10.1016/j.cmi.2025.03.002</pub-id><pub-id pub-id-type="medline">40113208</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Large language model distilling medication recommendation model</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 5, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2402.02803</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name></person-group><article-title>Large language models for generative recommendation: a survey and visionary discussions</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 3, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2309.01157</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chase</surname><given-names>A</given-names> </name><name name-style="western"><surname>Most</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Large language models management of complex medication regimens: a case-based evaluation</article-title><source>Front Pharmacol</source><year>2025</year><volume>16</volume><fpage>1514445</fpage><pub-id pub-id-type="doi">10.3389/fphar.2025.1514445</pub-id><pub-id pub-id-type="medline">41368579</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>A</given-names> </name></person-group><article-title>Improving large language model applications in biomedicine with retrieval-augmented generation: a systematic review, meta-analysis, and clinical development guidelines</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>04</month><day>1</day><volume>32</volume><issue>4</issue><fpage>605</fpage><lpage>615</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf008</pub-id><pub-id pub-id-type="medline">39812777</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name></person-group><article-title>Improving large language model applications in the medical and nursing domains with retrieval-augmented generation: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>21</day><volume>27</volume><fpage>e80557</fpage><pub-id pub-id-type="doi">10.2196/80557</pub-id><pub-id pub-id-type="medline">41118646</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>TK</given-names> </name></person-group><article-title>Multi-evidence clinical reasoning with retrieval-augmented generation for emergency triage: retrospective evaluation study</article-title><source>JMIR Med Inform</source><year>2026</year><month>01</month><day>26</day><volume>14</volume><fpage>e82026</fpage><pub-id pub-id-type="doi">10.2196/82026</pub-id><pub-id pub-id-type="medline">41587455</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Normand</surname><given-names>O</given-names> </name><name name-style="western"><surname>Borsi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Fruin</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A real-world evaluation of LLM medication safety reviews in NHS primary care</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 24, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.21127</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qu</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>B</given-names> </name></person-group><article-title>Guidelines for the diagnosis and treatment of adult community acquired pneumonia in China (2016 Edition) [Article in Chinese]</article-title><source>Zhonghua Jie He He Hu Xi Za Zhi</source><year>2016</year><month>04</month><day>12</day><volume>39</volume><issue>4</issue><fpage>241</fpage><lpage>242</lpage><pub-id pub-id-type="doi">10.3760/cma.j.issn.1001-0939.2016.04.001</pub-id><pub-id pub-id-type="medline">27117069</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt template, results, and technical implementation details.</p><media xlink:href="medinform_v14i1e98207_app1.docx" xlink:title="DOCX File, 34 KB"/></supplementary-material></app-group></back></article>