<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMI</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id>
      <journal-title>JMIR Medical Informatics</journal-title>
      <issn pub-type="epub">2291-9694</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v14i1e97520</article-id>
      <article-id pub-id-type="pmid">42735403</article-id>
      <article-id pub-id-type="doi">10.2196/97520</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Locally Deployed Large Language Models for AI-Assisted Outpatient Prescription Review: Crossover Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Thawinwisan</surname>
            <given-names>Nattawipa</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Kadmon</surname>
            <given-names>G</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Liu</surname>
            <given-names>Zhengyue</given-names>
          </name>
          <degrees>BM</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-4029-7799</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Ding</surname>
            <given-names>Yi</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-5448-5435</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Chen</surname>
            <given-names>Jingxia</given-names>
          </name>
          <degrees>BM</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0000-8991-9301</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Yan</surname>
            <given-names>Ziqiang</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0003-6182-5399</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Cheng</surname>
            <given-names>Xuxi</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0007-0124-6159</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Zhou</surname>
            <given-names>Wei</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-8036-1037</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Fu</surname>
            <given-names>Peng</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-2487-5246</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Wang</surname>
            <given-names>Zhuo</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <address>
            <institution>Department of Pharmacy, Shanghai Changhai Hospital, The First Affiliated Hospital of Navy Medical University</institution>
            <addr-line>168 Changhai Road, Yangpu District</addr-line>
            <addr-line>Shanghai, 200433</addr-line>
            <country>China</country>
            <phone>86 021 31162307</phone>
            <email>wztgyx223@163.com</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2918-8732</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>School of Clinical Pharmacy, Shenyang pharmaceutical university</institution>
        <addr-line>Shenyang</addr-line>
        <country>China</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Pharmacy, Shanghai Changhai Hospital, The First Affiliated Hospital of Navy Medical University</institution>
        <addr-line>Shanghai</addr-line>
        <country>China</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Zhuo Wang <email>wztgyx223@163.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>14</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>14</volume>
      <elocation-id>e97520</elocation-id>
      <history>
        <date date-type="received">
          <day>7</day>
          <month>4</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>11</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>22</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>25</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Zhengyue Liu, Yi Ding, Jingxia Chen, Ziqiang Yan, Xuxi Cheng, Wei Zhou, Peng Fu, Zhuo Wang. Originally published in JMIR Medical Informatics (https://medinform.jmir.org), 14.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on https://medinform.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://medinform.jmir.org/2026/1/e97520" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Pharmacists’ prescription review is key to medication safety, but rising numbers of outpatient prescriptions and expanding formularies leave less time per case, increasing error risk. Large language models (LLMs) show promise, yet two barriers hinder routine use: first, most systems rely on cloud-based commercial models, risking data breaches by transmitting protected patient information externally, and second, retrieval-augmented generation (RAG) pipelines used to reduce hallucinations depend on complex text vectorization and vector databases, which hospitals with limited IT resources struggle to build and maintain.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to evaluate the feasibility and usefulness of a locally deployed, knowledge-augmented LLM as a decision-support tool for pharmacist-led outpatient prescription review.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>The open-source Qwen3-14B model was deployed on a hospital intranet server using the Ollama framework. A structured knowledge base derived from drug package inserts served as the principal reference and supported lightweight knowledge augmentation through exact-match injection. A 2-period crossover design was used: 2 pharmacists independently reviewed 213 outpatient prescriptions under both unaided and AI-assisted conditions, yielding paired unaided and collaborative results for each prescription. Plain AI review and knowledge-augmented AI review were also run on the same 213-prescription test set to quantify hallucination rates, and stand-alone AI performance was assessed against the reference standard. The reference standard was established by independent consensus between two supervising pharmacists, with disagreements adjudicated by a deputy chief pharmacist. Accuracy, sensitivity, and specificity were compared between conditions using paired McNemar tests, and review time was compared using the Wilcoxon signed-rank test.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Overall review accuracy was 97.2% (207/213, 95% CI 94.0%-98.7%) in the human-AI collaborative condition vs 82.6% (176/213, 95% CI 77.0%-87.1%) in the pharmacist-alone condition (<italic>P</italic>&#60;.001). Sensitivity was 98% (61/62) vs 55% (34/62), and the false-negative rate fell from 45% to 2% (<italic>P</italic>&#60;.001). Specificity did not differ significantly (146/151, 96.7% vs 142/151, 94%; <italic>P</italic>=.29). Knowledge augmentation reduced the model hallucination rate from 19.7% (42/213 prescriptions) to 4.7% (10/213), an absolute reduction of 15.0 percentage points (relative reduction 76.2%; <italic>P</italic>&#60;.001). Collaborative review reduced the mean per-prescription review time from 2.33 (SD 0.97) to 1.12 (SD 0.49) minutes (approximately 51.9% reduction; Wilcoxon signed-rank <italic>Z</italic>=−12.65; <italic>P</italic>&#60;.001; <italic>r</italic>=0.87).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>A locally deployed, knowledge-augmented LLM used as a pharmacist-supervised prescreening tool was associated with substantially higher accuracy and sensitivity in outpatient retrospective prescription review, while keeping all prescription data within the hospital network. Locally deployed open-source models may offer hospitals a privacy-preserving and practical route to AI-assisted pharmacy decision support.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language model</kwd>
        <kwd>prescription review</kwd>
        <kwd>human–artificial intelligence collaboration</kwd>
        <kwd>human-AI collaboration</kwd>
        <kwd>on-premise deployment</kwd>
        <kwd>knowledge augmentation</kwd>
        <kwd>Ollama</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Prescription review is fundamental to ensuring medication safety and rational drug use in health care institutions. China’s Hospital Prescription Review Management Standards (Trial) designates pharmacists as the primary agents of prescription review across all levels of health care facilities [<xref ref-type="bibr" rid="ref1">1</xref>]. However, as outpatient prescription volumes and drug formularies steadily expand, pharmacists face growing pressure to evaluate indication appropriateness, dosing, drug interactions, and other factors within limited time, inevitably raising the risk of oversight [<xref ref-type="bibr" rid="ref2">2</xref>]. Finding effective supportive tools to enhance review quality and efficiency while preserving the pharmacist’s central role has therefore become an urgent priority.</p>
      <p>Since 2023, large language models (LLMs) have gained considerable traction in health care. Studies have demonstrated that LLMs can effectively encode clinical information and perform comparably to experts on medical question-answering tasks [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. In the domain of medication safety, researchers have explored the capacity of LLMs to correct medication direction errors [<xref ref-type="bibr" rid="ref5">5</xref>], identify drug-drug interactions [<xref ref-type="bibr" rid="ref6">6</xref>], and review prescriptions [<xref ref-type="bibr" rid="ref7">7</xref>]. Despite this progress, LLMs still fall short of matching pharmacist-level expertise. In a prospective crossover study, Ong et al [<xref ref-type="bibr" rid="ref8">8</xref>] showed that combining LLM-based screening with physician review achieved significantly higher medication error detection rates than either method alone. LLMs thus appear better suited as a complement to clinical practice rather than a replacement.</p>
      <p>Despite these advances, two critical gaps remain. First, nearly all existing studies rely on cloud-based platforms. Prescription data inherently contain legally protected patient information (eg, diagnoses and medication histories) that cannot be transmitted through external interfaces without substantial data breach and regulatory risk [<xref ref-type="bibr" rid="ref9">9</xref>]. Second, the hallucination problem intrinsic to LLMs poses particular dangers in clinical settings [<xref ref-type="bibr" rid="ref10">10</xref>]. Retrieval-augmented generation (RAG) is the prevailing strategy for mitigating hallucinations [<xref ref-type="bibr" rid="ref11">11</xref>]; however, conventional RAG pipelines require text vectorization, embedding models, and vector databases, creating a significant technical barrier for health care facilities with limited informatics resources.</p>
      <p>To address these gaps, we deployed the open-source Qwen3-14B model on a hospital intranet server via the Ollama framework and constructed a structured knowledge base from drug package inserts. Instead of the vectorization and semantic retrieval overhead of conventional RAG, we adopted exact-match injection for knowledge augmentation. Importantly, this study was not designed to evaluate stand-alone AI performance. We used a crossover design grounded in routine pharmacy workflows, comparing unaided pharmacist review with AI-assisted pharmacist review and systematically assessing the impact of knowledge augmentation on model hallucinations. This approach preserves regulatory compliance by maintaining pharmacists as the reviewers and reflects realistic implementation conditions. The aim of this study was to determine whether a locally deployed, knowledge-augmented LLM, serving as a pharmacist-supervised prescreening tool, improves accuracy, sensitivity, and efficiency of outpatient prescription review compared with unaided review, and to quantify the degree to which exact-match knowledge augmentation reduces hallucinations. We hypothesized that AI-assisted review would improve accuracy and sensitivity without meaningful loss of specificity, and that knowledge augmentation would substantially lower the hallucination rate relative to unaugmented AI review.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>System Architecture and Knowledge Augmentation</title>
        <p>The system adopts a browser or server architecture with a PHP (version 8.2.27; PHP Group) back end communicating with a locally deployed Ollama (version 0.10.1; Ollama Inc) instance serving Qwen3-14B (Alibaba Group; Q4_K_M quantization, greedy decoding, up to 5120 tokens per prescription) on the hospital intranet. All processing remained fully offline. The end-to-end pipeline and its comparison with conventional RAG are illustrated in <xref rid="figure1" ref-type="fig">Figures 1</xref> and <xref rid="figure2" ref-type="fig">2</xref>. Implementation details, including hardware specifications, are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>End-to-end knowledge-augmented review pipeline. The offline drug knowledge base feeds matched insert fields into prompt assembly; the local Qwen3-14B model performs inference and returns a structured advisory output. HIS: hospital information system.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e97520_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Deterministic exact matching vs conventional retrieval-augmented generation (RAG). Exact matching is traceable, needs no retraining, and removes the semantic mismatch risk of vector retrieval, although it still depends on accurate drug code correspondence; conventional RAG adds infrastructure and carries a risk of semantic mismatch.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e97520_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Drug knowledge was sourced exclusively from package inserts, which serve as the legally binding reference under Chinese regulations, decomposed into 5 standardized fields (indications, dosage and administration, contraindications, interactions, and special-population precautions) across 583 entries. Upon recognizing a drug code, the system performed deterministic exact matching and injected the corresponding fields into the prompt. This approach avoids the need for an embedding model, vector store, and semantic mismatch risks inherent to conventional RAG; provenance is fully traceable, and updates require only editing a single database record. Because matching is keyed on the drug code rather than on free text, it removes the semantic retrieval errors of vector search, although it still depends on consistent coding; a brand-vs-generic naming difference or a coding inconsistency in the hospital information system could therefore prevent a match. Such cases are handled by the same safeguard applied to unmatched drugs: they are treated as uninjected and flagged for manual review, so a failed match surfaces as an explicit no-injection alert rather than a silent or erroneous injection. Drugs absent from the knowledge base were flagged for manual pharmacist review; all 213 test prescriptions were fully covered. We chose package inserts over subscription compendia such as Micromedex (Merative LP) or UpToDate. Although these compendia offer richer indication and dosage information, they require licensing and external connectivity, which is incompatible with the fully offline, privacy-preserving design at the core of our study.</p>
        <p>The prompt comprised 4 components: role definition (advisory tool, not a decision-maker), task specification (9 review dimensions, including indication-diagnosis correlation, dosage, interactions, and special populations), injected package-insert content and review rules, and a structured output format (problem description, supporting evidence, and recommended action). The complete template is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
      </sec>
      <sec>
        <title>Study Design</title>
        <sec>
          <title>Population and Allocation</title>
          <p>The population of the study was outpatient prescriptions from a tertiary teaching hospital in October and November 2025. Inclusion criteria were prescriptions with a minimum of 1 chemical or biological pharmaceutical drug. Exclusion criteria were traditional Chinese herbal decoction prescriptions. Following the screening process, 213 prescriptions were randomly sampled at a rate of 0.1%, as specified by the Hospital Prescription Review Management Standards [<xref ref-type="bibr" rid="ref1">1</xref>].</p>
          <p>The sample size was fixed by this administrative sampling rule, and no a priori power calculation was performed before the study. We did not conduct a post hoc power analysis, because observed power is a deterministic function of the observed <italic>P</italic> value and adds little information beyond the CIs. Instead, the precision of every estimate is conveyed by its Wilson-score 95% CI, and the fixed sample size is acknowledged as a limitation. As a planning reference only, the sample size formula for the McNemar test described by Lachin [<xref ref-type="bibr" rid="ref12">12</xref>] indicates that, for a 2-sided α of .05, 80% power, and a between-condition accuracy difference of at least 10 percentage points, about 124 discordant-informative pairs would be required; the 213 paired observations available therefore provided a reasonable basis for the primary comparison. The therapeutic duplication (n=5, 2.3%) and dosage inappropriateness (n=11, 5.2%) subgroups contained too few cases for adequately powered testing and are reported as exploratory.</p>
        </sec>
        <sec>
          <title>Crossover Design and Collaborative Workflow</title>
          <p>A 2-period crossover design was used (<xref rid="figure3" ref-type="fig">Figure 3</xref>). The 213 prescriptions were randomly allocated to batch A (n=107, 50.2%) and batch B (n=106, 49.8%). Two junior pharmacists (P1 and P2; 1-2 years of experience) each performed 1 period of unaided review and 1 period of human-AI collaborative review, counterbalanced across batches with a 2-week washout interval, yielding 213 paired observations (1 unaided and 1 collaborative result per prescription).</p>
          <fig id="figure3" position="float">
            <label>Figure 3</label>
            <caption>
              <p>Two-period crossover design. Prescriptions were randomly split into 2 batches and counterbalanced across pharmacists and periods, separated by a 2-week washout.</p>
            </caption>
            <graphic xlink:href="medinform_v14i1e97520_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          </fig>
          <p>In the collaborative condition (<xref rid="figure4" ref-type="fig">Figure 4</xref>), the pharmacist reviewed system-generated alerts, each linked to package-insert evidence, and exercised sole decision-making authority to accept, revise, or dismiss each flag or to add findings not identified by the system. Every alert disposition was recorded as accepted or overridden according to the reviewing pharmacist’s judgment. Overridden alerts were subsequently verified against the reference standard and classified as correct or incorrect. The system automatically recorded the review time of each prescription as the wall-clock interval from opening to submission, capturing end-to-end time at the prescription level rather than the drug-item level. Model inference for every prescription in a batch was completed beforehand as an offline preprocessing step, so that pharmacists reviewed previously generated outputs; the recorded review time therefore reflects the pharmacist’s interaction with the system and excludes model inference latency. Per-prescription review time served as the primary efficiency outcome.</p>
          <fig id="figure4" position="float">
            <label>Figure 4</label>
            <caption>
              <p>Human-AI collaborative workflow. The pharmacist reviewed each system alert, accepting or overriding it, and added any issues not flagged by the system; overridden alerts were compared with the reference standard.</p>
            </caption>
            <graphic xlink:href="medinform_v14i1e97520_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
          </fig>
        </sec>
        <sec>
          <title>Comparison of AI Review Modes</title>
          <p>All 213 prescriptions were additionally processed in 2 stand-alone AI modes, plain (parametric knowledge only) and knowledge-augmented, to quantify the effect of knowledge injection on hallucinations. Hallucination was defined as any assertion contradicting the package insert or established clinical guidance, adjudicated at the prescription level by 2 senior pharmacists (disagreements resolved by a deputy chief pharmacist); Cohen κ quantified interrater reliability. The knowledge-augmented AI’s stand-alone diagnostic performance was also evaluated against the reference standard.</p>
        </sec>
        <sec>
          <title>Reference Standard</title>
          <p>Two supervising pharmacists (&#62;5 years of prescription-review experience), blinded to all experimental outputs, independently assessed all 213 prescriptions; discrepancies were resolved by a deputy chief pharmacist. A sensitivity analysis included all prescriptions flagged by any condition but classified as appropriate by the expert panel for post hoc review (with blinding lifted); verified inappropriate prescriptions were added to an extended reference standard, and all analyses were repeated.</p>
        </sec>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>Baseline characteristics of the 213 prescriptions were summarized overall and by crossover sequence (P1 and P2), and between-sequence balance was tested with the Mann-Whitney <italic>U</italic> test for age and drug items per prescription (both nonnormal by the Shapiro-Wilk test) and the Pearson chi-square test for sex and age category.</p>
        <p>The primary outcome was overall accuracy (correct classifications divided by total prescriptions); secondary outcomes included sensitivity, specificity, false-positive rate, false-negative rate, and mean review time. Paired conditions were compared with the McNemar test (Yates continuity correction; exact binomial McNemar test for small subgroups). A Mann-Whitney <italic>U</italic> test (with a <italic>P</italic>&#62;.10 threshold) verified batch exchangeability before pooling periods. All proportions are reported with Wilson 95% CIs. No multiplicity correction was applied to exploratory end points. Analyses were performed in Python (version 3.10.2; Python Software Foundation). Additional statistical analysis details are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This retrospective, noninterventional study analyzed existing, deidentified outpatient prescription records, with no contact with patients and no influence on clinical care. It therefore qualified for exemption from ethics committee review and individual informed consent under Article 32 of China’s 2023 Measures for Ethical Review of Life Science and Medical Research Involving Humans, and the institutional research-governance office confirmed this exemption (no case number applies, as no review was sought) [<xref ref-type="bibr" rid="ref13">13</xref>]. Direct identifiers were removed during extraction, and only the fields required for analysis (patient age, sex, diagnosis, drug, dose, route, and frequency) were retained, yielding a dataset from which no individual can be identified. All processing occurred within the hospital intranet on access-controlled servers. No participants were recruited or compensated, and no image or supplementary file contains patient-identifiable information.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Baseline Prescription Characteristics</title>
        <p>The 213 prescriptions came from patients aged 1 to 95 years (median 57, IQR 38-69); 110 (51.6%) were female, and the median number of drug items per prescription was 2 (IQR 1-3). Sequences P1 (n=107, 50.2%) and P2 (n=106, 49.8%) were well balanced with respect to age (<italic>P</italic>=.86), age category (<italic>P</italic>=.33), sex (<italic>P</italic>=.73), and drug items per prescription (<italic>P</italic>=.33), supporting pooling across periods (<xref ref-type="table" rid="table1">Table 1</xref>).</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Demographic and clinical characteristics of the 213 outpatient prescriptions sampled at a tertiary teaching hospital in China (October 2025-November 2025) for a 2-period crossover study comparing unaided pharmacist review with human-AI collaborative review, overall and by crossover sequence.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="270"/>
            <col width="0"/>
            <col width="180"/>
            <col width="0"/>
            <col width="220"/>
            <col width="0"/>
            <col width="210"/>
            <col width="0"/>
            <col width="0"/>
            <col width="90"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Characteristics</td>
                <td colspan="2">Overall (N=213)</td>
                <td colspan="2">Sequence P1 (n=107)</td>
                <td colspan="2">Sequence P2 (n=106)</td>
                <td colspan="2"><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="10">
                  <bold>Age (years)</bold>
                </td>
                <td>.86<sup>a</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Mean (SD)</td>
                <td colspan="2">53.5 (19.9)</td>
                <td colspan="2">53.0 (21.3)</td>
                <td colspan="2">54.1 (18.5)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Median (IQR)</td>
                <td colspan="2">57 (38-69)</td>
                <td colspan="2">57 (36-70)</td>
                <td colspan="2">56 (40-68)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Range</td>
                <td colspan="2">1-95</td>
                <td colspan="2">1-93</td>
                <td colspan="2">13-95</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Age group (years), n (%)</bold>
                </td>
                <td>.33<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Children (&#60;18)</td>
                <td colspan="2">12 (5.6)</td>
                <td colspan="2">8 (7.5)</td>
                <td colspan="2">4 (3.8)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Adults (18-59)</td>
                <td colspan="2">106 (49.8)</td>
                <td colspan="2">49 (45.8)</td>
                <td colspan="2">57 (53.8)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Older adults (≥60)</td>
                <td colspan="2">95 (44.6)</td>
                <td colspan="2">50 (46.7)</td>
                <td colspan="2">45 (42.4)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Sex, n (%)</bold>
                </td>
                <td>.73<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Male</td>
                <td colspan="2">103 (48.4)</td>
                <td colspan="2">50 (46.7)</td>
                <td colspan="2">53 (50)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Female</td>
                <td colspan="2">110 (51.6)</td>
                <td colspan="2">57 (53.3)</td>
                <td colspan="2">53 (50)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Drug items per prescription</bold>
                </td>
                <td>.33<sup>a</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Mean (SD)</td>
                <td colspan="2">2.16 (1.04)</td>
                <td colspan="2">2.07 (0.96)</td>
                <td colspan="2">2.25 (1.11)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Median (IQR)</td>
                <td colspan="2">2 (1-3)</td>
                <td colspan="2">2 (1-3)</td>
                <td colspan="2">2 (1-3)</td>
                <td colspan="3">
                  <break/>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>Mann-Whitney <italic>U</italic> test; age and drug items per prescription were nonnormally distributed (Shapiro-Wilk <italic>P</italic>&#60;.05 in ≥1 group).</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>Pearson chi-square test.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Comparison of Review Performance Between Groups</title>
        <p>Before pooling the two periods, we examined whether the two prescription batches were comparable across sequences; given the prescription-level paired design, this is a check of batch and sequence comparability rather than a formal person-level carryover test. The Mann-Whitney <italic>U</italic> test, using the per-prescription sum of the two-period outcomes as the test statistic, showed no significant difference between sequences (batch A in period 1: n=107, 50.2%, vs batch B in period 2: n=106, 49.8%; <italic>P</italic>=.37). Because the paired unit was the prescription and the sequences corresponded to the 2 prescription batches, this result indicates that the batches were exchangeable with no detectable sequence-level effect; the 2 periods were therefore pooled for the main analysis.</p>
        <p>Independent expert review and adjudication classified 62 (29.1%) of 213 prescriptions as inappropriate: 51 (24.0%) with an inappropriate indication, 11 (5.2%) with an inappropriate dosage, and 5 (2.3%) with therapeutic duplication (67 distinct errors in total, as some prescriptions had &#62;1 problem). No prescription in the test set contained a clinically significant drug-drug interaction, a physicochemical incompatibility, or a confirmed contraindication or special-population error. Consequently, the system performance reported here pertains to indication, dosage, and duplication errors; its proficiency in detecting interactions, incompatibilities, contraindications, and special-population problems could not be validated in this sample, even though each of these dimensions is specified in the prompt (addressed in the Limitations). All subsequent findings are based on this reference standard, with between-condition performance summarized in <xref ref-type="table" rid="table2">Table 2</xref>.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Prescription-review performance of unaided pharmacist review vs human-AI collaborative review on 213 outpatient prescriptions sampled at a tertiary teaching hospital in China (October 2025-November 2025) in a 2-period crossover study, using an independent expert reference standard (62 inappropriate prescriptions)<sup>a</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="200"/>
            <col width="140"/>
            <col width="150"/>
            <col width="150"/>
            <col width="180"/>
            <col width="180"/>
            <thead>
              <tr valign="top">
                <td>Groups</td>
                <td>Accuracy (%)</td>
                <td>Sensitivity (%)</td>
                <td>Specificity (%)</td>
                <td>False-positive rate (%)</td>
                <td>False-negative rate (%)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Pharmacist alone</td>
                <td>82.6</td>
                <td>55</td>
                <td>94</td>
                <td>6</td>
                <td>45</td>
              </tr>
              <tr valign="top">
                <td>Human-AI collaborative</td>
                <td>97.2</td>
                <td>98</td>
                <td>96.7</td>
                <td>3.3</td>
                <td>2</td>
              </tr>
              <tr valign="top">
                <td>Chi-square test (<italic>df</italic>)</td>
                <td>25.71 (1)</td>
                <td>25.04 (1)</td>
                <td>—<sup>b</sup></td>
                <td>—</td>
                <td>25.04 (1)</td>
              </tr>
              <tr valign="top">
                <td><italic>P</italic> value</td>
                <td>&#60;.001</td>
                <td>&#60;.001</td>
                <td>.29</td>
                <td>.29</td>
                <td>&#60;.001</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>Comparisons used paired McNemar tests with Yates continuity correction (213 paired observations). For accuracy, there were 33 discordant pairs that were correct only with collaboration and 2 that were correct only when unaided; for sensitivity and the false-negative rate, there were 27 and 0 discordant pairs, respectively. Accuracy denominators were 213; sensitivity, specificity, and the error rates used 62 inappropriate and 151 appropriate prescriptions, respectively.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>Specificity and false-positive-rate differences were not significant, and their test statistics are not reported.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Human-AI collaborative review significantly outperformed pharmacist-alone review in both overall accuracy and sensitivity (<xref ref-type="table" rid="table2">Table 2</xref>). Specificity did not differ between the 2 groups, indicating that AI support did not increase the misclassification of appropriate prescriptions as inappropriate and therefore added no meaningful false-positive risk.</p>
        <p>To characterize individual pharmacist differences, Cohen κ was calculated for each pharmacist’s unaided review against the reference standard. Because P1 and P2 reviewed different prescription batches, the 2 pharmacists were not directly compared. These values therefore reflect criterion agreement with the reference standard. They do not represent interrater reliability between P1 and P2. P1 yielded κ=0.554 (95% CI 0.360-0.749; observed agreement 84.1%) and P2 yielded κ=0.520 (95% CI 0.284-0.751; observed agreement 81.1%). By the Landis and Koch classification, both fall within the moderate-agreement interval (κ range: 0.41-0.60). The two estimates were close (&#124;Δκ&#124;=0.034), indicating similar criterion agreement, although κ values obtained on nonoverlapping batches cannot establish agreement between the raters themselves. A related caveat concerns the paired analysis: because different reviewers performed the two conditions for any given prescription, the prescription-level differences carry a rater component, which the counterbalanced crossover design balances across conditions at the aggregate level.</p>
        <p>Evaluated as a stand-alone classifier against the same reference standard, the knowledge-augmented AI flagged 77 prescriptions and achieved a sensitivity of 100% (62/62; 95% CI 94.2%-100%), a specificity of 90.1% (136/151; 95% CI 84.3%-93.9%), and an accuracy of 93% (198/213; 95% CI 88.7%-95.7%). The AI alone was thus highly sensitive but less specific than the collaborative condition, and its stand-alone accuracy (198/213, 93%) was lower than that achieved by the collaborative workflow (207/213, 97.2%). By retaining almost all the AI’s sensitivity while the pharmacist’s judgment raised specificity, the collaborative condition attained the highest overall accuracy, illustrating the complementary mechanism underlying the collaborative benefit. The increase in specificity, from 90.1% (136/151) in the stand-alone system to 96.7% (146/151) in the collaborative condition, was driven by pharmacist filtering of the model’s false alarms: of the 15 (7%) appropriate prescriptions that the AI overflagged, the pharmacists correctly dismissed 10 (4.7%) and accepted 5 (2.3%), while overriding 1 genuine alert (the single false negative in the collaborative arm).</p>
      </sec>
      <sec>
        <title>Detection of Different Problem Types</title>
        <p>Subgroup analysis showed that the magnitude of AI-assisted improvement varied by problem type (<xref ref-type="table" rid="table3">Table 3</xref>). Where sample sizes were adequate, the collaborative group consistently outperformed the pharmacist-alone group. Identification of inappropriate indications rose from 61% (31/51) to 98% (50/51; <italic>P</italic>&#60;.001), and identification of inappropriate dosages rose from 46% (5/11) to 100% (11/11; <italic>P</italic>=.03). Both categories rely on relatively objective criteria that can be verified against package inserts or diagnostic information, so the system’s knowledge base effectively compensated for gaps in pharmacist recall, yielding the clearest benefit from collaboration. Detection of therapeutic duplication improved from 20% (1/5) to 100% (5/5), but the small sample (N=5) precluded statistical significance (exact McNemar <italic>P</italic>=.13); this result is reported for reference only.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Detection rates for each type of prescription problem under unaided pharmacist review and human-AI collaborative review among 213 outpatient prescriptions from a tertiary teaching hospital in China<sup>a</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="350"/>
            <col width="240"/>
            <col width="220"/>
            <col width="190"/>
            <thead>
              <tr valign="top">
                <td>Problem types</td>
                <td>Pharmacist-alone, n (%)</td>
                <td>Collaborative, n (%)</td>
                <td>McNemar <italic>P</italic> values</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Indication inappropriateness (n=51)</td>
                <td>31 (61)</td>
                <td>50 (98)</td>
                <td>&#60;.001</td>
              </tr>
              <tr valign="top">
                <td>Dosage inappropriateness (n=11)</td>
                <td>5 (46)</td>
                <td>11 (100)</td>
                <td>.03</td>
              </tr>
              <tr valign="top">
                <td>Therapeutic duplication (n=5)</td>
                <td>1 (20)</td>
                <td>5 (100)</td>
                <td>.13</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>All subgroup sample sizes were small (n=5-51); exact McNemar tests (binomial exact) were used for paired comparisons. Detection rates were calculated per error: the denominators (n=51, 11, and 5) sum to the 67 distinct errors contained in the 62 inappropriate prescriptions, whereas the performance metrics in <xref ref-type="table" rid="table2">Table 2</xref> were calculated per prescription. The collaborative detection counts (50+11+5=66) therefore differ from the 61 true-positive prescriptions implied by <xref ref-type="table" rid="table2">Table 2</xref> because some prescriptions carried &#62;1 error.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Effect of Knowledge Augmentation on Model Hallucinations</title>
        <p>On the same 213-prescription test set, plain AI review produced 101 marked events, of which 42 were adjudicated as hallucinations affecting 42 distinct prescriptions, a per-prescription hallucination rate of 19.7% (42/213). All 42 hallucinations fell into 3 categories: erroneous dosing advice (45%, 19/42), fabricated drug interactions absent from the package insert (38%, 16/42), and false indications (17%, 7/42). After knowledge augmentation, the system produced 94 marked events with only 10 adjudicated hallucinations across 10 prescriptions (4.7%, 10/213), an absolute reduction of 15.0 percentage points and a relative reduction of 76.2%. Because both modes processed the same 213 prescriptions, the comparison was paired: hallucinations occurred only in plain mode for 33 (15.5%) prescriptions and only in augmented mode for 1 (0.5%), while 9 (4.2%) prescriptions had hallucinations in both modes and 170 (79.8%) had hallucinations in neither mode. The resulting McNemar <italic>χ</italic><sup>2</sup><sub>1</sub> was 28.26 (<italic>P</italic>&#60;.001). Among the 10 residual hallucinations in augmented mode, 8 (80%) were drug interaction errors, mostly in complex prescriptions with &#62;2 coprescribed drugs, suggesting the model may still extrapolate beyond injected knowledge when package-insert information for several drugs overlaps.</p>
        <p>Interrater reliability of hallucination adjudication was good, with Cohen κ of 0.858 (95% CI 0.756-0.959) for plain mode and 0.810 (95% CI 0.627-0.992) for knowledge-augmented mode, indicating substantial to almost perfect agreement.</p>
      </sec>
      <sec>
        <title>Review Efficiency</title>
        <p>The pharmacist-alone condition required 496.3 minutes to complete the full batch, whereas the human-AI collaborative condition required 238.6 minutes (N=213 prescriptions), yielding per-prescription review times of 2.33 (SD 0.97; median 1.95, IQR 1.47-2.92) minutes and 1.12 (SD 0.49; median 0.92, IQR 0.70-1.45) minutes, respectively. The mean within-prescription reduction was 1.21 (SD 0.50, 95% CI 1.14-1.28) minutes, representing an approximately 51.9% decrease under the collaborative condition, with shorter times observed in all 213 paired prescriptions. As the paired differences deviated from normality (Shapiro-Wilk <italic>W</italic>=0.91; <italic>P</italic>&#60;.001), the primary comparison used the Wilcoxon signed-rank test, which confirmed significantly shorter review times in the collaborative condition (<italic>Z</italic>=−12.65; <italic>P</italic>&#60;.001; <italic>r</italic>=.87); a paired <italic>t</italic> test produced a consistent result (t<sub>212</sub>=35.34; <italic>P</italic>&#60;.001; Cohen <italic>d<sub>z</sub></italic>=2.42, a very large effect).</p>
        <p>Two factors likely contributed: the system preselected potentially problematic prescriptions, enabling pharmacists to focus on higher-risk cases, and relevant package-insert information was retrieved automatically, eliminating manual reference lookup. Timing was recorded at the prescription level rather than the individual drug-item level, so the estimates reflect end-to-end review time per prescription.</p>
      </sec>
      <sec>
        <title>Pharmacist Acceptance and Feedback on System Suggestions</title>
        <p>When working collaboratively, the system sent a warning of a possible problem with 77 prescriptions. The reviewing pharmacist accepted 86% (66/77) of these alerts and overrode the remaining 14% (11/77) as overly sensitive false alarms. Acceptance records the pharmacist’s disposition rather than verified clinical validity. As noted previously, 5 (7.6%) of the 66 accepted alerts concerned prescriptions that the reference standard ultimately classified as appropriate. Of the 11 overrides, 10 (90.9%) proved correct on verification against that standard; the single incorrect override produced the only false negative in the collaborative group. During unstructured verbal debriefings after the final review session, both participating pharmacists expressed willingness to use the system in routine practice, remarking that it eased the burden of manual reference lookup and the stress of possible omissions.</p>
      </sec>
      <sec>
        <title>Sensitivity Analysis</title>
        <p>Some prescriptions flagged by experimental arms had been classified as appropriate under the reference standard. To assess whether such omissions affected the conclusions, these cases were returned to the expert panel for post hoc reevaluation. Three were confirmed as inappropriate and added to an extended reference standard (65 inappropriate prescriptions in total). Because both the pharmacist-alone and human-AI teams had correctly identified all 3, the discordant-pair structure between groups was unchanged.</p>
        <p>Under the expanded standard, the pharmacist-alone group achieved 84% accuracy (95% CI 78.5-88.3) and 57% sensitivity (95% CI 44.8-68.2), while the collaborative group reached 98.6% (95% CI 95.9-99.5) and 98% (95% CI 91.8-99.7), respectively. The McNemar test statistic remained identical to the main analysis (χ<sup>2</sup>=25.71; <italic>P</italic>&#60;.001), confirming that the overall conclusions are robust to minor imperfections in the reference standard.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>In this 2-period crossover study, pairing a locally deployed, knowledge-augmented LLM with pharmacist review substantially improved outpatient prescription review. Overall accuracy rose from 82.6% with unaided review to 97.2% with collaborative review, and sensitivity rose from 55% to 98%, while specificity was essentially unchanged; the gain in error detection was therefore not obtained at the cost of additional false alarms. Exact-match knowledge augmentation lowered the model hallucination rate from 19.7% to 4.7%. Collaborative review was also markedly faster, with mean per-prescription review time decreasing from 2.33 (SD 0.97) minutes to 1.12 (SD 0.49) minutes (≈51.9% reduction). Taken together, these findings indicate that a modest, locally hosted open-source model, used as a pharmacist-supervised prescreening tool, can meaningfully strengthen medication safety review without transmitting prescription data outside the hospital network.</p>
      </sec>
      <sec>
        <title>Rationale and Significance of Local-Deployment Human-AI Collaboration</title>
        <p>Using a crossover design, we compared unaided pharmacist review with review assisted by a locally deployed LLM. The collaborative mode achieved substantially higher accuracy, sensitivity, and efficiency, with the pharmacist retaining decisional authority and the LLM serving as a prescreening agent. This copilot approach is consistent with Ong et al [<xref ref-type="bibr" rid="ref8">8</xref>], who found that collaborative review outperformed either the model or the clinician alone, although their work relied on cloud-based GPT-4. Our study extends this rationale by relocating deployment from the cloud to the hospital intranet. Because Ollama keeps prescription data within the institutional network, it addresses the data-privacy requirements of Chinese medical facilities [<xref ref-type="bibr" rid="ref9">9</xref>]. These findings suggest that effective human-AI cooperation in pharmacy need not depend on commercial cloud models, since locally deployed open-source solutions can deliver comparable collaborative benefits while preserving privacy.</p>
        <p>Prior literature on LLMs in pharmacy has focused largely on isolated model evaluation, including drug interaction identification [<xref ref-type="bibr" rid="ref6">6</xref>], benchmarking against clinical pharmacists [<xref ref-type="bibr" rid="ref7">7</xref>], community pharmacy decision support [<xref ref-type="bibr" rid="ref14">14</xref>], and real-world drug information accuracy [<xref ref-type="bibr" rid="ref15">15</xref>]. These studies collectively show that ChatGPT responses are often incomplete or contain errors. Although they delineate the performance boundaries of LLMs, none addresses the more practically meaningful question of whether pharmacist work is improved by LLM assistance. By taking the pharmacist-LLM dyad as the unit of evaluation, our study quantifies net real-world benefit and thereby complements the existing evidence base.</p>
      </sec>
      <sec>
        <title>Technical Characteristics of the Knowledge Augmentation Strategy and Its Hallucination-Mitigation Mechanism</title>
        <p>Conventional RAG is significantly different from the exact-match injection approach. The clinical decision-support system outlined by Ong et al [<xref ref-type="bibr" rid="ref8">8</xref>] involves document chunking, vector indexing, and semantic retrieval, a pipeline that incurs substantial development and maintenance costs. We use the unique nature of drug identifiers to minimize knowledge retrieval to an individual deterministic database search and avoid the need for embedding models and vector storage. In prescription review in particular, the query goal is a clear-cut drug name as opposed to free text; exact matching is thus more reliable than fuzzy semantic retrieval and far less prone to noise.</p>
        <p>The assessment of hallucination provides additional support regarding the mechanism of the collaboration effect. Knowledge augmentation lowered the hallucination rate from 19.7% to 4.7%, corresponding to a relative reduction of approximately 76.2%, which indicates that exact-match injection meaningfully narrows the model’s output space and confers adequate trustworthiness for system-generated suggestions for routine pharmacist consultation. This further supports the finding that the acceptance rate in the collaborative group was 86%: a small hallucination rate is required so that pharmacists could trust and use system outputs effectively. It should also be noted that the evaluation of hallucinations was intended as an explanatory analysis in the current research; that is, it aims to elucidate the technical basis of the benefit of collaboration but not to act as an isolated main end point equivalent to accuracy and sensitivity.</p>
      </sec>
      <sec>
        <title>Clinical Effectiveness and Translational Value of the Collaborative Mode</title>
        <p>The collaborative mode demonstrated its principal value in sensitivity. The miss rate for inappropriate prescriptions fell from 45% under pharmacist-only review to 2% with AI support. System prescreening reminders served as a safety net for lapses arising from heavy workloads or imperfect recall of drug-specific information during independent review. Specificity remained high in both groups (142/151, 94% vs 146/151, 96.7%) with no statistically significant difference, indicating that AI support added no meaningful false-positive risk. The pharmacists’ appraisal of system alerts was nonetheless imperfect: two-thirds of the model’s false alarms were correctly dismissed (10/15, 66.7%), while the remaining third (5/15, 33.3%) were accepted. This finding aligns with Ong et al [<xref ref-type="bibr" rid="ref8">8</xref>], who reported that collaborative review does not increase false positives, and with Singhal et al [<xref ref-type="bibr" rid="ref3">3</xref>], who showed that LLMs can perform at an expert level when provided with adequate contextual knowledge. Subgroup analysis revealed the greatest gains in knowledge-intensive, rule-based domains such as indication and dosage appropriateness, whereas improvements were smaller for judgments requiring integration of individualized patient information, defining both the current scope of the system and directions for future development.</p>
        <p>From a clinical standpoint, raising sensitivity from 55% to 98% carries substantial weight. As a post hoc quality assurance measure, prescription review derives its core value from systematic identification of irrational drug use and the continuous feedback that subsequently shapes prescribing behavior. A high miss rate directly undermines this loop: undetected inappropriate prescriptions never enter quality control, the prescribing physicians receive no corrective feedback, and the same errors tend to recur. Reducing the false-negative rate from 45% to 2% produces a more representative profile of prescribing problems in review outputs, offering a more accurate basis for departmental feedback and behavioral intervention. Over time, improved detection should reinforce a positive review-feedback-improvement cycle and curtail inappropriate prescribing at its source. The benefit is particularly relevant for primary care institutions, where pharmacist staffing is limited and full review coverage is difficult to sustain; the collaborative approach markedly improves efficiency, substantially reduces missed detections, and offers a practical solution for medication safety oversight in such settings.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>The 213-prescription sample reflected an administrative rule rather than an a priori power calculation; we report Wilson-score 95% CIs so that estimate precision is directly visible. Subgroup intervals are wide, and the therapeutic duplication (n=5, 2.3%) and dosage inappropriateness (n=11, 5.2%) findings should be treated as exploratory pending validation in larger multicenter samples covering uncommon combinations and special populations. Both reviewers were junior pharmacists (1-2 years’ experience), so the AI-related gain may be smaller for senior staff and should not be extrapolated across seniority without further study. The 2-period crossover with a 2-week washout balanced period and rater effects but cannot fully exclude asymmetrical learning: knowledge acquired under AI assistance may have raised later unaided performance, biasing the estimated benefit toward the null and rendering the reported improvement, if anything, conservative.</p>
        <p>The system itself has intrinsic scope limits. It analyzes each prescription in isolation, with no access to longitudinal data such as prior medications, allergies, or renal and hepatic function; dosage and special-population review were therefore judged only against package-insert dose ranges, indications, and population statements alongside the recorded diagnosis, and cumulative polypharmacy was not evaluated. Some dosing decisions cannot be fully adjudicated without laboratory parameters. The 14-billion-parameter Qwen3 model is less capable than leading proprietary models, which may constrain deep-reasoning tasks; moreover, the model ran under 4 bits (Q4_K_M) quantization, which may modestly reduce reasoning performance or alter hallucination behavior relative to full-precision weights, so the stand-alone results reported here reflect a quantized deployment and may represent a performance floor. Even so, the collaborative gains achieved under local deployment indicate that scale is not the sole determinant of assistive value. Inference took 15 to 30 seconds per prescription on consumer-grade hardware. Because this inference was performed as a batch preprocessing step before the review session, the latency was not included in the recorded per-prescription review time and did not affect the efficiency comparison; it would, however, lengthen overall batch turnaround and could become a bottleneck at very high outpatient volumes, where more capable hardware, batched processing, or parallel inference would be required. Beyond hardware, local deployment carries recurring operating costs that we did not quantify, including server maintenance, electricity, and the staff time needed to keep the package-insert knowledge base current. We also did not perform a formal economic evaluation, which should precede any claim about affordability or cost-effectiveness. The 583-entry knowledge base is a subset of the hospital’s formulary; off-base drugs receive no knowledge injection and are routed for manual review, so generalizability depends on continued expansion both locally and at sites with broader formularies. The reference standard classified 29.1% (62/213) of sampled prescriptions as inappropriate, a rate that reflects the 0.1% administrative sampling of a tertiary teaching hospital and may exceed the prevalence in many outpatient or primary care settings. Because positive predictive value falls as prevalence declines, the same model operating at a lower error prevalence would return a larger proportion of false-positive alerts: for the stand-alone system (sensitivity approximately 100%, specificity approximately 90%), illustrative positive predictive value decreases from approximately 80% at the observed prevalence to approximately 53% at a prevalence of 10% and 35% at 5%. A higher false-alarm fraction could aggravate alert fatigue and automation complacency, although the model’s high specificity limits the absolute number of false alerts and the human-in-the-loop design allows pharmacists to filter them; characterizing the real-world alert burden will therefore require prospective evaluation across settings with differing baseline error rates.</p>
        <p>Measurement issues also apply. The absolute hallucination counts were small (42 plain and 10 augmented), limiting the precision of the type distribution; severity was not graded, and a clinical-significance taxonomy would enable finer comparison of augmentation strategies. Only explicit problem statements were adjudicated, so hallucinations embedded in justifications for prescriptions deemed appropriate were missed, making the per-prescription rate probably conservative. The sample contained no confirmed drug-drug interaction or physicochemical incompatibility error, leaving proficiency in these clinically critical domains unvalidated despite prompt coverage. The post hoc sensitivity analysis was unblinded, introducing some incorporation bias, and is reported only as a robustness check.</p>
        <p>Finally, the collaborative efficiency advantage partly reflected preflagging of suspect prescriptions, raising a real risk that busy pharmacists rely on alerts and underscrutinize unflagged orders. Safe deployment therefore requires AI alerts to supplement rather than replace complete pharmacist review, with workflow safeguards, periodic audit of nonflagged prescriptions, and targeted training to manage automation complacency before broader implementation.</p>
      </sec>
      <sec>
        <title>Future Directions</title>
        <p>Future work should pursue several complementary directions. Domain-specific fine-tuning informed by expert-reviewed prescriptions and pharmacist feedback could further refine accuracy for particular drug classes [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Deeper integration with the hospital information system would enable automated retrieval of laboratory results, allergy history, and medication history, supporting genuinely patient-level rather than prescription-level analysis. A closed-loop learning mechanism that incorporates pharmacist accept or reject decisions could progressively improve advisory quality. Privacy-preserving, deidentified sharing of knowledge bases and exemplar cases across institutions also warrants exploration. If licensing and offline deployment allow, adding richer resources such as Micromedex or UpToDate to the package-insert knowledge base could broaden indication and dosage coverage.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>In this study, the pharmacist was kept at the center of decision-making, and a prescription-review decision-support system was built on the Qwen3-14B LLM, deployed inside the hospital through the Ollama platform together with a structured drug knowledge base. The crossover evaluation showed that AI assistance increased pharmacist review accuracy from 82.6% to 97.2% and sensitivity from 55% to 98% (both <italic>P</italic>&#60;.001), while shortening the per-prescription review time from 2.33 (SD 0.97) to 1.12 (SD 0.49) minutes, an approximately 51.9% reduction (Wilcoxon signed-rank <italic>Z</italic>=−12.65; <italic>P</italic>&#60;.001; <italic>r</italic>=0.87; N=213). Knowledge augmentation reduced the model hallucination rate from 19.7% to 4.7% (relative reduction 76.2%). By keeping all prescription data within the hospital’s own network, local deployment offers health care institutions a workable and secure way to integrate AI-driven pharmacy services without stepping outside their existing data-compliance requirements.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Hardware and software specifications, additional details of the reference standard, and additional statistical analysis details.</p>
        <media xlink:href="medinform_v14i1e97520_app1.docx" xlink:title="DOCX File , 17 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Verbatim prompt template used by the locally deployed large language model for outpatient prescription review (with the Chinese original and English translation), including the role definition, task specification, knowledge context, and output format components.</p>
        <media xlink:href="medinform_v14i1e97520_app2.docx" xlink:title="DOCX File , 26 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">RAG</term>
          <def>
            <p>retrieval-augmented generation</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare the use of the generative AI tool DeepSeek (DeepSeek-V3; DeepSeek AI) for language editing and formatting of the manuscript.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The datasets analyzed in this study are deidentified routine clinical records held by the institution and cannot be released openly under the hospital’s data-governance policy. The deidentified dataset may be shared for noncommercial academic use upon reasonable request to the corresponding author, subject to a data-use agreement and relevant institutional approvals.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The authors declare that no financial support was received for the research or publication of this article.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: ZW</p>
        <p>Data curation: JC</p>
        <p>Investigation: YD, ZY, JC, XC, WZ</p>
        <p>Methodology: ZL</p>
        <p>Project administration: ZW</p>
        <p>Software: ZL</p>
        <p>Supervision: PF, ZW</p>
        <p>Validation: PF, ZW</p>
        <p>Writing—original draft: ZL, JC</p>
        <p>Writing—review and editing: YD, ZY, XC, WZ, PF, ZW</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="web">
          <article-title>Issuing the "hospital prescription review management standards (trial)"</article-title>
          <source>National Health and Family Planning Commission of the People's Republic of China</source>
          <year>2013</year>
          <access-date>2025-01-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.nhc.gov.cn/zwgkzt/glgf/201306/094ebc83dddc47b5a4a63ebde7224615.shtml">https://www.nhc.gov.cn/zwgkzt/glgf/201306/094ebc83dddc47b5a4a63ebde7224615.shtml</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Qian</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Application and development of artificial intelligence in hospital pharmacy services</article-title>
          <source>Adv Clin Med</source>
          <year>2023</year>
          <volume>13</volume>
          <issue>8</issue>
          <fpage>12536</fpage>
          <lpage>41</lpage>
          <pub-id pub-id-type="doi">10.12677/ACM.2023.1381758</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>80</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gutierrez</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>Large language models in medicine</article-title>
          <source>Nat Med</source>
          <year>2023</year>
          <month>08</month>
          <volume>29</volume>
          <issue>8</issue>
          <fpage>1930</fpage>
          <lpage>40</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
          <pub-id pub-id-type="medline">37460753</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-023-02448-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pais</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Voigt</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Wade</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Bayati</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Large language models for preventing medication direction errors in online pharmacies</article-title>
          <source>Nat Med</source>
          <year>2024</year>
          <month>06</month>
          <volume>30</volume>
          <issue>6</issue>
          <fpage>1574</fpage>
          <lpage>82</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38664535"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41591-024-02933-8</pub-id>
          <pub-id pub-id-type="medline">38664535</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-02933-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC11186789</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Roosan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Padua</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Verzosa</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Effectiveness of ChatGPT in clinical pharmacy and the role of artificial intelligence in medication therapy management</article-title>
          <source>J Am Pharm Assoc (2003)</source>
          <year>2024</year>
          <volume>64</volume>
          <issue>2</issue>
          <fpage>422</fpage>
          <lpage>8.e8</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1544-3191(23)00384-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.japh.2023.11.023</pub-id>
          <pub-id pub-id-type="medline">38049066</pub-id>
          <pub-id pub-id-type="pii">S1544-3191(23)00384-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Estau</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Qin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Evaluating the performance of ChatGPT in clinical pharmacy: a comparative study of ChatGPT and clinical pharmacists</article-title>
          <source>Br J Clin Pharmacol</source>
          <year>2024</year>
          <month>01</month>
          <volume>90</volume>
          <issue>1</issue>
          <fpage>232</fpage>
          <lpage>8</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1111/bcp.15896"/>
          </comment>
          <pub-id pub-id-type="doi">10.1111/bcp.15896</pub-id>
          <pub-id pub-id-type="medline">37626010</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ong</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Jin</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>GY</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>DY</given-names>
            </name>
            <name name-style="western">
              <surname>Sng</surname>
              <given-names>GG</given-names>
            </name>
            <name name-style="western">
              <surname>Ke</surname>
              <given-names>YH</given-names>
            </name>
            <name name-style="western">
              <surname>Tung</surname>
              <given-names>JY</given-names>
            </name>
            <name name-style="western">
              <surname>Zhong</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Koh</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>KZ</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Ch'ng</surname>
              <given-names>JK</given-names>
            </name>
            <name name-style="western">
              <surname>Than</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Goh</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>CP</given-names>
            </name>
            <name name-style="western">
              <surname>Ng</surname>
              <given-names>TM</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>Large language model as clinical decision support system augments medication safety in 16 clinical specialties</article-title>
          <source>Cell Rep Med</source>
          <year>2025</year>
          <month>10</month>
          <day>21</day>
          <volume>6</volume>
          <issue>10</issue>
          <fpage>102323</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2666-3791(25)00396-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102323</pub-id>
          <pub-id pub-id-type="medline">40997804</pub-id>
          <pub-id pub-id-type="pii">S2666-3791(25)00396-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC12629785</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="web">
          <article-title>Notice from the general office of the national health commission on issuing reference guidelines for artificial intelligence application scenarios in the health sector</article-title>
          <source>National Health Commission of the People’s Republic of China</source>
          <year>2024</year>
          <access-date>2025-01-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.nhc.gov.cn/guihuaxxs/c100133/202411/3dee425b8dc34f739d63483c4e5c334c.shtml">https://www.nhc.gov.cn/guihuaxxs/c100133/202411/3dee425b8dc34f739d63483c4e5c334c.shtml</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Frieske</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Su</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ishii</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Bang</surname>
              <given-names>YJ</given-names>
            </name>
            <name name-style="western">
              <surname>Madotto</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Fung</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Survey of hallucination in natural language generation</article-title>
          <source>ACM Comput Surv</source>
          <year>2023</year>
          <month>03</month>
          <day>03</day>
          <volume>55</volume>
          <issue>12</issue>
          <fpage>1</fpage>
          <lpage>38</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1145/3571730"/>
          </comment>
          <pub-id pub-id-type="doi">10.1145/3571730</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ning</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yin</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chua</surname>
              <given-names>TS</given-names>
            </name>
            <name name-style="western">
              <surname>LI</surname>
              <given-names>Q</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Baeza-Yates</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bonchi</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>A survey on RAG meeting LLMs: towards retrieval-augmented large language models</article-title>
          <source>KDD '24: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining</source>
          <year>2024</year>
          <publisher-loc>New York, NY</publisher-loc>
          <publisher-name>Association for Computing Machinery</publisher-name>
          <fpage>6491</fpage>
          <lpage>501</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lachin</surname>
              <given-names>JM</given-names>
            </name>
          </person-group>
          <article-title>Power and sample size evaluation for the McNemar test with application to matched case-control studies</article-title>
          <source>Stat Med</source>
          <year>1992</year>
          <month>06</month>
          <day>30</day>
          <volume>11</volume>
          <issue>9</issue>
          <fpage>1239</fpage>
          <lpage>51</lpage>
          <pub-id pub-id-type="doi">10.1002/sim.4780110909</pub-id>
          <pub-id pub-id-type="medline">1509223</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="web">
          <article-title>Notice on issuing the measures for ethical review of life science and medical research involving human subjects</article-title>
          <source>National Health Commission of the People’s Republic of China</source>
          <access-date>2023-02-18</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.gov.cn/zhengce/zhengceku/2023-02/28/content_5743658.htm">https://www.gov.cn/zhengce/zhengceku/2023-02/28/content_5743658.htm</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Hartman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ramanathan</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Performance of the ChatGPT large language model for decision support in community pharmacy</article-title>
          <source>Br J Clin Pharmacol</source>
          <year>2024</year>
          <month>12</month>
          <volume>90</volume>
          <issue>12</issue>
          <fpage>3320</fpage>
          <lpage>33</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1111/bcp.16215"/>
          </comment>
          <pub-id pub-id-type="doi">10.1111/bcp.16215</pub-id>
          <pub-id pub-id-type="medline">39191671</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Morath</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Chiriac</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Jaszkowski</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Deiß</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Nürnberg</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Hörth</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Hoppe-Tichy</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Green</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Performance and risks of ChatGPT used in drug information: an exploratory real-world analysis</article-title>
          <source>Eur J Hosp Pharm</source>
          <year>2024</year>
          <month>10</month>
          <day>25</day>
          <volume>31</volume>
          <issue>6</issue>
          <fpage>491</fpage>
          <lpage>7</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://ejhp.bmj.com/lookup/pmidlookup?view=long&#38;pmid=37263772"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/ejhpharm-2023-003750</pub-id>
          <pub-id pub-id-type="medline">37263772</pub-id>
          <pub-id pub-id-type="pii">ejhpharm-2023-003750</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Cheung</surname>
              <given-names>KC</given-names>
            </name>
            <name name-style="western">
              <surname>See</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>NL</given-names>
            </name>
          </person-group>
          <article-title>TCM-FTP: fine-tuning large language models for herbal prescription prediction</article-title>
          <source>arXiv. Preprint posted online on July 15, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2407.10510</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sayres</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wulczyn</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Amin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>SR</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Neal</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Rashid</surname>
              <given-names>QM</given-names>
            </name>
            <name name-style="western">
              <surname>Schaekermann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dash</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>JH</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>NH</given-names>
            </name>
            <name name-style="western">
              <surname>Lachgar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Prakash</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Green</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Dominowska</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tomašev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>JK</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>DR</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Toward expert-level medical question answering with large language models</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <month>03</month>
          <volume>31</volume>
          <issue>3</issue>
          <fpage>943</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="medline">39779926</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-03423-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11922739</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
