<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e89173</article-id><article-id pub-id-type="doi">10.2196/89173</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Locally Executable AI System for Improving Preoperative Patient Communication: Multidomain Clinical Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Sato</surname><given-names>Motoki</given-names></name><degrees>MD, MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nagata</surname><given-names>Sou</given-names></name><degrees>DDS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ohnuma</surname><given-names>Mizuho</given-names></name><degrees>DDS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Takahashi</surname><given-names>Hidekazu</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kakazu</surname><given-names>Tomoaki</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yamamura</surname><given-names>Masayuki</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yoshikawa</surname><given-names>Atsushi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Matsushita</surname><given-names>Yuki</given-names></name><degrees>DDS, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Skeletal Development and Regenerative Biology, Graduate School of Biomedical Sciences, Nagasaki University</institution><addr-line>3F Building for Basic Dental Science, 1-7-1 Sakamoto</addr-line><addr-line>Nagasaki</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Boston Medical Sciences, Inc.</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff3"><institution>Digestive Diseases Center, Showa University Koto Toyosu Hospital</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff4"><institution>Department of Computer Science, School of Computing, Institute of Science Tokyo</institution><addr-line>Tokyo</addr-line><country>Japan</country></aff><aff id="aff5"><institution>College of Informatics, Kanto Gakuen University</institution><addr-line>Yokohama</addr-line><addr-line>Kanagawa</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Kuziemsky</surname><given-names>Craig</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Rivera</surname><given-names>David</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Motoki Sato, MD, MS, Department of Skeletal Development and Regenerative Biology, Graduate School of Biomedical Sciences, Nagasaki University, 3F Building for Basic Dental Science, 1-7-1 Sakamoto, Nagasaki, Japan, 81 95-819-7633, 81 95-819-7633; <email>m070039@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e89173</elocation-id><history><date date-type="received"><day>09</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>09</day><month>05</month><year>2026</year></date><date date-type="accepted"><day>10</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Motoki Sato, Sou Nagata, Mizuho Ohnuma, Hidekazu Takahashi, Tomoaki Kakazu, Masayuki Yamamura, Atsushi Yoshikawa, Yuki Matsushita. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 21.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e89173"/><abstract><sec><title>Background</title><p>Patients undergoing invasive procedures frequently experience anxiety and often have unanswered questions regarding the procedure. Although large language models show considerable promise for supporting patient communication in many cases, their deployment in health care is limited by the risk of hallucinations, data-privacy constraints, and high energy costs&#x2014;factors that impede equitable access in resource-limited settings.</p></sec><sec><title>Objective</title><p>This study aims to develop and evaluate LENOHA (Low Energy, No Hallucination, Leave No One Behind Architecture), a locally executable dialog system for safe, equitable, and sustainable preprocedural communication.</p></sec><sec sec-type="methods"><title>Methods</title><p>We built expert-curated FAQ (frequently asked question) databases and independent test sets for 2 domains (tooth extraction and gastroscopy; 200 utterances per domain: 100 clinical questions and 100 casual). A sentence-transformer classifier routed inputs: clinical questions were answered verbatim from the vetted FAQs (nongenerative path), while casual conversation was handled by a locally hosted 8-billion-parameter small language model (Swallow-8B). We evaluated 4 sentence-transformer models (including E5-large-instruct) against cloud large language models (ChatGPT [GPT-4o] and Gemini Advanced) using accuracy, <italic>F</italic><sub>1</sub>-score, and area under the receiver operating characteristic curve, and measured the on-device inference energy on a consumer graphics processing unit (RTX 3080).</p></sec><sec sec-type="results"><title>Results</title><p>Across both domains (N=400), E5-large-instruct achieved an accuracy of 98.3% (393/400; 95% CI 96.4%&#x2010;99.1%) and an area under the curve of 0.996, with only 7 out of 400 (1.8%) misclassifications. This performance was not statistically different from that of ChatGPT (GPT-4o), which had 6 out of 400 (1.5%) errors (McNemar test with Holm adjustment; <italic>P</italic>&#x003E;.99). Sustainability measurements showed approximately 2.23 mWh per request (latency&#x2248;0.10 s; video RAM&#x2248;2.2 GiB average, &#x2248;2.5 GiB peak) for the nongenerative clinical path vs approximately 168.27 mWh (latency&#x2248;8.51 s; video RAM&#x2248;13.3 GiB average, &#x2248;14.0 GiB peak) for small language model small talk&#x2014;approximately a 75-fold higher energy footprint per reply for the generative path.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>High-precision, nongenerative clinical support is feasible using local, low-cost hardware without cloud dependence. By decoupling clinical information retrieval from generative chitchat, LENOHA enhances safety, preserves privacy, and markedly reduces energy use, offering a practical blueprint for sustainable and equitable medical AI deployment across diverse care settings.</p></sec></abstract><kwd-group><kwd>AI</kwd><kwd>artificial intelligence</kwd><kwd>large language models</kwd><kwd>natural language processing</kwd><kwd>privacy</kwd><kwd>energy metabolism</kwd><kwd>doctor-patient communication</kwd><kwd>implementation science</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Patients undergoing invasive medical procedures, such as tooth extraction or upper gastrointestinal endoscopy (gastroscopy), commonly experience preprocedural questions and anxiety [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Providing patients with appropriate information and reassurance through effective communication is critical for building trust and promoting active engagement in treatment, which are essential factors in successful clinical care [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Interventions, such as educational materials and structured communication strategies, have been shown to significantly enhance patient understanding, satisfaction, and involvement in decision-making [<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>Digital technologies have been shown to improve early comprehension of surgical informed consent without increasing anxiety or reducing satisfaction [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>In today&#x2019;s high-pressure clinical environments, it is not always feasible for health care professionals to provide fully personalized explanations to every patient. In Japan, physician shortages in rural areas and surgical specialties limit the time available for thorough preprocedural communication, while patients often hesitate to ask questions for fear of burdening already busy staff [<xref ref-type="bibr" rid="ref5">5</xref>]. Prior studies have shown that presenting multimedia or electronic consent materials before the consultation can reduce face-to-face explanation time by approximately 30% to 60% while maintaining patient understanding, anxiety levels, and satisfaction, suggesting that such approaches can improve time efficiency for both patients and health care professionals [<xref ref-type="bibr" rid="ref6">6</xref>]. Moreover, microcosting analyses from the UK National Health Service indicate that digital consent pathways can be at least cost-neutral and, in some scenarios, slightly less expensive compared with paper-based workflows [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>In recent years, large language models (LLMs) have demonstrated remarkable potential in patient education, psychological support, and workflow optimization [<xref ref-type="bibr" rid="ref8">8</xref>]. For example, recent evidence suggests that AI-based LLMs offer a promising avenue for improving the quality and readability of oral surgery informed consent documents, outperforming conventional web-based materials in standardized assessments [<xref ref-type="bibr" rid="ref9">9</xref>]. However, the deployment of LLMs in real-world health care contexts poses several challenges.</p><p>First, ensuring the reliability and safety of the information generated by LLMs is a paramount concern [<xref ref-type="bibr" rid="ref10">10</xref>]. LLMs are inherently prone to generating factually incorrect content, referred to as &#x201C;hallucinations,&#x201D; and may reproduce biases present in their training data, potentially exacerbating existing health disparities and posing significant risks to patient safety [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>From a theoretical perspective, recent work has shown that hallucinations cannot be eliminated for any computable LLM, even with more parameters, more data, or sophisticated prompting, indicating that some residual risk of incorrect output is inherent to the technology [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Retrieval-augmented generation (RAG) has emerged as a promising strategy to enhance reliability by referencing external knowledge bases [<xref ref-type="bibr" rid="ref13">13</xref>]. RAG itself is not a panacea and has inherent challenges, such as the &#x201C;lost-in-the-middle&#x201D; problem, where LLMs may fail to use retrieved information effectively [<xref ref-type="bibr" rid="ref14">14</xref>]. Consequently, recent research has shifted toward architectural safeguards that constrain how LLMs access and use external information. One representative example is the Model Context Protocol (MCP; Anthropic), which standardizes how language models ingest and isolate external data, thereby decoupling stochastic reasoning from deterministic execution [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. For instance, a recent study by Avila et al [<xref ref-type="bibr" rid="ref17">17</xref>] demonstrated that using an MCP-based architecture in structural analysis reduced predictive deviations from over 400% (in unconstrained LLMs) to under 1.5% [<xref ref-type="bibr" rid="ref17">17</xref>]. Furthermore, such architectural compartmentalization ensures reproducibility and traceability, which are critical requirements for high-stakes environments. These findings strongly indicate that rigid architectural constraints, rather than basic RAG or prompt engineering, are essential for maintaining rigorous safety standards.</p><p>Second, practical barriers to deployment hinder equitable access to AI. The reliance on cloud-based APIs for most leading general-purpose LLMs raises substantial privacy and security concerns, whereas their high computational and energy demands render local execution infeasible for many health care facilities, contributing to a significant environmental footprint. Health care systems are already responsible for 4% to 5% of global greenhouse gas emissions, making the environmental impact of AI a pressing issue [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. As highlighted in a recent review by Ueda et al [<xref ref-type="bibr" rid="ref20">20</xref>], the rapid expansion of AI in Japanese health care necessitates a shift toward sustainable practices to mitigate its environmental impact, including greenhouse gas emissions from intensive computing resources [<xref ref-type="bibr" rid="ref20">20</xref>]. Furthermore, avenues for accessibility face their own equity challenges; automatic speech recognition (ASR) systems exhibit performance disparities across diverse accents [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>], and recent studies show that leading LLMs, such as ChatGPT (GPT-4o), produce clinical vignettes that stereotype demographic presentations and alter recommendations based on patient race and gender [<xref ref-type="bibr" rid="ref23">23</xref>]. These findings underscore the critical need for novel architectural approaches that explicitly prioritize safety, equity, and environmental sustainability. While large models dominate the current AI discourse, Jeanquartier et al [<xref ref-type="bibr" rid="ref24">24</xref>] emphasized that small language models (SLMs) offer significant promise for application scenarios where resource efficiency and data privacy are paramount, such as in offline medical devices. By circumventing the massive computational costs of larger models, SLMs provide a more accessible and sustainable pathway for health care informatics. Rather than pursuing incremental performance gains within the existing flawed holistic evaluation framework, this paper proposes an architectural framework&#x2014;a blueprint designed from first principles to be safe, equitable, and sustainable through the integration of &#x201C;eco-design&#x201D; and local execution.</p><p>Third, a holistic framework is required for the ethical and sustainable implementation of AI in health care. Morley et al [<xref ref-type="bibr" rid="ref19">19</xref>] recently advocated for a systems approach, proposing five core infrastructural requirements for ethical AI implementation: (1) robust data exchange, (2) epistemic certainty with staff autonomy, (3) actively protected health care values, (4) validated outcomes with meaningful accountability, and (5) environmental sustainability. Similarly, to systematize ethical considerations, Ning et al [<xref ref-type="bibr" rid="ref25">25</xref>] advocated for a comprehensive assessment checklist for generative AI research grounded in 9 core principles: accountability, autonomy, equity, nonmaleficence, privacy, security, integrity (in medical education and quality of clinical research), transparency, and trust [<xref ref-type="bibr" rid="ref25">25</xref>]. These multifaceted challenges underscore the need for new architectural approaches that prioritize safety, equity, and practical applicability. Building upon these foundations to address the specific needs of resource-constrained settings, the recently developed SAFE-AI (Scalable Agile Framework for Execution in AI) framework by Nemteanu et al [<xref ref-type="bibr" rid="ref26">26</xref>] emphasizes the necessity of embedding lightweight, &#x201C;minimum necessary safeguards&#x201D; and risk interpretability directly into the AI development life cycle.</p><p>To overcome these limitations, this study adopts a different approach. We introduce the LENOHA system (Low Energy, No Hallucination, Leave No One Behind Architecture). Its underlying principle aligns with the fundamentals of sustainable AI, particularly in settings constrained by limited infrastructure and resources. To evaluate the utility and generalizability of this architecture, we constructed independent test datasets from 2 distinct clinical domains&#x2014;oral and maxillofacial surgery (tooth extraction) and gastroenterology (gastroscopy)&#x2014;and conducted a comparative performance assessment across multiple model architectures, including open-source sentence-transformer (ST) models and commercial LLMs. The results demonstrate that high-precision classification is achievable even on modest local hardware, offering a feasible and scalable solution to one of the most pressing bottlenecks of health care AI.</p><p>The primary aim of this study was to develop and evaluate LENOHA, a locally executable dual-pathway architecture that routes clinical queries to a nongenerative, clinician-curated FAQs (frequently asked questions) pathway and casual conversations to a local generative model.</p><p>We assessed whether this design can reliably separate clinical from casual utterances in preprocedural settings while reducing privacy risks and operational energy compared to a fully generative approach.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Clinical Domains</title><p>In this study, we focused on a single clinical problem: preprocedural patient communication for invasive but relatively low-risk procedures. We explicitly followed a stepwise AI life cycle approach, focusing on the early development and technical validation phases rather than the in-workflow clinical deployment. Consistent with stepwise implementation models, such as those of van de Sande et al [<xref ref-type="bibr" rid="ref27">27</xref>] and broader life cycle concepts proposed by Kuziemsky et al [<xref ref-type="bibr" rid="ref28">28</xref>], this study corresponds to phase 1 (AI model development) and phase 2 (assessment of AI performance and reliability) and does not yet encompass phase 3 clinical testing with actual patients. To test whether our architectural approach can be generalized across heterogeneous clinical settings, we deliberately selected 2 distinct domains&#x2014;oral and maxillofacial surgery (tooth extraction) and gastroenterology (gastroscopy)&#x2014;that differ in specialty, workflow, and risk framing, while sharing the same need for scalable preprocedural counseling. For each domain, we prepared 3 resources:</p><list list-type="bullet"><list-item><p>An expert-curated FAQ database was used as the target of the ST-based matcher.</p></list-item><list-item><p>A validation dataset was used to determine the operating thresholds.</p></list-item><list-item><p>An independent test dataset was used for the final evaluation.</p></list-item></list><p>Utterances classified as clinical questions were routed to the nongenerative FAQ path, and utterances classified as casual conversation were intended to be handled by a locally executed SLM.</p><p>A schematic overview of the system&#x2019;s data flow is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Schematic overview of the LENOHA system data flow. All processing was performed locally on a single graphics processing unit (GPU) to preserve patient privacy. A local sentence-transformer (ST) classifier first categorized the patient input. Inputs identified as clinical questions were routed to a nongenerative pathway, which retrieved a verbatim, clinician-curated answer from the frequently asked question (FAQ) database, minimizing the risk of hallucination. In contrast, casual conversations were routed to a generative pathway, where a local small language model (SLM; Swallow-8B) produced an empathetic response.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e89173_fig01.png"/></fig></sec><sec id="s2-2"><title>FAQ Databases</title><sec id="s2-2-1"><title>Tooth Extraction</title><p>Under the supervision of dentists and oral and maxillofacial surgeons, we defined categories based on FAQs observed in routine Japanese practice (eg, travel, postoperative daily life, jaw or mouth opening, anxiety, anesthesia, dysesthesia, return to work, bleeding, diet, swelling, pain, indication, procedure time, bone resection, medication, temporomandibular joint disorder, infection, cost, postoperative prosthetics, pregnancy or breastfeeding, comorbidities, cancellation or interruption, presence of wisdom teeth, tooth longevity, and suture removal). The complete clinical schema template used for this categorization is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>For each category, we collected multiple colloquial paraphrases in standard Japanese rather than a single template to reflect real patient utterances. In total, we compiled approximately 4000 domain-specific questions, which substantially exceeded the coverage reported in previous FAQ-style systems and was intended to reduce false negatives during inference.</p></sec><sec id="s2-2-2"><title>Gastroscopy</title><p>Using the tooth extraction schema (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) as a template and in consultation with physicians and endoscopists, we adapted the categories to the endoscopy setting: items that were not relevant (eg, &#x201C;jaw&#x201D;) were removed, and endoscopy-specific topics (eg, <italic>Helicobacter pylori</italic>, preparation, sedation or throat anesthesia, infection control, postexamination driving or work, emergency contact, and rescheduling) were added. The final endoscopy FAQ contained approximately 1600 clinician-curated questions, each paraphrased in standard Japanese.</p></sec></sec><sec id="s2-3"><title>Definitions</title><p>To rigorously separate utterances that require medical oversight from those that do not, we defined 2 operational categories that were applied across both domains. This distinction was designed to filter out LLM hallucination risks for clinical content while still enabling supportive replies for nonclinical talk.</p><sec id="s2-3-1"><title>Clinical Question</title><p>Utterances explicitly seeking medical, clinical, or practical judgments and information require a clinically consistent and institutionally aligned response. These utterances were considered unsafe for free-form LLM generation.</p><p>Examples of intent included questions about postprocedural life, work, or diet; medical concerns regarding anesthesia, bleeding, pain, or infection; specific logistical queries (such as duration, cost, and precautions); consultations regarding regular medications, comorbidities, pregnancy, or breastfeeding; and requests for specific medical advice from health care staff.</p></sec><sec id="s2-3-2"><title>Casual Conversation (Small Talk)</title><p>Utterances without clinical intent that are not mapped to any FAQ category.</p><p>Examples of intent include greetings, weather discussions, general small talk, personal updates (excluding medical symptom consultations), light expressions of gratitude, and simple expressions of emotion (eg, &#x201C;I&#x2019;m nervous&#x201D; and &#x201C;I&#x2019;m scared&#x201D;) that do not explicitly ask for medical coping strategies.</p></sec></sec><sec id="s2-4"><title>Operational Rule and Decision Boundary</title><p>The critical criterion for classification was whether the speaker expected a clinically aligned medical response. For instance, the simple emotional expression, &#x201C;I am nervous&#x201D; is classified as Casual (Small Talk), whereas &#x201C;Are there any ways to ease my nerves?&#x201D; is classified as a clinical question requiring a safe, vetted response. The classifier&#x2019;s operational rule was strictly defined: if an utterance did not match any clinical or FAQ category above the confidence threshold, it was treated as casual conversation.</p></sec><sec id="s2-5"><title>Synthetic Validation and Test Datasets</title><sec id="s2-5-1"><title>Rationale</title><p>Because the purpose of this study was to benchmark the core architecture (classifier+routing) and not to evaluate the deployment quality in a single hospital, we used expert-supervised synthetic text rather than deidentified patient records. This approach avoided privacy concerns and removed confounders related to the recording style.</p><p>Crucially, based on our preliminary internal validation, we observed that the semantic decision boundary between &#x201C;medical queries&#x201D; and &#x201C;casual conversation&#x201D; is highly sensitive to linguistic nuances. To rigorously evaluate the architectural feasibility of our novel routing logic without confounding variables, we established a study protocol to validate the system using standardized Japanese before introducing linguistic variations, such as dialects. This approach ensured that the baseline performance reflected architectural capability rather than linguistic noise.</p></sec><sec id="s2-5-2"><title>Validation Datasets</title><p>For each domain, we first generated candidate utterances using multiple frontier LLMs (Groq Compound-Beta; Llama-3.1-Nemotron-Ultra-253B-v1, NVIDIA Corporation; Google Gemma-3-27B-IT; Mixtral-8&#x00D7;7B, Mistral AI) to obtain a broad spectrum of Japanese sentences. To ensure transparency and reproducibility, the specific prompts used for data generation, including instructions to simulate diverse patient personas (eg, varying anxiety levels), are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Two independent domain experts per field (dentists or oral surgeons and physicians or endoscopists), who were not involved in the prompt engineering process, reviewed, corrected, and labeled each utterance as either a clinical question or a casual conversation. This process yielded 400 validation utterances per domain (200 clinical and 200 casual). These validation sets were used only to set the operating thresholds for the ST classifiers.</p></sec><sec id="s2-5-3"><title>Test Datasets</title><p>To ensure independence from the construction and threshold-tuning phases, we generated test candidates using different LLMs (Qwen3-35B-A2 and Claude [Sonnet 4]), followed by the same rigorous review and labeling processes conducted by independent specialists. The final test datasets consisted of 200 utterances per domain (100 clinical and 100 casual). These test sets were the only data used for the final between-model comparisons.</p><p>To establish the reliability of the ground-truth labels, interreviewer agreement for the binary classification task (clinical vs casual) was evaluated, demonstrating perfect consensus (Cohen &#x03BA;=1.0).</p></sec></sec><sec id="s2-6"><title>Ethical Considerations</title><p>The study protocol was approved by the Ethics Committee of Nagasaki University (approval: 23090101 and 24092701). As all datasets consisted of synthetic utterances generated under expert supervision and contained no personal data from real patients, informed consent was not required and no identifiable information was collected or stored. The study design adhered to ethical guidelines for medical research involving human participants.</p></sec><sec id="s2-7"><title>Models</title><sec id="s2-7-1"><title>ST-Based Classifiers (Primary)</title><p>We evaluated 4 ST encoders: a Japanese-specific SBERT (sonoisa/sentence-bert-base-ja-mean-tokens), a lightweight multilingual MiniLM (paraphrase-multilingual-MiniLM-L12-v2), a high-performance multilingual E5-large (intfloat/multilingual-e5-large), and its instruction-tuned variant, E5-large-instruct. All FAQ items in each domain were embedded once more. Incoming patient utterances were then embedded using the same model, and the cosine similarities were computed. The maximum similarity score for each utterance was compared with a domain-specific threshold, which was objectively determined in the validation set using the Youden index.</p></sec><sec id="s2-7-2"><title>LLM Baselines (Comparators)</title><p>For comparative context, we evaluated 2 representative frontier LLM services available between April 2025 and May 2025: ChatGPT (GPT-4o; OpenAI) and Gemini Advanced 2.5 Pro (Google). Both models were accessed through their web interfaces. They were provided with an identical prompt, which explicitly detailed the class definitions and required output formats, and were forced to output a single deterministic label (either clinical or casual). It is important to note that these baseline models were strictly evaluated as classifiers and were not used to generate synthetic test datasets.</p></sec><sec id="s2-7-3"><title>Local SLM Path</title><p>A locally hosted Japanese SLM (Llama-3.1-Swallow-8B-Instruct) was used to test the feasibility of on-device small talk generation for utterances classified as casual. The decoding parameters were constrained to avoid clinical advice.</p></sec></sec><sec id="s2-8"><title>Evaluation</title><sec id="s2-8-1"><title>Metrics</title><p>We calculated standard binary metrics: accuracy, precision, recall, <italic>F</italic><sub>1</sub>-score, specificity, balanced accuracy, and receiver operating characteristic&#x2013;AUC (area under the receiver operating characteristic curve; AUC only for ST models, as they output continuous scores; LLMs were treated as deterministic labelers).</p></sec><sec id="s2-8-2"><title>Sample Size and Statistics</title><p>Each domain had a total of 200 test utterances (100 clinical and 100 casual). For an expected accuracy of approximately 0.95, the 95% Wilson CI half-width was approximately 0.03. Prespecified pairwise comparisons (E5-large-instruct vs the other STs vs both LLMs) were tested using 2-sided McNemar tests; the Holm adjustment was applied for multiplicity. No missing data were found.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overall Performance</title><p>The primary outcome of this study demonstrates that a locally executable ST classifier can approach the classification quality of leading cloud-based LLMs for the safety-critical task of routing patient utterances to the appropriate department. Furthermore, end-to-end local execution proved to be both highly safe and sustainable: no clinically unsafe content was generated under the constrained decoding settings for casual conversations, and the nongenerative clinical pathway (FAQ retrieval) was approximately 75 times more energy-efficient than the generative small-talk pathway.</p></sec><sec id="s3-2"><title>Validation Performance</title><sec id="s3-2-1"><title>Tooth Extraction</title><p>In the tooth extraction validation set (N=400; 200 clinical, 200 casual), all ST classifiers achieved high discrimination when the thresholds were optimized using the Youden index. E5-large (AUC=0.990; <italic>F</italic><sub>1</sub>-score=0.965) and E5-large-instruct (AUC=0.989; <italic>F</italic><sub>1</sub>-score=0.971) showed the best balance between sensitivity and specificity. SBERT also performed well (AUC=0.961; <italic>F</italic><sub>1</sub>-score=0.930). Despite its small size, MiniLM remained usable (AUC=0.973) but showed a lower <italic>F</italic><sub>1</sub>-score (0.825), indicating more false negatives in this domain. These validation thresholds were frozen and reused for test evaluation.</p></sec><sec id="s3-2-2"><title>Gastroscopy</title><p>In the gastroscopy validation set (N=400), the performance was similarly strong.</p><p>E5-large-instruct achieved near-perfect classification (accuracy=0.988; recall=0.995; <italic>F</italic><sub>1</sub>-score=0.988). E5-large followed closely (accuracy=0.965; <italic>F</italic><sub>1</sub>-score=0.965). Lighter models such as SBERT (accuracy=0.928; <italic>F</italic><sub>1</sub>-score=0.928) and MiniLM (accuracy=0.925; <italic>F</italic><sub>1</sub>-score=0.926) still achieved acceptable performance, confirming that the classification task can be solved reliably across 2 clinically distinct domains. These thresholds were also applied in the test phase.</p></sec></sec><sec id="s3-3"><title>Test Performance on Independent Data</title><p>Among the ST models, E5-large-instruct achieved the best overall performance across both clinical domains, with a mean accuracy of 0.983, a recall of 0.985, and an <italic>F</italic><sub>1</sub>-score of 0.983, corresponding to 7 misclassifications out of 400 test utterances (<xref ref-type="table" rid="table1">Table 1</xref>). Discrimination remained high under distribution shift, with AUCs of 0.994 (tooth extraction) and 0.998 (gastroscopy), indicating a robust separation between clinical and casual inputs. The E5-large model also generalized well, showing only a marginal decrease relative to the instruction-tuned variant. In contrast, SBERT exhibited the largest degradation, particularly in the gastroscopy domain (accuracy=0.835), and produced the highest number of errors (51/400; <xref ref-type="table" rid="table1">Table 1</xref>). These results suggest that relying solely on a Japanese-specific ST model may be less robust when inputs reflect LLM-influenced phrasing or clinically structured language.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Test performance metrics for the tooth extraction and gastroscopy domains, along with overall performance.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model and domain</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Misclassification (FP<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>+FN<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>)<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="middle" colspan="5">SBERT<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><break/>Tooth extraction</td><td align="char" char="." valign="middle">0.910</td><td align="char" char="." valign="middle">0.950</td><td align="char" char="." valign="middle">0.962</td><td align="char" char="." valign="middle">18 (13+5)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">0.835</td><td align="char" char="." valign="middle">0.910</td><td align="char" char="." valign="middle">0.936</td><td align="char" char="." valign="middle">33 (24+9)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">0.873</td><td align="char" char="." valign="middle">0.930</td><td align="char" char="." valign="middle">0.949</td><td align="char" char="." valign="middle">51 (37+14)</td></tr><tr><td align="left" valign="middle" colspan="5">MiniLM</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><break/>Tooth extraction</td><td align="char" char="." valign="middle">0.915</td><td align="char" char="." valign="middle">0.900</td><td align="char" char="." valign="middle">0.972</td><td align="char" char="." valign="middle">17 (7+10)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">0.965</td><td align="char" char="." valign="middle">0.950</td><td align="char" char="." valign="middle">0.984</td><td align="char" char="." valign="middle">7 (2+5)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">0.940</td><td align="char" char="." valign="middle">0.925</td><td align="char" char="." valign="middle">0.978</td><td align="char" char="." valign="middle">24 (9+15)</td></tr><tr><td align="left" valign="middle" colspan="5">E5</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><break/>Tooth extraction</td><td align="char" char="." valign="middle">0.930</td><td align="char" char="." valign="middle">0.960</td><td align="char" char="." valign="middle">0.991</td><td align="char" char="." valign="middle">14 (10+4)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">0.965</td><td align="char" char="." valign="middle">0.940</td><td align="char" char="." valign="middle">0.994</td><td align="char" char="." valign="middle">7 (1+6)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">0.948</td><td align="char" char="." valign="middle">0.950</td><td align="char" char="." valign="middle">0.993</td><td align="char" char="." valign="middle">21 (11+10)</td></tr><tr><td align="left" valign="middle" colspan="5">E5-large-instruct</td></tr><tr><td align="left" valign="middle">&#x2003;&#x2003;<break/>&#x2003;&#x2003;Tooth extraction</td><td align="char" char="." valign="middle">0.980</td><td align="char" char="." valign="middle">0.990</td><td align="char" char="." valign="middle">0.994</td><td align="char" char="." valign="middle">4 (3+1)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">0.985</td><td align="char" char="." valign="middle">0.980</td><td align="char" char="." valign="middle">0.998</td><td align="char" char="." valign="middle">3 (1+2)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">0.983</td><td align="char" char="." valign="middle">0.985</td><td align="char" char="." valign="middle">0.996</td><td align="char" char="." valign="middle">7 (4+3)</td></tr><tr><td align="left" valign="middle" colspan="5">Gemini</td></tr><tr><td align="left" valign="middle">&#x2003;&#x2003;<break/>&#x2003;&#x2003;Tooth extraction</td><td align="char" char="." valign="middle">1.000</td><td align="char" char="." valign="middle">1.000</td><td align="left" valign="middle">N/A<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="char" char="." valign="middle">0</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">1.000</td><td align="char" char="." valign="middle">1.000</td><td align="left" valign="middle">N/A</td><td align="char" char="." valign="middle">0</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">1.000</td><td align="char" char="." valign="middle">1.000</td><td align="left" valign="middle">N/A</td><td align="char" char="." valign="middle">0</td></tr><tr><td align="left" valign="middle" colspan="5">ChatGPT (GPT-4o)</td></tr><tr><td align="left" valign="middle">&#x2003;&#x2003;<break/>&#x2003;&#x2003;Tooth extraction</td><td align="char" char="." valign="middle">1.000</td><td align="char" char="." valign="middle">1.000</td><td align="left" valign="middle">N/A</td><td align="char" char="." valign="middle">0</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gastroscopy</td><td align="char" char="." valign="middle">0.970</td><td align="char" char="." valign="middle">0.9434</td><td align="left" valign="middle">N/A</td><td align="char" char="." valign="middle">6 (6+0)</td></tr><tr><td align="left" valign="middle"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="char" char="." valign="middle">0.985</td><td align="char" char="." valign="middle">0.9717</td><td align="left" valign="middle">N/A</td><td align="char" char="." valign="middle">6 (6+0)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AUC: area under the curve.</p></fn><fn id="table1fn2"><p><sup>b</sup>FP: false positive.</p></fn><fn id="table1fn3"><p><sup>c</sup>FN: false negative.</p></fn><fn id="table1fn4"><p><sup>d</sup>The number of misclassifications represents the total number of FPs and FNs across both domains.</p></fn><fn id="table1fn5"><p><sup>e</sup>SBERT: sonoisa/sentence-bert-base-ja-mean-tokens.</p></fn><fn id="table1fn6"><p><sup>f</sup>Overall metrics for the sentence transformer models were calculated as the unweighted average of the domain-specific values.</p></fn><fn id="table1fn7"><p><sup>g</sup>N/A: not applicable. AUC was not computed for large language model (LLM) baselines because, as stated in the Metrics section, LLMs were treated as deterministic labelers outputting discrete labels rather than continuous scores, making AUC not applicable.</p></fn></table-wrap-foot></table-wrap><p>For frontier cloud LLMs, Gemini classified all the test items correctly (0 errors; <xref ref-type="table" rid="table1">Table 1</xref>). ChatGPT (GPT-4o) produced 6 errors, all of which were casual utterances misclassified as clinical (<xref ref-type="table" rid="table1">Table 1</xref>). Detailed performance metrics, including precision, specificity, and 95% CIs for all models, are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. Notably, the error counts of ChatGPT (GPT-4o) (6/400) and the locally executable E5-large-instruct (7/400) were comparable. The receiver operating characteristic curves and validation-optimized operating thresholds are shown for the tooth extraction (<xref ref-type="fig" rid="figure2">Figure 2</xref>) and gastroscopy (<xref ref-type="fig" rid="figure3">Figure 3</xref>) domains.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Receiver operating characteristic (ROC) curves with validation-based optimal thresholds (tooth extraction domain). AUC: area under the curve; SBERT: sonoisa/sentence-bert-base-ja-mean-tokens.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e89173_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Receiver operating characteristic (ROC) curves with validation-based optimal thresholds (gastroscopy domain). AUC: area under the curve; SBERT: sonoisa/sentence-bert-base-ja-mean-tokens.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e89173_fig03.png"/></fig></sec><sec id="s3-4"><title>Statistical Comparison of Models</title><p>To rigorously assess the performance differences between the key models, we conducted pairwise McNemar tests with continuity correction (<xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Pairwise McNemar test results with effect sizes and statistical power.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model 1</td><td align="left" valign="bottom">Model 2</td><td align="left" valign="bottom">b (model 1 only correct)</td><td align="left" valign="bottom">c (model 2 only correct)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Cohen <italic>h</italic></td><td align="left" valign="bottom">Power</td></tr></thead><tbody><tr><td align="left" valign="top">SBERT<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">MiniLM</td><td align="left" valign="top">18</td><td align="left" valign="top">45</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.8858</td><td align="left" valign="top">0.999</td></tr><tr><td align="left" valign="top">SBERT</td><td align="left" valign="top">E5</td><td align="left" valign="top">12</td><td align="left" valign="top">42</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.1781</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">SBERT</td><td align="left" valign="top">E5-large-instruct</td><td align="left" valign="top">4</td><td align="left" valign="top">48</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">2.0175</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">SBERT</td><td align="left" valign="top">ChatGPT (GPT-4o)</td><td align="left" valign="top">4</td><td align="left" valign="top">49</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">2.0284</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">SBERT</td><td align="left" valign="top">Gemini</td><td align="left" valign="top">0</td><td align="left" valign="top">51</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">3.1416</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">MiniLM</td><td align="left" valign="top">E5</td><td align="left" valign="top">13</td><td align="left" valign="top">16</td><td align="left" valign="top">.71</td><td align="left" valign="top">0.2073</td><td align="left" valign="top">0.124</td></tr><tr><td align="left" valign="top">MiniLM</td><td align="left" valign="top">E5-large-instruct</td><td align="left" valign="top">4</td><td align="left" valign="top">21</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.4955</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">MiniLM</td><td align="left" valign="top">ChatGPT (GPT-4o)</td><td align="left" valign="top">6</td><td align="left" valign="top">24</td><td align="left" valign="top">.002</td><td align="left" valign="top">1.287</td><td align="left" valign="top">0.999</td></tr><tr><td align="left" valign="top">MiniLM</td><td align="left" valign="top">Gemini</td><td align="left" valign="top">0</td><td align="left" valign="top">24</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">3.1416</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">E5-large</td><td align="left" valign="top">E5-large-instruct</td><td align="left" valign="top">1</td><td align="left" valign="top">15</td><td align="left" valign="top">.001</td><td align="left" valign="top">2.1309</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">E5-large</td><td align="left" valign="top">ChatGPT (GPT-4o)</td><td align="left" valign="top">6</td><td align="left" valign="top">21</td><td align="left" valign="top">.007</td><td align="left" valign="top">1.1781</td><td align="left" valign="top">0.991</td></tr><tr><td align="left" valign="top">E5-large</td><td align="left" valign="top">Gemini</td><td align="left" valign="top">0</td><td align="left" valign="top">21</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">3.1416</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">E5-large-instruct</td><td align="left" valign="top">ChatGPT (GPT-4o)</td><td align="left" valign="top">6</td><td align="left" valign="top">7</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">0.154</td><td align="left" valign="top">0.068</td></tr><tr><td align="left" valign="top">E5-large-instruct</td><td align="left" valign="top">Gemini</td><td align="left" valign="top">0</td><td align="left" valign="top">7</td><td align="left" valign="top">.02</td><td align="left" valign="top">3.1416</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">ChatGPT (GPT-4o)</td><td align="left" valign="top">Gemini</td><td align="left" valign="top">0</td><td align="left" valign="top">6</td><td align="left" valign="top">.04</td><td align="left" valign="top">3.1416</td><td align="left" valign="top">1</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>SBERT: sonoisa/sentence-bert-base-ja-mean-tokens.</p></fn></table-wrap-foot></table-wrap><p>Compared with other ST models, E5-large-instruct demonstrated statistically significant superiority over SBERT (<italic>P</italic>&#x003C;.001) and MiniLM (<italic>P</italic>&#x003C;.001) with high statistical power (&#x003E;0.999). Overall, these statistical results indicate that the local E5-large-instruct model achieves a classification performance comparable to that of the evaluated frontier cloud-based LLMs for this specific routing task.</p><p><xref ref-type="table" rid="table2">Table 2</xref> presents the results of pairwise comparisons between models using the McNemar test with continuity correction. For each model pair, the number of discordant cases, where only 1 model made a correct prediction, is reported as b (model 1 only correct) and c (model 2 only correct). The associated <italic>P</italic>-value indicates whether the difference in performance is statistically significant. Cohen <italic>h</italic> provides the effect size, and the post hoc power (assuming <italic>&#x03B1;</italic>=.05) reflects the likelihood of detecting such an effect with the observed sample size. Tests with low power (eg, power &#x003C;0.8) may fail to detect true differences even if they exist (eg, comparisons involving MiniLM vs E5 and E5-large-instruct vs ChatGPT [GPT-4o]). In contrast, model pairs involving SBERT or Gemini consistently showed large effect sizes (Cohen <italic>h</italic>&#x003E;0.8) and sufficient power (power&#x2248;1) indicate robust and meaningful performance differences.</p></sec><sec id="s3-5"><title>Local Small-Talk Generation</title><p>To check whether the nonclinical branch can also be kept on the device, we generated responses to 200 casual utterances with a local 8B Japanese SLM (Llama-3.1-Swallow-8B-Instruct-v0.3). All prompts produced coherent, context-appropriate replies, and no clinically unsafe content was observed under constrained decoding settings. The representative outputs are presented in <xref ref-type="table" rid="table3">Table 3</xref>.</p><p>Taken together with the classification results, this confirms that both clinical query filtering and nonclinical response generation can be run end-to-end on a single local graphics processing unit (GPU).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Examples of casual conversation responses generated by Swallow-8B.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Patient&#x2019;s casual utterance</td><td align="left" valign="bottom">Swallow-8B&#x2019;s response</td></tr></thead><tbody><tr><td align="left" valign="top">There is a lovely shop nearby.</td><td align="left" valign="top">Oh really? What type of shop is it?</td></tr><tr><td align="left" valign="top">A colleague at work has just gotten married.</td><td align="left" valign="top">Congratulations! That&#x2019;s wonderful. Best wishes to them!</td></tr><tr><td align="left" valign="top">There is a great bakery in my neighborhood.</td><td align="left" valign="top">That sounds great! I have been looking for a new bakery. Do you have any recommendations for this?</td></tr><tr><td align="left" valign="top">The weather was pleasant that day.</td><td align="left" valign="top">Yes, the weather is beautiful today. A perfect day to go out. Do you have any plans?</td></tr><tr><td align="left" valign="top">I recently watched a funny variety show on TV.</td><td align="left" valign="top">I have been watching it too! That comedian&#x2019;s act really made me laugh. Do you have any recommendations for shows?</td></tr><tr><td align="left" valign="top">I recently watched an interesting movie at the theater.</td><td align="left" valign="top">That&#x2019;s great! What type of movie was it?</td></tr></tbody></table></table-wrap></sec><sec id="s3-6"><title>Sustainability and Energy Use</title><p>We quantified the on-device energy use by sampling the GPU board power at 1 Hz using the NVIDIA Management Library and integrating the trace, excluding the one-time model load. For the LENOHA ST classifier (nongenerative FAQ pathway), the energy consumption was 0.445 Wh per 200 inputs, corresponding to approximately 2.23 mWh per request (latency&#x2248;0.10 s; video RAM&#x2248;2.2 GiB average, &#x2248;2.5 GiB peak). In contrast, the local SLM pathway (Swallow-8B) consumed 33.65 Wh per 200 replies, corresponding to approximately 168 mWh per reply (latency&#x2248;8.51 s; video RAM&#x2248;13.3 GiB average, &#x2248;14.0 GiB peak).</p><p>Thus, the nongenerative clinical pathway is approximately 75 times more energy efficient than the generative small-talk pathway, despite operating on the same hardware (<xref ref-type="table" rid="table4">Table 4</xref>). We also report, for context, a comprehensive cloud figure for a single text prompt, which is approximately 240 mWh (including idle power, CPU, RAM, and power usage efficiency) [<xref ref-type="bibr" rid="ref29">29</xref>]. Because the measurement methods differ, these values should not be interpreted as a strict like-for-like comparison; however, they provide objective context indicating that a locally routed, nongenerative path is substantially more energy-efficient.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Energy and latency comparison of local nongenerative, local generative, and cloud-generative settings.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">System</td><td align="left" valign="bottom">Parameters (billion)</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Environment</td><td align="left" valign="bottom">Measurement method</td><td align="left" valign="bottom">Energy (mWh/request)</td><td align="left" valign="bottom">Latency (s)</td></tr></thead><tbody><tr><td align="left" valign="top">LENOHA<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> ST<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> classifier (nongenerative, FAQ)<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.56</td><td align="left" valign="top">Nongenerative</td><td align="left" valign="top">Local</td><td align="left" valign="top">Device-integral (GPU<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> power integration)</td><td align="left" valign="top">2.23</td><td align="left" valign="top">0.1008</td></tr><tr><td align="left" valign="top">LENOHA SLM<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup> generation (small-talk, 8B)</td><td align="left" valign="top">8</td><td align="left" valign="top">Generative</td><td align="left" valign="top">Local</td><td align="left" valign="top">Device-integral (GPU power integration)</td><td align="left" valign="top">168</td><td align="left" valign="top">8.5147</td></tr><tr><td align="left" valign="top">Google Cloud: median Gemini text prompt [<xref ref-type="bibr" rid="ref26">26</xref>]<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table4fn7">g</xref></sup></td><td align="left" valign="top">Generative</td><td align="left" valign="top">Cloud</td><td align="left" valign="top">Comprehensive (idle/CPU/RAM/data-center PUE<sup><xref ref-type="table-fn" rid="table4fn8">h</xref></sup> included)</td><td align="left" valign="top">240</td><td align="left" valign="top">N/A</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>LENOHA: Low Energy, No Hallucination, Leave No One Behind Architecture.</p></fn><fn id="table4fn2"><p><sup>b</sup>ST: sentence transformer.</p></fn><fn id="table4fn3"><p><sup>c</sup>FAQ: frequently asked question.</p></fn><fn id="table4fn4"><p><sup>d</sup>GPU: graphics processing unit.</p></fn><fn id="table4fn5"><p><sup>e</sup>SLM: small language model.</p></fn><fn id="table4fn6"><p><sup>f</sup>The cloud value reflects 0.24 Wh (240 mWh) per text prompt (comprehensive, including idle/CPU/RAM/data-center PUE&#x2248;1.09) [<xref ref-type="bibr" rid="ref29">29</xref>]. Because the measurement methods differ between local device-integral and cloud-comprehensive reports, these values provide a contextual reference and should not be overinterpreted as a strict like-for-like comparison.</p></fn><fn id="table4fn7"><p><sup>g</sup>N/A: not applicable.</p></fn><fn id="table4fn8"><p><sup>h</sup>PUE: power usage effectiveness.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-7"><title>Failure Analysis</title><p><xref ref-type="table" rid="table5">Table 5</xref> lists examples of the E5-large-instruct misclassifications. Qualitative analysis of these errors revealed 2 distinct patterns. First, false positives occurred when casual scheduling or holiday inquiries (eg, &#x201C;Are you available at the end of this month?&#x201D;) were misclassified as clinical questions. This pattern results in sending nonclinical utterances to the deterministic FAQ pathway, which, while reducing conversational naturalness, remains acceptable under the safety-first architecture.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Examples of misclassified utterances by E5-large-instruct.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Patient utterance (English translation)</td><td align="left" valign="bottom">Ground truth</td><td align="left" valign="bottom">E5-instruct prediction</td></tr></thead><tbody><tr><td align="left" valign="top">Are there any medical conditions for which I should be cautious?</td><td align="left" valign="top">Clinical question</td><td align="left" valign="top">Casual</td></tr><tr><td align="left" valign="top">Should I remove my dentures?</td><td align="left" valign="top">Clinical question</td><td align="left" valign="top">Casual</td></tr><tr><td align="left" valign="top">What should I do if I experience numbness in my tongue or lips?</td><td align="left" valign="top">Clinical question</td><td align="left" valign="top">Casual</td></tr><tr><td align="left" valign="top">Do you have any plans for the upcoming holidays?</td><td align="left" valign="top">Casual</td><td align="left" valign="top">Clinical question</td></tr><tr><td align="left" valign="top">Are you available at the end of this month?</td><td align="left" valign="top">Casual</td><td align="left" valign="top">Clinical question</td></tr><tr><td align="left" valign="top">Did you take time off during the New Year&#x2019;s holidays?</td><td align="left" valign="top">Casual</td><td align="left" valign="top">Clinical question</td></tr></tbody></table></table-wrap><p>Conversely, false negatives occurred when specific clinical queries (eg, inquiries about postoperative numbness, dentures, or underlying medical conditions) were misclassified as casual conversation. Routing these medical inquiries to the generative small-talk pathway bypasses the intended structural safeguards of the system. Although the overall error rate was extremely low, these specific misclassifications highlighted critical edge cases in which semantic similarities blurred the decision boundary. This indicates that further optimization of the cosine similarity threshold or the implementation of a secondary keyword-based safety net is necessary to ensure zero leakage of safety-critical questions in future deployments.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The principal finding of this study is that a locally executable ST classifier can achieve near-perfect accuracy in separating clinical queries from casual conversations, matching the performance of frontier cloud LLMs while using approximately 1/75th of the energy.</p><p>By evaluating this safety-first dialog architecture (LENOHA) across 2 clinically distinct preprocedural settings&#x2014;oral and maxillofacial surgery, and upper gastrointestinal endoscopy&#x2014;we showed that lightweight, locally executable models can achieve exceptional discrimination performance. This finding is important because it demonstrates that high-precision filtering of patient utterances does not necessarily require large, energy-intensive models or external data transmission.</p></sec><sec id="s4-2"><title>Architectural and Sustainability Implications</title><p>The second contribution is architectural. Many current health care chatbot designs attempt to &#x201C;do everything&#x201D; using a single generative model. However, as demonstrated by Safrai and Azaria [<xref ref-type="bibr" rid="ref30">30</xref>], mixing casual dialog with clinical queries in a single prompt can lead to &#x201C;context contamination,&#x201D; severely degrading the model&#x2019;s medical reasoning. By contrast, LENOHA enforces a hard separation: if the input is clinical, the system does not generate a response; instead, it returns a canonical, clinician-authorized FAQ answer. Only low-risk, nonclinical inputs are routed to the SLM.</p><p>This path is conceptually simple but highly aligned with implementation checklists, such as those proposed by Morley et al [<xref ref-type="bibr" rid="ref19">19</xref>], and recently established digital health ethics frameworks, such as SAFE-AI [<xref ref-type="bibr" rid="ref26">26</xref>], because it keeps epistemic uncertainty low, preserves institutional control over the content, and makes auditability straightforward. To address the risks of context contamination, there is a growing recognition that structural safeguards are required. A highly relevant emerging standard is the MCP, which mitigates such vulnerabilities by architecturally compartmentalizing the input data [<xref ref-type="bibr" rid="ref15">15</xref>]. Unlike standard open-ended prompting, MCP strictly formats and isolates the context provided to the model, reducing the likelihood that extraneous details or casual &#x201C;small talk&#x201D; will be misinterpreted as clinical facts [<xref ref-type="bibr" rid="ref16">16</xref>]. Our dual-pathway design shares this fundamental architectural philosophy. By enforcing a strict separation between clinical retrieval and casual generation, our approach illustrates how rigid architectural constraints&#x2014;rather than relying solely on model training or prompt engineering&#x2014;can play a central role in maintaining clinical safety and preventing reasoning degradation.</p><p>Energy and sustainability considerations further support this design. Our measurements showed that the nongenerative, FAQ-returning path consumed approximately 2.23 mWh per request, whereas local small-talk generation with an 8B model cost approximately 168 mWh per request. By routing safety-critical clinical queries to a deterministic, low-energy module and reserving high-energy generative inference only for low-risk, nonclinical small talk, our hybrid architecture substantially reduces operational carbon debt. Based on the FY2023 Japanese national average emission factor (0.423 kg-CO&#x2082;/kWh), this corresponds to roughly 0.94 mg-CO&#x2082; vs 71.1 mg-CO&#x2082; per interaction&#x2014;a 75-fold reduction in the per-request carbon footprint [<xref ref-type="bibr" rid="ref31">31</xref>]. This design directly addresses the calls for more sustainable natural language processing practices and helps ensure that the digital transformation of health care does not come at the cost of environmental integrity.</p><p>In other words, most of the energy cost lies in the &#x201C;nice-to-have&#x201D; conversational layer, not in the safety-critical layer. An architecture that allows clinics to serve 100% of clinical queries on the low-energy path and reserves the high-energy path only for casual rapport-building is, therefore, substantially more scalable for facilities with limited GPUs, limited power budgets, or intermittent connectivity. This is particularly relevant for regions with many inhabited remote islands and persistent difficulties in accessing specialists; in such settings, a locally executable, bandwidth-independent system is not only privacy-preserving but also feasible.</p><p>Our research base in Nagasaki Prefecture includes 51 inhabited islands with approximately 110,000 residents, representing one of Japan&#x2019;s highest concentrations of remote islands. Such regions exemplify the urgent need to bridge health care access gaps, as many residents face substantial difficulties in accessing specialized medical services. In underserved island regions where the specialist workforce and reliable connectivity are limited, such an architecture could support clinicians by offloading routine preprocedural communication while preserving privacy and feasibility, thereby helping to reduce the working time burden on medical specialists, especially because the system can operate continuously throughout the day.</p><p>A recent study by Wewetzer et al [<xref ref-type="bibr" rid="ref32">32</xref>] highlighted that the implementation of AI-supported health care in rural areas faces significant barriers, primarily due to technological limitations and a pronounced lack of patient trust in AI compared to urban populations.</p></sec><sec id="s4-3"><title>Limitations</title><p>Several limitations should be considered when interpreting the findings of this study.</p><p>First, this study deliberately adopted a text-based interface and did not integrate ASR. Our preliminary tests indicated that the current ASR error rates for Japanese clinical dialog could obscure the performance of the core classification architecture; therefore, a rigorous ASR-inclusive evaluation is deferred to future work. This choice improves internal validity but limits immediate use in fully speech-based encounters, particularly in dialectal or low-audio-quality settings. Fairness and safety considerations in ASR for health care were major factors in this decision.</p><p>Second, we focused on the performance of the input-filtering (classification) module rather than an end-to-end user experience. This reflects a safety-first philosophy: before assessing rapport, empathy, or patient satisfaction, it is ethically necessary to demonstrate that the system can reliably separate high-risk clinical questions from low-risk casual conversations. Evaluating whether the chatbot can support complex social or emotional roles was beyond the scope of this foundational work and will require qualitative, context-sensitive studies. Our classifier relies on externally developed models for sentence embeddings and thus inevitably inherits any representational biases encoded in the model&#x2019;s pretraining data. In our architecture, however, these biases can only affect the routing decision (ie, whether an utterance is classified as clinical or casual) and cannot alter the content of clinical advice itself, which is strictly constrained to clinician-authored FAQ texts. We partly mitigated setting-specific overfitting by validating and testing the classifier across 2 distinct preprocedural domains; however, we did not perform a formal audit of E5-large&#x2019;s subgroup biases in Japanese patient speech, which remains an important area for future work.</p><p>Third, this study used standard Japanese to establish a robust performance baseline. Our pilot experiments indicated that optimizing the cosine similarity thresholds for the ST requires precise tuning, and introducing linguistic noise (eg, dialects or slang) at this initial stage could obscure the true performance of the routing architecture. Having now demonstrated that the system achieves high separation accuracy (AUC &#x003E;0.99) under controlled conditions, future work will focus on expanding the system&#x2019;s robustness to diverse patient personas, including those with low health literacy or regional dialects.</p><p>From a life cycle perspective, our study is best understood as an early-phase study. In line with step-wise implementation models, such as those of van de Sande et al [<xref ref-type="bibr" rid="ref27">27</xref>], this study primarily covers phase 1 (AI model development) and phase 2 (assessment of AI performance and reliability), rather than phase 3 (clinical testing of AI with actual patients). This is also consistent with broader AI life cycle concepts proposed by Kuziemsky et al [<xref ref-type="bibr" rid="ref28">28</xref>], which emphasize that development and validation phases should precede in-workflow clinical evaluation. Furthermore, this approach intentionally aligns with the latest core requirements for ethical AI implementation proposed by Morley et al [<xref ref-type="bibr" rid="ref19">19</xref>], which mandate the establishment of &#x201C;epistemic certainty&#x201D; and &#x201C;validated outcomes&#x201D; before real-world deployment. Accordingly, we used expert-supervised synthetic utterances to isolate routing performance and energy use under controlled conditions, and we do not yet claim full clinical readiness; a phase-3, patient-facing evaluation in real clinical workflows remains an important next step.</p></sec><sec id="s4-4"><title>Conclusions</title><p>The main insight from this study is that the performance of clinical AI is not solely a function of computational scale. Human clinical expertise&#x2014;practice-based experiential knowledge that is often underrepresented in web-derived training corpora&#x2014;remains central to safety and reliability. Consistent with recent studies on clinical prediction, our findings illustrate that comparable task performance can be obtained without relying on frontier LLMs when such expertise is carefully encoded and constrained, highlighting a knowledge-centric, complementary path alongside scale-driven model development [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>Taken together, these findings support a dual-pathway design for medical dialog systems: (1) a generation-free clinical pathway that limits medical queries to auditable retrieval (eg, verbatim FAQ answers) and rejects or defers unverifiable inputs, and (2) a dedicated small-talk pathway that delivers brief, structured, empathic interactions through a lightweight generative model.</p></sec></sec></body><back><ack><p>The authors gratefully acknowledge their colleagues for their contributions to the creation of the FAQ dataset used in this study. The authors declare the use of generative AI (GAI) in the research and writing processes. According to the GAIDeT taxonomy (2025), the following tasks were delegated to GAI tools under complete human supervision: code optimization and translation. The GAI tool used was ChatGPT (GPT-5.1). Responsibility for the final manuscript lies entirely with the authors. The GAI tools were not listed as authors and did not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the 8020 Promotion Foundation: 24-6-15 (to YM).</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed in this study, including the curated FAQ database specifically designed for the Japanese medical context, are not publicly available due to ethical considerations and unavailability reasons. The code used in this study is available in reference [<xref ref-type="bibr" rid="ref35">35</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: MS, YM, AY, MY</p><p>Formal analysis: MS, YM</p><p>Investigation: SN, MO, HT, TK</p><p>Methodology: MS, YM, AY, MY</p><p>Resources: SN, MO, HT, TK, MS, YM</p><p>Software: MS, AY, MY</p><p>Supervision: YM</p><p>Validation: SN, MO, HT, TK</p><p>Writing &#x2013; original draft: MS, YM</p><p>Writing &#x2013; review &#x0026; editing: MS, SN, MO, HT, TK, MY, AY, YM</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ASR</term><def><p>automatic speech recognition</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb3">FAQ</term><def><p>frequently asked question</p></def></def-item><def-item><term id="abb4">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb5">LENOHA</term><def><p> Low Energy, No Hallucination, Leave No One Behind Architecture</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">MCP</term><def><p>Model Context Protocol</p></def></def-item><def-item><term id="abb8">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb9">SAFE-AI</term><def><p>Scalable Agile Framework for Execution in AI</p></def></def-item><def-item><term id="abb10">SBERT</term><def><p> sonoisa/sentence-bert-base-ja-mean-tokens</p></def></def-item><def-item><term id="abb11">SLM</term><def><p>small language model</p></def></def-item><def-item><term id="abb12">ST</term><def><p>sentence transformer</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Armfield</surname><given-names>JM</given-names> </name></person-group><article-title>The extent and nature of dental fear and phobia in Australia</article-title><source>Aust Dent J</source><year>2010</year><month>12</month><volume>55</volume><issue>4</issue><fpage>368</fpage><lpage>377</lpage><pub-id pub-id-type="doi">10.1111/j.1834-7819.2010.01256.x</pub-id><pub-id pub-id-type="medline">21174906</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Earl</surname><given-names>P</given-names> </name></person-group><article-title>Patients&#x2019; anxieties with third molar surgery</article-title><source>Br J Oral Maxillofac Surg</source><year>1994</year><month>10</month><volume>32</volume><issue>5</issue><fpage>293</fpage><lpage>297</lpage><pub-id pub-id-type="doi">10.1016/0266-4356(94)90049-3</pub-id><pub-id pub-id-type="medline">7999736</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schenker</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fernandez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sudore</surname><given-names>R</given-names> </name><name name-style="western"><surname>Schillinger</surname><given-names>D</given-names> </name></person-group><article-title>Interventions to improve patient comprehension in informed consent for medical and surgical procedures: a systematic review</article-title><source>Med Decis Making</source><year>2011</year><volume>31</volume><issue>1</issue><fpage>151</fpage><lpage>173</lpage><pub-id pub-id-type="doi">10.1177/0272989X10364247</pub-id><pub-id pub-id-type="medline">20357225</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kiernan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fahey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Guraya</surname><given-names>SS</given-names> </name><etal/></person-group><article-title>Digital technology in informed consent for surgery: systematic review</article-title><source>BJS Open</source><year>2023</year><month>01</month><day>6</day><volume>7</volume><issue>1</issue><fpage>zrac159</fpage><pub-id pub-id-type="doi">10.1093/bjsopen/zrac159</pub-id><pub-id pub-id-type="medline">36694387</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Numata</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Matsumoto</surname><given-names>M</given-names> </name></person-group><article-title>Labor shortage of physicians in rural areas and surgical specialties caused by Work Style Reform Policies of the Japanese government: a quantitative simulation analysis</article-title><source>J Rural Med</source><year>2024</year><month>07</month><volume>19</volume><issue>3</issue><fpage>166</fpage><lpage>173</lpage><pub-id pub-id-type="doi">10.2185/jrm.2023-047</pub-id><pub-id pub-id-type="medline">38975037</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haack</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fischer</surname><given-names>ND</given-names> </name><name name-style="western"><surname>Frey</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Digital informed consent for urological surgery: randomized controlled study comparing multimedia-supported vs. traditional paper-based informed consent concerning satisfaction, anxiety, information gain and time efficiency</article-title><source>Prostate Cancer Prostatic Dis</source><year>2024</year><month>12</month><volume>27</volume><issue>4</issue><fpage>715</fpage><lpage>719</lpage><pub-id pub-id-type="doi">10.1038/s41391-023-00737-4</pub-id><pub-id pub-id-type="medline">37925488</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Houten</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hussain</surname><given-names>MI</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>AP</given-names> </name><etal/></person-group><article-title>Digital versus paper-based consent from the UK NHS perspective: a micro-costing analysis</article-title><source>Pharmacoecon Open</source><year>2025</year><month>01</month><volume>9</volume><issue>1</issue><fpage>27</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1007/s41669-024-00536-0</pub-id><pub-id pub-id-type="medline">39499439</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sallam</surname><given-names>M</given-names> </name></person-group><article-title>ChatGPT utility in healthcare education, research, and practice: systematic review on the promising perspectives and valid concerns</article-title><source>Healthcare (Basel)</source><year>2023</year><month>03</month><day>19</day><volume>11</volume><issue>6</issue><fpage>887</fpage><pub-id pub-id-type="doi">10.3390/healthcare11060887</pub-id><pub-id pub-id-type="medline">36981544</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaessler</surname><given-names>J</given-names> </name><name name-style="western"><surname>Remschmidt</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jopp</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Arefnia</surname><given-names>B</given-names> </name><name name-style="western"><surname>Franke</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rieder</surname><given-names>M</given-names> </name></person-group><article-title>Quality of conventional versus artificial intelligence oral surgery consent forms: comparative analysis</article-title><source>J Med Internet Res</source><year>2026</year><month>01</month><day>5</day><volume>28</volume><fpage>e59851</fpage><pub-id pub-id-type="doi">10.2196/59851</pub-id><pub-id pub-id-type="medline">41490384</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Mitigating the risk of health inequity exacerbated by large language models</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>4</day><volume>8</volume><issue>1</issue><fpage>246</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01576-4</pub-id><pub-id pub-id-type="medline">40319154</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aguiar de Sousa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Costa</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Almeida Figueiredo</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Camargos</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Ribeiro</surname><given-names>BC</given-names> </name><name name-style="western"><surname>Alves E Silva</surname><given-names>MRM</given-names> </name></person-group><article-title>Is ChatGPT a reliable source of scientific information regarding third-molar surgery?</article-title><source>J Am Dent Assoc</source><year>2024</year><month>03</month><volume>155</volume><issue>3</issue><fpage>227</fpage><lpage>232</lpage><pub-id pub-id-type="doi">10.1016/j.adaj.2023.11.004</pub-id><pub-id pub-id-type="medline">38206257</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jain</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kankanhalli</surname><given-names>M</given-names> </name></person-group><article-title>Hallucination is inevitable: an innate limitation of large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 22, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.11817</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ng</surname><given-names>KKY</given-names> </name><name name-style="western"><surname>Matsuba</surname><given-names>I</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>PC</given-names> </name></person-group><article-title>RAG in health care: a novel framework for improving communication and decision-making by addressing LLM limitations</article-title><source>NEJM AI</source><year>2025</year><month>01</month><volume>2</volume><issue>1</issue><fpage>AIra2400380</fpage><pub-id pub-id-type="doi">10.1056/AIra2400380</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>NF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hewitt</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Lost in the middle: how language models use long contexts</article-title><source>Trans Assoc Comput Linguist</source><year>2024</year><month>02</month><day>23</day><volume>12</volume><fpage>157</fpage><lpage>173</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00638</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Introducing the model context protocol</article-title><source>Anthropic</source><access-date>2026-02-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.anthropic.com/news/model-context-protocol">https://www.anthropic.com/news/model-context-protocol</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name></person-group><article-title>Model context protocol (MCP): landscape, security threats, and future research directions</article-title><source>ACM Trans Softw Eng Methodol</source><year>2026</year><pub-id pub-id-type="doi">10.1145/3796519</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Avila</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ilbay</surname><given-names>D</given-names> </name><name name-style="western"><surname>Rivera</surname><given-names>D</given-names> </name></person-group><article-title>Human&#x2013;AI teaming in structural analysis: a model context protocol approach for explainable and accurate generative AI</article-title><source>Buildings</source><year>2025</year><month>09</month><volume>15</volume><issue>17</issue><fpage>3190</fpage><pub-id pub-id-type="doi">10.3390/buildings15173190</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benavent</surname><given-names>D</given-names> </name><name name-style="western"><surname>Venerito</surname><given-names>V</given-names> </name><name name-style="western"><surname>Michelena</surname><given-names>X</given-names> </name></person-group><article-title>RAGing ahead in rheumatology: new language model architectures to tame artificial intelligence</article-title><source>Ther Adv Musculoskelet Dis</source><year>2025</year><volume>17</volume><fpage>1759720X251331529</fpage><pub-id pub-id-type="doi">10.1177/1759720X251331529</pub-id><pub-id pub-id-type="medline">40292012</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hine</surname><given-names>E</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Global health in the age of AI: charting a course for ethical implementation and societal benefit</article-title><source>Minds Mach</source><volume>35</volume><issue>3</issue><fpage>1</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1007/s11023-025-09730-3</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ueda</surname><given-names>D</given-names> </name><name name-style="western"><surname>Walston</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Fujita</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Climate change and artificial intelligence in healthcare: review and recommendations towards a sustainable future</article-title><source>Diagn Interv Imaging</source><year>2024</year><month>11</month><volume>105</volume><issue>11</issue><fpage>453</fpage><lpage>459</lpage><pub-id pub-id-type="doi">10.1016/j.diii.2024.06.002</pub-id><pub-id pub-id-type="medline">38918123</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Adedeji</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sanni</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ayodele</surname><given-names>E</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Olatunji</surname><given-names>T</given-names> </name></person-group><article-title>The multicultural medical assistant: can LLMs improve medical ASR errors across borders?</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.15310</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Dichristofano</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shuster</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chandra</surname><given-names>S</given-names> </name><name name-style="western"><surname>Patwari</surname><given-names>N</given-names> </name></person-group><article-title>Global performance disparities between English-language accents in automatic speech recognition</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 1, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2208.01157</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeanquartier</surname><given-names>F</given-names> </name><name name-style="western"><surname>Jean-Quartier</surname><given-names>C</given-names> </name><name name-style="western"><surname>Rieder</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Assessing the carbon footprint of language models: towards sustainability in AI</article-title><source>Resour Conserv Recycl</source><year>2026</year><month>02</month><volume>226</volume><fpage>108670</fpage><pub-id pub-id-type="doi">10.1016/j.resconrec.2025.108670</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ning</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Teixayavong</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Generative artificial intelligence and ethical considerations in health care: a scoping review and ethics checklist</article-title><source>Lancet Digit Health</source><year>2024</year><month>11</month><volume>6</volume><issue>11</issue><fpage>e848</fpage><lpage>e856</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00143-2</pub-id><pub-id pub-id-type="medline">39294061</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nemteanu</surname><given-names>I</given-names> </name><name name-style="western"><surname>Mancebo</surname><given-names>A</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Joe</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pettine</surname><given-names>WW</given-names> </name></person-group><article-title>Scalable agile framework for execution in AI for medical AI ethics policy design in small- and medium-sized enterprises</article-title><source>J Med Internet Res</source><year>2026</year><month>02</month><day>25</day><volume>28</volume><fpage>e80028</fpage><pub-id pub-id-type="doi">10.2196/80028</pub-id><pub-id pub-id-type="medline">41740154</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van de Sande</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Genderen</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Smit</surname><given-names>JM</given-names> </name><etal/></person-group><article-title>Developing, implementing and governing artificial intelligence in medicine: a step-by-step approach to prevent an artificial intelligence winter</article-title><source>BMJ Health Care Inform</source><year>2022</year><month>02</month><volume>29</volume><issue>1</issue><fpage>e100495</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2021-100495</pub-id><pub-id pub-id-type="medline">35185012</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kuziemsky</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Chrimes</surname><given-names>D</given-names> </name><name name-style="western"><surname>Minshall</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mannerow</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lau</surname><given-names>F</given-names> </name></person-group><article-title>AI quality standards in health care: rapid umbrella review</article-title><source>J Med Internet Res</source><year>2024</year><month>05</month><day>22</day><volume>26</volume><fpage>e54705</fpage><pub-id pub-id-type="doi">10.2196/54705</pub-id><pub-id pub-id-type="medline">38776538</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Elsworth</surname><given-names>C</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Patterson</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Measuring the environmental impact of delivering AI at google scale</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 21, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.15734</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Safrai</surname><given-names>M</given-names> </name><name name-style="western"><surname>Azaria</surname><given-names>A</given-names> </name></person-group><article-title>Does small talk with a medical provider affect ChatGPT&#x2019;s medical counsel? Performance of ChatGPT on USMLE with and without distractions</article-title><source>PLoS One</source><year>2024</year><volume>19</volume><issue>4</issue><fpage>e0302217</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0302217</pub-id><pub-id pub-id-type="medline">38687696</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="report"><person-group person-group-type="author"><collab>Cabinet Secretariat, Financial Services Agency, Ministry of Finance, Ministry of Economy, Trade and Industry, Ministry of the Environment</collab></person-group><article-title>Japan climate transition bonds allocation and impact report for FY2023 issuance/allocation report for FY2024 issuance</article-title><year>2026</year><access-date>2026-05-06</access-date><publisher-name>Ministry of Economy, Trade and Industry (METI)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.meti.go.jp/policy/energy_environment/global_warming/transition/climate.transition.bond.allocation.impact.report.eng.pdf">https://www.meti.go.jp/policy/energy_environment/global_warming/transition/climate.transition.bond.allocation.impact.report.eng.pdf</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wewetzer</surname><given-names>L</given-names> </name><name name-style="western"><surname>Goetz</surname><given-names>K</given-names> </name><name name-style="western"><surname>Freischmidt</surname><given-names>S</given-names> </name><name name-style="western"><surname>Steinhauser</surname><given-names>J</given-names> </name></person-group><article-title>Trust in AI-supported screening in general practice among urban and rural citizens: cross-sectional study</article-title><source>JMIR Med Inform</source><year>2026</year><month>02</month><day>12</day><volume>14</volume><fpage>e69777</fpage><pub-id pub-id-type="doi">10.2196/69777</pub-id><pub-id pub-id-type="medline">41678715</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>ClinicalBench: can LLMs beat traditional ML models in clinical prediction?</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 10, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2411.06469</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Small language models learn enhanced reasoning skills from medical textbooks</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>2</day><volume>8</volume><issue>1</issue><fpage>240</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01653-8</pub-id><pub-id pub-id-type="medline">40316765</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="web"><article-title>motokinaru/LENOHA-medical-dialog</article-title><source>GitHub</source><access-date>2026-07-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/motokinaru/LENOHA-medical-dialog">https://github.com/motokinaru/LENOHA-medical-dialog</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Clinical schema and annotation guidelines.</p><media xlink:href="medinform_v14i1e89173_app1.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Prompts used for synthetic data generation.</p><media xlink:href="medinform_v14i1e89173_app2.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Complete test performance metrics, including 95% CIs.</p><media xlink:href="medinform_v14i1e89173_app3.xlsx" xlink:title="XLSX File, 10 KB"/></supplementary-material></app-group></back></article>