<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e86145</article-id><article-id pub-id-type="doi">10.2196/86145</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Models for Endodontic Symptom Assessment and Treatment Planning Using Image-Free Clinical Records: Comparative Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Seo</surname><given-names>Dahyun</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Cheong</surname><given-names>Jieun</given-names></name><degrees>DDS, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Choi</surname><given-names>Yiseul</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Shin</surname><given-names>Yooseok</given-names></name><degrees>DDS, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Park</surname><given-names>Wonse</given-names></name><degrees>DDS, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Advanced General Dentistry, College of Dentistry, Yonsei University</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Yonsei University, Institute for Innovation in Digital Healthcare</institution><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>Department of Conservative Dentistry, Yonsei University College of Dentistry</institution><addr-line>50-1 Yonsei-ro, Seodaemun-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Dhawan</surname><given-names>Pankaj</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liang</surname><given-names>Xiaolong</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Baek</surname><given-names>Yoongyu</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Yooseok Shin, DDS, PhD, Department of Conservative Dentistry, Yonsei University College of Dentistry, 50-1 Yonsei-ro, Seodaemun-gu, Seoul, 03722, Republic of Korea, +82-2-2228-3146, +82-2-313-7575; <email>densys@yuhs.ac</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>24</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e86145</elocation-id><history><date date-type="received"><day>19</day><month>10</month><year>2025</year></date><date date-type="rev-recd"><day>14</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>16</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Dahyun Seo, Jieun Cheong, Yiseul Choi, Yooseok Shin, Wonse Park. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 24.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e86145"/><abstract><sec><title>Background</title><p>Accurate assessment of pulpal status is essential for achieving successful endodontic outcomes. However, direct evaluation remains inherently challenging because the pulp is surrounded by calcified tissue, necessitating reliance on clinical and radiographic examinations for diagnostic and prognostic decision-making. These procedures demand substantial clinical expertise and time, and less-experienced clinicians often face challenges that may lead to errors in diagnosis and treatment planning. Recent advancements in large language models (LLMs) offer promising opportunities to enhance clinical reasoning by facilitating the integration of evidence and supporting methodical diagnostic decision-making.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the clinical applicability of LLMs by comparing their text-based clinical screening performance and the clinical validity of their treatment plan responses with those of human evaluators.</p></sec><sec sec-type="methods"><title>Methods</title><p>Between January 2011 and December 2022, 100 clinical cases involving primary endodontic disease were randomly selected from the clinical records of outpatients who visited the Department of Conservative Dentistry or Advanced General Dentistry (AGD) at Yonsei University Dental Hospital. Four prompt types, combining 2 variables (language and role), were used as input for 4 LLMs. Both LLMs and human evaluators (AGD specialists, AGD residents, endodontic residents, and senior dental students) assessed the cases using text-based clinical records. Radiographic images were not directly provided. Screening performance was evaluated using a 0-to-2-point concordance scale, and treatment plan validity and relevance were assessed using a 5-point Likert scale.</p></sec><sec sec-type="results"><title>Results</title><p>Among the 4 LLMs evaluated, ChatGPT achieved the highest mean concordance score on Korean-doctor prompts (mean 0.98, SD 0.82). However, this score did not reach the partially correct criterion of 1 on the 0 to 2-point scale. Clova X recorded the lowest mean score on English-patient prompts (mean 0.23, SD 0.63). Across both diagnostic categories, AGD specialists demonstrated the highest diagnostic accuracy (pulpal: 0.70; periapical: 0.65), with higher sensitivity but lower specificity than those exhibited by the other groups. ChatGPT also showed favorable performance among the LLMs, with accuracies of 0.65 (95% CI 0.55&#x2010;0.74) for pulpal disease and 0.57 (95% CI 0.47&#x2010;0.69) for periapical disease, which were comparable to those of AGD and endodontic residents.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Under image-free clinical record review conditions, ChatGPT 4.0 showed relatively higher and more consistent performance in symptom screening and treatment planning compared to the other LLMs evaluated. However, its highest mean score of 0.98 (SD 0.82) did not reach the partially correct criterion of 1 on the 0 to 2-point scale. Hallucinations generated by LLMs and experience-dependent interpretation biases among human evaluators remain key challenges that require attention. Therefore, continuous clinical supervision and comprehensive user training are necessary for the safe and effective clinical application.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>decision support systems</kwd><kwd>clinical</kwd><kwd>endodontics</kwd><kwd>prompt design</kwd><kwd>ChatGPT</kwd><kwd>Gemini</kwd><kwd>Microsoft Bing</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Accurate diagnosis of pulp pathologies plays a critical role in the success of root canal treatment, with the ultimate objective being the preservation of natural dentition [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. In many clinical situations, root canal treatment is combined with periodontal and prosthodontic procedures, which complicates the process of diagnosis and treatment planning. A thorough evaluation of pulp and periapical tissue conditions, including the assessment of pulp vitality, remains essential before treatment initiation [<xref ref-type="bibr" rid="ref2">2</xref>]. However, the pulp is surrounded by calcified tissue, which structurally limits direct examination [<xref ref-type="bibr" rid="ref3">3</xref>]. Consequently, radiographic assessment remains an indispensable diagnostic modality for treatment planning and prognosis evaluation, playing a vital role in the objective confirmation of lesions [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>The integration of imaging findings with clinical data often requires substantial time and expertise. Limited resources and time constraints frequently prevent clinicians from promptly incorporating the latest evidence or clinical guidelines into practice. Less experienced clinicians, in particular, are more vulnerable to diagnostic bias and errors, which may compromise diagnostic accuracy and clinical outcomes [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Recent advances in large language models (LLMs), which can process and generate human-like responses from unstructured patient records, have introduced new possibilities for supporting clinical assessment and treatment planning based on patient information [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>LLMs can facilitate systematic clinical decision-making by integrating recent research findings and clinical guidelines. They may also provide value in complex cases by reconfirming overlooked information, reducing uncertainty, and supporting the development of individualized treatment strategies. This capability enables clinicians to refine diagnostic reasoning and design patient-specific treatment plans [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Therefore, evaluating the potential of LLMs as auxiliary tools to enhance symptom-based screening performance and treatment planning validity is a priority.</p><p>Most existing research on LLMs in conservative dentistry has focused on educational applications, such as responding to patient-related clinical questions or addressing multiple-choice examinations [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. Furthermore, few studies have evaluated LLM performance using real-world clinical cases and compared model outputs with those of human evaluators. To address this gap, this study evaluated the clinical validity and relevance of LLM-generated symptom-based assessments and treatment planning responses using routine clinical records. This study examined how LLMs may function during the initial clinical assessment, where narrative clinical information guides early screening decisions, and compared their outputs with those of clinicians and senior students across different levels of clinical experience. In addition, we provide practical evidence on where LLM support may be clinically useful and where expert oversight remains essential. Therefore, this study aimed to provide clinically validated criteria for different applications of LLMs within the limited context of symptom-based screening in the field of dentistry.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This retrospective study was approved by the institutional review board of the Dental Hospital of Yonsei University (IRB 2-2024-0075). All procedures were part of standard clinical care, and the committee waived the need for informed consent. To maintain privacy and confidentiality, all data were fully anonymized before analysis. This study did not involve participant compensation.</p></sec><sec id="s2-2"><title>Data Collection</title><p>This retrospective analysis included records of patients who visited the outpatient clinics of the Departments of Advanced General Dentistry (AGD) or Conservative Dentistry between January 2011 and December 2022. Eligible participants were adults aged 19 years or older, who had a recorded chief complaint at the initial visit, and were diagnosed with pulpal or periapical disease by a specialist. Among those who underwent pretreatment periapical radiography with corresponding clinical and radiographic records, 100 cases without previous endodontic treatment were randomly selected. The demographic and clinical characteristics of the included cases are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Chief complaints recorded in the clinical records reflect the patient-reported symptom history, whereas objective examination findings reflect the clinical status at the time of presentation.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographic and clinical characteristics of the study cases.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Values (N=100)</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="left" valign="top">57.22 (14.50)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x003C;40</td><td align="left" valign="top">9 (9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>40&#x2010;59</td><td align="left" valign="top">44 (44)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;60</td><td align="left" valign="top">47 (47)</td></tr><tr><td align="left" valign="top" colspan="2">Sex, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">34 (34)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">66 (66)</td></tr><tr><td align="left" valign="top" colspan="2">Arch, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Maxillary</td><td align="left" valign="top">46 (46)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mandibular</td><td align="left" valign="top">54 (54)</td></tr><tr><td align="left" valign="top" colspan="2">Tooth type, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Anterior tooth</td><td align="left" valign="top">12 (12)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Premolar</td><td align="left" valign="top">17 (17)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Molar</td><td align="left" valign="top">71 (71)</td></tr><tr><td align="left" valign="top" colspan="2">Symptoms, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pain on stimulus</td><td align="left" valign="top">47 (47)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Spontaneous pain</td><td align="left" valign="top">25 (25)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Asymptomatic</td><td align="left" valign="top">18 (18)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tenderness on percussion</td><td align="left" valign="top">7 (7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other symptoms</td><td align="left" valign="top">3 (3)</td></tr><tr><td align="left" valign="top" colspan="2">Restoration status, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Indirect restoration</td><td align="left" valign="top">45 (45)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>None</td><td align="left" valign="top">30 (30)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Direct restoration</td><td align="left" valign="top">14 (14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bridge</td><td align="left" valign="top">11 (11)</td></tr></tbody></table></table-wrap></sec><sec id="s2-3"><title>Study Design</title><p>Four LLMs were evaluated, namely ChatGPT (version 4.0, OpenAI Inc), Gemini (version 1.5 Pro, Google), Bing (Microsoft), and Clova X (version Hyper Clova X, Naver Corporation). All LLM queries were conducted between November 30, 2024, and December 7, 2024. Four prompt types were created by combining 2 variables&#x2014;language (Korean vs English) and role (doctor vs patient): (1) Korean-doctor, (2) Korean-patient, (3) English-doctor, and (4) English-patient (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). All LLMs were queried using platform-default generation settings, without any manual adjustments. Clinical cases were provided in a narrative clinical record format, including patient age, sex, chief complaint, reason for visit, and clinical symptoms. Radiographic images were not directly provided to either the LLMs or the human evaluators. However, because radiographic interpretations are commonly documented in routine clinical information, summarized radiographic descriptors could be included within the chart text. Screening performance was compared with that of human evaluators using Korean-doctor prompts reflecting the clinical background of the evaluators. All treatment plan responses were reformatted into a narrative format before evaluation to ensure assessor blinding to the source of the response. English prompts were generated from the original Korean prompts using Google Translate (Google LLC) and reviewed by a domain expert for accuracy. To ensure reproducibility, prompt structure and generation settings were standardized. Persona prompts were adopted to reflect differences in clinical communication between clinicians and patients. A zero-shot prompting design was used to allow standardized evaluation without example-based guidance, thereby improving comparability across models and evaluator groups.</p></sec><sec id="s2-4"><title>Evaluator Groups</title><p>A conservative dentistry specialist with more than 20 years of clinical experience initially reviewed and selected suitable clinical cases for this study based on clinical data and radiographic assessments. The same specialist established the reference standard diagnosis using the full clinical and radiographic records and was blinded to the LLM-generated outputs and the diagnostic assessments of all evaluators. Twelve evaluators then independently assessed the cases using the same clinical information in the patient records provided to the LLMs. The evaluators included 3 AGD specialists with more than 5 years of experience, 3 AGD residents with more than 2 years of experience, 3 endodontic residents with more than 2 years of experience, and 3 senior dental students. Subsequently, the same evaluators assessed the clinical validity and relevance of the treatment plans generated by each LLM.</p></sec><sec id="s2-5"><title>Evaluation Criteria</title><p>Symptom-based screening accuracy was evaluated on a 0 to 2-point scale according to pathological concordance and the criteria of the <italic>ICD-10</italic> (<italic>International Statistical Classification of Diseases</italic>, <italic>Tenth Revision</italic>) [<xref ref-type="bibr" rid="ref17">17</xref>]. Two points were assigned for diagnoses fully consistent with the reference standard, whereas 1 point was assigned for identification of the correct disease category with minor discrepancies in diagnostic specificity or terminology (eg, correct classification as pulpal disease with an imprecise subtype). No points were assigned for a complete mismatch (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Because the doctor prompts requested <italic>ICD-10</italic>&#x2013;based diagnoses while the reference standard used American Association of Endodontists terminology, differences in terminology between the 2 systems were handled within the existing assessment criteria, where responses identifying the correct disease category with a minor terminological discrepancy were scored as 1, and those fully consistent with the reference diagnosis were scored as 2. For screening performance analyses, the ordinal scores were binarized, with scores of 1 and 2 classified as positive and a score of 0 classified as negative. Diagnostic metrics were calculated using a one-versus-other scheme, in which pulpal cases (58/100) served as positives and periapical cases (42/100) as negatives for pulpal disease metrics, with the reverse applied for periapical disease metrics. For the human evaluator groups, responses from the 3 evaluators within each group were combined before calculating diagnostic metrics, resulting in an effective observation count of 174 per group for pulpal disease and 126 for periapical disease; 95% CIs reflect this aggregated count. The corresponding 2&#x00D7;2 contingency cell counts (true positive, false negative, false positive, and true negative) for all groups are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The clinical validity and relevance of LLM-generated treatment plans were assessed using a 5-point Likert scale [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Higher scores reflected greater clinical appropriateness and concordance with evidence-based practice. Evaluators independently rated the realism and evidence-based nature of each plan within a real-world clinical context; they were blinded to the LLM outputs (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p></sec><sec id="s2-6"><title>Statistical Analysis</title><p>To analyze differences between groups, the Kruskal-Wallis test was used, with post hoc pairwise comparisons conducted using the Dunn test with Bonferroni correction. Interrater agreement among the 4 evaluator groups was assessed using the intraclass correlation coefficient (ICC). Screening sensitivity, specificity, positive predictive value, negative predictive value, accuracy, and 95% CIs of the LLMs and clinicians were analyzed and compared for each disease. Diagnostic metrics and corresponding 95% CIs were calculated using the MedCalc Diagnostic Test Evaluation Calculator [<xref ref-type="bibr" rid="ref20">20</xref>]. All statistical analyses were performed using SPSS (version 27, IBM Corp). Statistical significance was set at <italic>P</italic>&#x003C;.05.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Text-Based Screening Performance by Prompt Type</title><p>The case selection and diagnostic classification process is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. The text-based screening performance of LLMs, measured on a 0 to 2-point scale, was compared across prompt types (doctor vs patient) and languages (Korean vs English; <xref ref-type="fig" rid="figure2">Figure 2</xref>). ChatGPT achieved the highest concordance on Korean-doctor prompts (mean 0.98, SD 0.82), with a significant within-model difference between languages (<italic>P</italic>&#x003C;.001). Conversely, Clova X recorded the lowest agreement on English-patient prompts (mean 0.23, SD 0.63), with a significant within-model difference (<italic>P</italic>=.004).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow diagram of case selection and diagnostic classification.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e86145_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Diagnostic performance and consistency of 4 large language models across prompt types. Diagnostic scores (0 to 2-point scale) for ChatGPT 4.0, Gemini 1.5, Bing, and Clova X using (A) doctor and (B) patient prompts in Korean and English. Bars represent mean (SD). Statistically significant differences in diagnostic agreement were observed across languages within each model (**<italic>P</italic>&#x003C;.01; *** <italic>P</italic>&#x003C;.001).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e86145_fig02.png"/></fig></sec><sec id="s3-2"><title>Evaluation of Treatment Plan Relevance and Validity</title><p>Across all evaluator groups, ChatGPT demonstrated higher mean ranks for treatment plan validity than other LLMs, whereas Clova X showed the lowest mean ranks (<xref ref-type="fig" rid="figure3">Figures 3</xref> and <xref ref-type="fig" rid="figure4">4</xref>). Statistically significant differences in both outcomes were observed among LLMs within each evaluator group.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Relevance scores across large language models (LLMs) as evaluated by 4 human evaluator groups: (A) Advanced General Dentistry (AGD) specialists, (B) AGD residents, (C) endodontic residents, and (D) senior students. Box plots show median (IQR) and outliers of relevance scores on a 5-point Likert scale. Relevance reflects how directly LLM responses addressed the clinical questions posed in each prompt.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e86145_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Validity scores across large language models (LLMs) as evaluated by 4 human evaluator groups: (A) Advanced General Dentistry (AGD) specialists, (B) AGD residents, (C) endodontic residents, and (D) senior students. Box plots show the median, IQR, and outliers of validity scores on a 5-point Likert scale. Validity reflects the clinical appropriateness and concordance of LLM responses with evidence-based diagnostic reasoning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e86145_fig04.png"/></fig></sec><sec id="s3-3"><title>Comparison of Symptom-Based Assessment Scores</title><p>Symptom-based assessment scores on the 0 to 2-point scale were compared between LLMs and human evaluators using the Kruskal-Wallis test. AGD specialists achieved the highest mean rank (mean 1.10, SD 0.71; 466.81), followed by endodontic residents (mean 0.97, SD 0.79; 425.49), ChatGPT (mean 0.98, SD 0.82; 425.21), and AGD residents (mean 0.94, SD 0.78; 416.07). Senior dental students showed intermediate performance (mean 0.90, SD 0.71; 410.20), whereas Bing (mean 0.77, SD 0.90; 361.55), Gemini (mean 0.74, SD 0.79; 357.28), and Clova X (mean 0.68, SD 0.74; 341.40) demonstrated lower mean ranks (<xref ref-type="table" rid="table2">Table 2</xref>). There was a significant difference in symptom-based assessment scores among the groups (<italic>&#x03C7;</italic>&#x00B2;<sub>7</sub>=25.75; <italic>P</italic>&#x003C;.001).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparison of symptom-based assessment scores across large language models and human evaluator groups using the Kruskal-Wallis test (N=100)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Group</td><td align="left" valign="bottom">Mean (SD)</td><td align="left" valign="bottom">Mean rank</td></tr></thead><tbody><tr><td align="left" valign="top">AGD<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> specialist</td><td align="char" char="." valign="top">1.10 (0.71)</td><td align="char" char="." valign="top">466.81<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Endodontic residents</td><td align="char" char="." valign="top">0.97 (0.79)</td><td align="char" char="." valign="top">425.49<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">ChatGPT 4.0</td><td align="char" char="." valign="top">0.98 (0.82)</td><td align="char" char="." valign="top">425.21<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">AGD residents</td><td align="char" char="." valign="top">0.94 (0.78)</td><td align="char" char="." valign="top">416.07<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Senior student</td><td align="char" char="." valign="top">0.90 (0.71)</td><td align="char" char="." valign="top">410.20<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">Bing</td><td align="char" char="." valign="top">0.77 (0.90)</td><td align="char" char="." valign="top">361.55</td></tr><tr><td align="left" valign="top">Gemini 1.5</td><td align="char" char="." valign="top">0.74 (0.79)</td><td align="char" char="." valign="top">357.28</td></tr><tr><td align="left" valign="top">Clova X</td><td align="char" char="." valign="top">0.68 (0.74)</td><td align="char" char="." valign="top">341.40</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup><italic>&#x03C7;</italic>&#x00B2;<sub>7</sub>=25.75; <italic>P</italic>&#x003C;.001.</p></fn><fn id="table2fn2"><p><sup>b</sup>AGD: Advanced General Dentistry.</p></fn><fn id="table2fn3"><p><sup>c</sup>Mean ranks of AGD specialist are significantly different from Bing, Gemini 1.5, and Clova X.</p></fn><fn id="table2fn4"><p><sup>d</sup>Mean ranks of endodontic residents, ChatGPT 4.0, AGD residents, and senior students are not significantly different from either AGD specialists or Bing, Gemini 1.5, and Clova X based on Dunn post hoc test with Bonferroni correction (<italic>P</italic>&#x003E;.05).</p></fn></table-wrap-foot></table-wrap><p>Interrater agreement among the 4 evaluator groups, measured by the ICC, was lowest for AGD specialists (ICC 0.579, 95% CI 0.413&#x2010;0.704; <italic>P</italic>&#x003C;.001). Moderate agreement was observed for AGD residents (ICC 0.774, 95% CI 0.685&#x2010;0.841; <italic>P</italic>&#x003C;.001), endodontic residents (ICC 0.793, 95% CI 0.711&#x2010;0.854; <italic>P</italic>&#x003C;.001), and senior dental students (ICC 0.750, 95% CI 0.651&#x2010;0.824; <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table3">Table 3</xref>). Post hoc pairwise comparisons using the Dunn test with Bonferroni correction showed that AGD specialists scored significantly higher than Clova X, Gemini, and Bing (<xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Interrater reliability of diagnostic evaluations across evaluator groups.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluator group</td><td align="left" valign="bottom">ICC<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>, average measure (95% CI)</td><td align="left" valign="bottom"><italic>F</italic> value (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">AGD<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> specialists</td><td align="left" valign="top">0.579 (0.413&#x2010;0.704)</td><td align="left" valign="top">2.377 (99, 198)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">AGD residents</td><td align="left" valign="top">0.774 (0.685&#x2010;0.841)</td><td align="left" valign="top">4.431 (99, 198)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Endodontic residents</td><td align="left" valign="top">0.793 (0.711&#x2010;0.854)</td><td align="left" valign="top">4.830 (99, 198)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Senior dental students</td><td align="left" valign="top">0.750 (0.651&#x2010;0.824)</td><td align="left" valign="top">3.995 (99, 198)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table3fn2"><p><sup>b</sup>AGD: Advanced General Dentistry.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Comparative Diagnostic Metrics</title><p><xref ref-type="table" rid="table4">Tables 4</xref> and <xref ref-type="table" rid="table5">5</xref> present the diagnostic metrics for pulpal and periapical diseases, respectively. AGD specialists showed the highest diagnostic accuracy among all groups for both pulpal (0.70) and periapical diseases (0.65). Among the LLMs, ChatGPT demonstrated the highest accuracy for pulpal disease (0.65, 95% CI 0.55&#x2010;0.74), whereas Clova X showed the lowest accuracy (0.47, 95% CI 0.37&#x2010;0.57). For periapical disease, the overall diagnostic performance of LLMs was lower, with ChatGPT achieving an accuracy of 0.57 (95% CI 0.47&#x2010;0.69).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Diagnostic metrics (accuracy, sensitivity, and specificity with 95% CI) of large language models (LLMs) and human evaluators for pulpal disease (n=58)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Diagnostic metrics</td><td align="left" valign="bottom">ChatGPT 4.0</td><td align="left" valign="bottom">Gemini 1.5 Pro</td><td align="left" valign="bottom">Bing</td><td align="left" valign="bottom">Clova X</td><td align="left" valign="bottom">AGD<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> specialist</td><td align="left" valign="bottom">AGD residents</td><td align="left" valign="bottom">Endodontic residents</td><td align="left" valign="bottom">Senior students</td></tr></thead><tbody><tr><td align="left" valign="top">Sensitivity</td><td align="left" valign="top">0.60 (0.47&#x2010;0.73)</td><td align="left" valign="top">0.55 (0.42&#x2010;0.68)</td><td align="left" valign="top">0.45 (0.32&#x2010;0.58)</td><td align="left" valign="top">0.40 (0.27&#x2010;0.53)</td><td align="left" valign="top">0.79 (0.72&#x2010;0.85)</td><td align="left" valign="top">0.69 (0.62&#x2010;0.76)</td><td align="left" valign="top">0.71<break/>(0.64&#x2010;0.78)</td><td align="left" valign="top">0.51 (0.43&#x2010;0.59)</td></tr><tr><td align="left" valign="top">Specificity</td><td align="left" valign="top">0.71 (0.55&#x2010;0.84)</td><td align="left" valign="top">0.55 (0.39&#x2010;0.70)</td><td align="left" valign="top">0.69 (0.53&#x2010;0.82)</td><td align="left" valign="top">0.57 (0.41&#x2010;0.72)</td><td align="left" valign="top">0.57 (0.48&#x2010;0.66)</td><td align="left" valign="top">0.65 (0.56&#x2010;0.73)</td><td align="left" valign="top">0.67<break/>(0.58&#x2010;0.75)</td><td align="left" valign="top">0.61 (0.52&#x2010;0.70)</td></tr><tr><td align="left" valign="top">PPV<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.74 (0.63&#x2010;0.83)</td><td align="left" valign="top">0.63 (0.53&#x2010;0.71)</td><td align="left" valign="top">0.67 (0.54&#x2010;0.77)</td><td align="left" valign="top">0.56 (0.44&#x2010;0.67)</td><td align="left" valign="top">0.72 (0.67&#x2010;0.76)</td><td align="left" valign="top">0.73 (0.68&#x2010;0.78)</td><td align="left" valign="top">0.75<break/>(0.69&#x2010;0.79)</td><td align="left" valign="top">0.64 (0.58&#x2010;0.70)</td></tr><tr><td align="left" valign="top">NPV<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.57 (0.48&#x2010;0.68)</td><td align="left" valign="top">0.47 (0.37&#x2010;0.57)</td><td align="left" valign="top">0.48 (0.40&#x2010;0.55)</td><td align="left" valign="top">0.41 (0.33&#x2010;0.49)</td><td align="left" valign="top">0.66 (0.58&#x2010;0.73)</td><td align="left" valign="top">0.60 (0.54&#x2010;0.66)</td><td align="left" valign="top">0.63<break/>(0.56&#x2010;0.69)</td><td align="left" valign="top">0.48 (0.42&#x2010;0.52)</td></tr><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">0.65 (0.55&#x2010;0.74)</td><td align="left" valign="top">0.55 (0.45&#x2010;0.65)</td><td align="left" valign="top">0.55 (0.45&#x2010;0.65)</td><td align="left" valign="top">0.47 (0.37&#x2010;0.57)</td><td align="left" valign="top">0.70 (0.64&#x2010;0.75)</td><td align="left" valign="top">0.67 (0.62&#x2010;0.73)</td><td align="left" valign="top">0.69<break/>(0.64&#x2010;0.75)</td><td align="left" valign="top">0.55 (0.50&#x2010;0.61)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Human evaluator metrics were calculated by combining responses across 3 evaluators per group (effective n=174); LLM metrics reflect single-pass evaluations (n=58).</p></fn><fn id="table4fn2"><p><sup>b</sup>AGD: Advanced General Dentistry.</p></fn><fn id="table4fn3"><p><sup>c</sup>PPV: positive predictive value.</p></fn><fn id="table4fn4"><p><sup>d</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Diagnostic metrics (accuracy, sensitivity, specificity with 95% CI) of large language models (LLMs) and human evaluators for periapical disease (n=42)<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Diagnostic metrics</td><td align="left" valign="bottom">ChatGPT 4.0</td><td align="left" valign="bottom">Gemini 1.5 Pro</td><td align="left" valign="bottom">Bing</td><td align="left" valign="bottom">Clova X</td><td align="left" valign="bottom">AGD<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup> specialist</td><td align="left" valign="bottom">AGD residents</td><td align="left" valign="bottom">Endodontic residents</td><td align="left" valign="bottom">Senior students</td></tr></thead><tbody><tr><td align="left" valign="top">Sensitivity</td><td align="left" valign="top">0.43 (0.28&#x2010;0.59)</td><td align="left" valign="top">0.29 (0.16&#x2010;0.45)</td><td align="left" valign="top">0.43<break/>(0.28&#x2010;0.59)</td><td align="left" valign="top">0.24 (0.12&#x2010;0.39)</td><td align="left" valign="top">0.64 (0.55&#x2010;0.73)</td><td align="left" valign="top">0.45 (0.36&#x2010;0.54)</td><td align="left" valign="top">0.52<break/>(0.43&#x2010;0.61)</td><td align="left" valign="top">0.40 (0.31&#x2010;0.49)</td></tr><tr><td align="left" valign="top">Specificity</td><td align="left" valign="top">0.67 (0.54&#x2010;0.79)</td><td align="left" valign="top">0.62 (0.48&#x2010;0.74)</td><td align="left" valign="top">0.59 (0.45&#x2010;0.71)</td><td align="left" valign="top">0.72 (0.59&#x2010;0.83)</td><td align="left" valign="top">0.66 (0.59&#x2010;0.73)</td><td align="left" valign="top">0.64 (0.56&#x2010;0.71)</td><td align="left" valign="top">0.62<break/>(0.54-0.69)</td><td align="left" valign="top">0.58 (0.50&#x2010;0.65)</td></tr><tr><td align="left" valign="top">PPV<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="top">0.49 (0.36&#x2010;0.61)</td><td align="left" valign="top">0.35 (0.23&#x2010;0.49)</td><td align="left" valign="top">0.43 (0.32&#x2010;0.54)</td><td align="left" valign="top">0.38 (0.24&#x2010;0.55)</td><td align="left" valign="top">0.58 (0.52&#x2010;0.64)</td><td align="left" valign="top">0.48 (0.41&#x2010;0.54)</td><td align="left" valign="top">0.50<break/>(0.44-0.56)</td><td align="left" valign="top">0.41 (0.34&#x2010;0.47)</td></tr><tr><td align="left" valign="top">NPV<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">0.62 (0.54&#x2010;0.69)</td><td align="left" valign="top">0.55 (0.48&#x2010;0.61)</td><td align="left" valign="top">0.59 (0.50&#x2010;0.67)</td><td align="left" valign="top">0.57 (0.51&#x2010;0.62)</td><td align="left" valign="top">0.72 (0.66&#x2010;0.77)</td><td align="left" valign="top">0.62 (0.57&#x2010;0.66)</td><td align="left" valign="top">0.64<break/>(0.59-0.69)</td><td align="left" valign="top">0.57 (0.52&#x2010;0.62)</td></tr><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">0.57 (0.47&#x2010;0.69)</td><td align="left" valign="top">0.48 (0.38&#x2010;0.58)</td><td align="left" valign="top">0.52 (0.42&#x2010;0.62)</td><td align="left" valign="top">0.52 (0.42&#x2010;0.62)</td><td align="left" valign="top">0.65 (0.60&#x2010;0.71)</td><td align="left" valign="top">0.56 (0.50&#x2010;0.62)</td><td align="left" valign="top">0.58<break/>(0.52-0.64)</td><td align="left" valign="top">0.50 (0.45&#x2010;0.56)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Human evaluator metrics were calculated by combining responses across 3 evaluators per group (effective n=126). LLM metrics reflect single-pass evaluations (n=42).</p></fn><fn id="table5fn2"><p><sup>b</sup>AGD: Advanced General Dentistry.</p></fn><fn id="table5fn3"><p><sup>c</sup>PPV: positive predictive value.</p></fn><fn id="table5fn4"><p><sup>d</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Distribution of Diagnostic Scores</title><p>Distribution analysis of diagnostic scores demonstrated that AGD specialists had the highest proportion of correct diagnoses at 48% (48/100), followed by endodontic residents at 43% (43/100), AGD residents at 41% (41/100), and senior dental students at 33% (33/100; <xref ref-type="fig" rid="figure5">Figure 5</xref>). Among the LLMs, ChatGPT produced the highest proportion of correct responses at 32% (32/100), surpassing that of senior dental students but remaining lower than that of specialists and residents. Regarding incorrect responses, Bing had the highest proportion at 54% (54/100), followed by Clova X at 48% (48/100). An exploratory subgroup analysis comparing patients aged younger than 60 years and 60 years or older demonstrated no significant difference in diagnostic accuracy between the groups (<italic>P</italic>=.051).</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Distribution of diagnostic accuracy scores among large language models (LLMs) and human evaluator groups. Stacked bar graphs show the proportion (%) of diagnostic accuracy scores (0=&#x201C;incorrect,&#x201D; 1=&#x201C;partially correct,&#x201D; and 2=&#x201C;correct&#x201D;) for 4 LLMs and human evaluator groups. Among all groups, Advanced General Dentistry (AGD) specialists demonstrated the highest proportion of correct responses, followed by endodontic residents and AGD residents. ChatGPT 4.0 outperformed Gemini, Bing, and Clova X in diagnostic accuracy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e86145_fig05.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study compared the performance of LLM-generated outputs with that of clinicians and senior dental students, evaluating the clinical validity and relevance of LLM-derived treatment plans. AGD specialists achieved the highest symptom-based screening accuracy and sensitivity. Among the LLMs, ChatGPT consistently showed the highest performance, with screening accuracy and diagnostic metrics comparable to those of AGD and endodontic residents. However, its relatively low sensitivity for pulpal disease diagnosis represents an important safety limitation, indicating that LLM-generated outputs should not be used for independent definitive diagnosis without expert supervision.</p><p>The 0 to 2 points scoring system enabled standardized comparison across LLMs and human evaluators. However, partially correct responses included a broad range of diagnostic differences, from minor imprecision to clinically meaningful misclassification. Therefore, the findings should be interpreted with caution, as this simplified scoring may not fully reflect the clinical impact of specific diagnostic errors. Although AGD specialists demonstrated high screening accuracy, their interrater agreement was relatively low, likely reflecting variability in expert interpretation of ambiguous or incomplete clinical information. This variability does not indicate poorer diagnostic ability but highlights the complexity of decision-making under limited clinical information [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. LLMs did not achieve specialist-level performance; however, ChatGPT showed potential as a supportive tool for less experienced clinicians. Post hoc pairwise analysis further showed that AGD specialists scored significantly higher than Clova X, Gemini, and Bing. In contrast, no statistically significant difference was detected between ChatGPT and any of the human evaluator groups. These results are consistent with our overall finding that ChatGPT performed relatively better than the other LLMs evaluated, while overall performance remained modest. Nonetheless, hallucinations were observed, including fabricated clinical details and treatment plans inconsistent with established guidelines (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>).</p></sec><sec id="s4-2"><title>Comparisons With Prior Work</title><p>Although recent studies have investigated the use of LLMs in conservative dentistry, most have focused on educational settings, patient inquiries, or multiple-choice examinations [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref16">16</xref>] (<xref ref-type="table" rid="table6">Table 6</xref>). In contrast, this study used real-world data under multiple prompting scenarios. Zero-shot prompts and personas were applied to reflect practical clinical use, where optimized prompting cannot be assumed [<xref ref-type="bibr" rid="ref23">23</xref>].</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Comparison of this study with prior large language model (LLM) studies in dentistry.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Data source</td><td align="left" valign="bottom">Clinical task</td><td align="left" valign="bottom">Evaluators</td><td align="left" valign="bottom">Outcomes assessed</td></tr></thead><tbody><tr><td align="left" valign="top">Abdulrab et al [<xref ref-type="bibr" rid="ref12">12</xref>], 2025</td><td align="left" valign="top">Exam-style questions</td><td align="left" valign="top">Knowledge assessment</td><td align="left" valign="top">Students</td><td align="left" valign="top">Accuracy</td></tr><tr><td align="left" valign="top">Aljamani et al [<xref ref-type="bibr" rid="ref15">15</xref>], 2025</td><td align="left" valign="top">Patient inquiries</td><td align="left" valign="top">Patient information support</td><td align="left" valign="top">None</td><td align="left" valign="top">Response quality</td></tr><tr><td align="left" valign="top">B&#x00FC;ker and Mercan [<xref ref-type="bibr" rid="ref13">13</xref>], 2025</td><td align="left" valign="top">Patient inquiries</td><td align="left" valign="top">Patient information quality assessment</td><td align="left" valign="top">None</td><td align="left" valign="top">Readability, accuracy, and quality</td></tr><tr><td align="left" valign="top">Jalali et al [<xref ref-type="bibr" rid="ref14">14</xref>], 2025</td><td align="left" valign="top">Board-style exam questions</td><td align="left" valign="top">Board exam performance</td><td align="left" valign="top">None</td><td align="left" valign="top">Accuracy</td></tr><tr><td align="left" valign="top">Ar&#x0131;l&#x0131; &#x00D6;zt&#x00FC;rk et al [<xref ref-type="bibr" rid="ref16">16</xref>], 2025</td><td align="left" valign="top">Endodontic exam-style questions</td><td align="left" valign="top">Knowledge assessment in endodontics</td><td align="left" valign="top">None</td><td align="left" valign="top">Answer accuracy</td></tr><tr><td align="left" valign="top">This study</td><td align="left" valign="top">Text-based clinical records (image-free)</td><td align="left" valign="top">Symptom-based diagnostic screening and treatment planning</td><td align="left" valign="top">Specialists, residents, and students</td><td align="left" valign="top">Diagnostic accuracy, treatment plan validity, and interrater reliability</td></tr></tbody></table></table-wrap><p>Outputs varied substantially by language and prompt type, underscoring the importance of prompt design in clinical contexts [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Consistent with prior medical studies, role-specific prompts improved clinical relevance, whereas vague prompts reduced consistency [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. The student group rated the LLM-generated treatment plans more positively than experienced clinicians, likely reflecting differences in error recognition ability based on clinical experience. Tangadulrat et al [<xref ref-type="bibr" rid="ref28">28</xref>] reported similar findings in medicine, where medical students tended to trust ChatGPT&#x2019;s results more than physicians did. These results suggest that clinical experience is critical for evaluating AI-generated outputs. As summarized in <xref ref-type="table" rid="table6">Table 6</xref>, this study differs from prior work.</p><p>Recent advances in multimodal LLMs, such as GPT-4V and Gemini Pro Vision, demonstrate improved diagnostic performance when radiographic images are incorporated into diagnostic assessments [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. In this study, radiographic images were not directly reviewed, although summarized radiographic findings were available in clinical records. Given the indispensable role of radiographic examination in endodontic diagnosis, limited access to image interpretation may partly explain the lower sensitivity observed for LLMs compared with specialists. Incorporating radiographic image analysis into future dental AI systems may improve diagnostic performance in clinical practice.</p><p>Previous studies in conservative dentistry have mainly examined the use of LLMs in educational contexts and multiple-choice examinations. Previous studies by Ar&#x0131;l&#x0131; &#x00D6;zt&#x00FC;rk et al [<xref ref-type="bibr" rid="ref16">16</xref>], Abdulrab et al [<xref ref-type="bibr" rid="ref12">12</xref>], and K&#x00FC;nzle and Paris [<xref ref-type="bibr" rid="ref31">31</xref>] consistently demonstrated that more advanced models, such as ChatGPT-4 and ChatGPT-4o, showed higher accuracy and stability than earlier versions or other models. Durmazpinar and Ekmekci [<xref ref-type="bibr" rid="ref32">32</xref>] found that ChatGPT-4o outperformed students in multiple-choice questions. In addition, Nguyen et al [<xref ref-type="bibr" rid="ref33">33</xref>] observed that ChatGPT performed well in text-based questions but showed lower accuracy in image-based questions.</p><p>Similar findings have been reported in studies addressing patients&#x2019; clinical inquiries. Aljamani et al [<xref ref-type="bibr" rid="ref15">15</xref>], Baris and Baris [<xref ref-type="bibr" rid="ref34">34</xref>], and B&#x00FC;ker and Mercan [<xref ref-type="bibr" rid="ref13">13</xref>] showed that several models achieved high accuracy and quality in answering patients&#x2019; questions related to endodontic treatment. However, these studies consistently highlighted important limitations, including higher reading levels, variability in reliability, and the potential for misinterpretation without expert supervision.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study had several limitations. First, a zero-shot prompting design was used, and follow-up questions were intentionally excluded, which limited the evaluation of iterative clinical reasoning [<xref ref-type="bibr" rid="ref35">35</xref>]. Consequently, this study could not assess whether multiturn prompting, which allows iterative clarification and the inclusion of additional information, would alter diagnostic accuracy or reasoning performance.</p><p>Second, although the English prompts were reviewed by a domain expert, translation nuances may have influenced model interpretation. Notably, <italic>per</italic> (<italic>+</italic>) was used instead of the standard English shorthand for percussion sensitivity, and <italic>conservative dentist</italic> was used instead of the internationally recognized term <italic>endodontist</italic>. The study was also limited to English and Korean prompts, which may have restricted the generalizability of the findings to other linguistic or cultural contexts.</p><p>Third, the analysis relied on clinical record information. Radiographic images were not directly reviewed by LLMs or human evaluators, although summarized radiographic interpretations were available in routine documentation. Diagnostic performance may, therefore, have been underestimated. Accordingly, direct comparability between the reference standard and test conditions was limited. Therefore, it remains unclear whether diagnostic failures reflect limited reasoning capability or simply reflect missing diagnostic data. A single specialist determined the reference diagnosis, and pre-2017 cases were not reclassified under updated criteria. In addition, no formal sample size calculation was performed, as the sample size was determined by the available eligible cases. This may have limited statistical power for some comparisons. Furthermore, 139 of 312 eligible cases were excluded due to multiple concurrent symptoms, which may have resulted in a less clinically diverse set of included cases. Therefore, the diagnostic performance of both LLMs and human evaluators may have been higher than that observed in routine clinical practice, where patients frequently present with overlapping symptoms.</p><p>Fourth, doctor and patient prompts differed in both persona and the amount of clinical information provided, as doctor prompts included objective examination findings and radiographic descriptors, while patient prompts relied solely on subjective symptoms. Therefore, it remains unclear whether the observed performance differences were attributable to persona framing or to the difference in available clinical data. Additionally, the doctor prompts requested <italic>ICD-10</italic>&#x2013;based diagnoses, while the reference standard used AAE terminology. Where the 2 classification systems do not correspond one-to-one in certain categories, model responses may have been more likely to receive a score of 1 rather than 2, potentially underestimating performance under doctor-prompt conditions.</p><p>Fifth, the Kruskal-Wallis test assumes independent samples; however, because all groups assessed the identical set of 100 clinical cases, this assumption is not fully satisfied. This limitation should be considered when interpreting both <xref ref-type="table" rid="table2">Table 2</xref> and the post hoc pairwise comparisons in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Additionally, the one-versus-other classification scheme assigns each case to a single disease category. It may not fully capture clinically linked dual diagnoses, potentially classifying biologically comprehensive responses as false positives. Furthermore, pooling responses across 3 evaluators within each group may constitute pseudoreplication, potentially narrowing the reported 95% CIs in <xref ref-type="table" rid="table4">Tables 4</xref> and <xref ref-type="table" rid="table5">5</xref>. In addition, scores of 1 and 2 were both classified as positive in the binary diagnostic performance analysis. Although this approach was intended to reflect clinically meaningful partial agreement, grouping partially correct diagnoses as positive outcomes may have inflated the reported sensitivity and accuracy metrics (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>).</p><p>Finally, hallucination and interpretation biases were observed, but qualitative error classification was not performed. Future studies incorporating qualitative analyses and multimodal inputs are warranted for better characterization of safety risks and improved clinical applicability. Despite these limitations, this study provides a comprehensive evaluation using real-world clinical data under diverse prompting conditions.</p></sec><sec id="s4-4"><title>Future Outlook</title><p>Future research should evaluate multimodal LLMs integrating text and imaging data under identical clinical conditions. In addition, advanced prompting strategies&#x2014;including chain-of-thought and few-shot prompting&#x2014;should be examined to enhance structured clinical reasoning and diagnostic accuracy. Approaches that constrain model outputs using established clinical guidelines or retrieval-augmented generation may help mitigate hallucinations. Moreover, interactive conversational frameworks that enable clarification of missing information may also enhance clinical reasoning and reliability. From a clinical implementation perspective, a human-in-the-loop workflow may be more appropriate than independent model use. Given the relatively high specificity but limited sensitivity observed, LLMs may be better suited for generating differential diagnosis lists rather than for definitive screening. In this workflow, clinicians review model-generated suggestions and integrate them with clinical examination and radiographic findings before making final decisions.</p></sec><sec id="s4-5"><title>Conclusions</title><p>Compared to other LLMs, ChatGPT provides more stable and reliable responses in endodontic diagnosis, demonstrating potential as a supportive tool for clinicians. However, limitations such as hallucinations, low sensitivity, and variability depending on user expertise remain significant challenges to its use as an independent diagnostic aid. For safe and effective clinical application, the development of standardized prompt templates, the implementation of comprehensive user training, and continued clinical supervision will be essential.</p></sec></sec></body><back><ack><p>No generative AI tools were used in the preparation or writing of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the Basic Science Research Program through the National Research Foundation of Korea and funded by the Ministry of Education, Science, and Technology (grant RS-2023-00243783). The funder had no involvement in the study design, data collection, analysis, interpretation, or the writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are not publicly available due to restrictions imposed by the Institutional Review Board of Yonsei University Dental Hospital (IRB number 2-2024-0075) and applicable patient privacy regulations. All data are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: DS, JC, YC, YS, WP</p><p>Data curation: DS, JC, YC</p><p>Formal analysis: DS</p><p>Funding acquisition: JC</p><p>Investigation: DS, JC</p><p>Methodology: DS, JC</p><p>Project administration: YS, WP</p><p>Supervision: YS, WP</p><p>Visualization: DS</p><p>Writing &#x2013; original draft: DS, JC</p><p>Writing &#x2013; review &#x0026; editing: DS, JC, YC, YS, WP</p><p>YS and WP served as co-corresponding authors. All the authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AGD</term><def><p>Advanced General Dentistry</p></def></def-item><def-item><term id="abb2">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb3"><italic>ICD-10</italic></term><def><p><italic>International Statistical Classification of Diseases, Tenth Revision</italic></p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sigurdsson</surname><given-names>A</given-names> </name></person-group><article-title>Pulpal diagnosis</article-title><source>Endod Topics</source><year>2003</year><month>07</month><volume>5</volume><issue>1</issue><fpage>12</fpage><lpage>25</lpage><pub-id pub-id-type="doi">10.1111/j.1601-1546.2003.00024.x</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rosen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Paul</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tsesis</surname><given-names>I</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rosen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Nemcovsky</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Tsesis</surname><given-names>I</given-names> </name></person-group><article-title>Evidence-based decision making in dentistry: the endodontic perspective</article-title><source>Evidence-Based Decision Making in Dentistry: Multidisciplinary Management of the Natural Dentition</source><year>2017</year><publisher-name>Springer</publisher-name><fpage>19</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-45733-8_3</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weisleder</surname><given-names>R</given-names> </name><name name-style="western"><surname>Yamauchi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Caplan</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Trope</surname><given-names>M</given-names> </name><name name-style="western"><surname>Teixeira</surname><given-names>FB</given-names> </name></person-group><article-title>The validity of pulp testing: a clinical study</article-title><source>J Am Dent Assoc</source><year>2009</year><month>08</month><volume>140</volume><issue>8</issue><fpage>1013</fpage><lpage>1017</lpage><pub-id pub-id-type="doi">10.14219/jada.archive.2009.0312</pub-id><pub-id pub-id-type="medline">19654254</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pimentel</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Abella</surname><given-names>F</given-names> </name><name name-style="western"><surname>Durack</surname><given-names>C</given-names> </name></person-group><article-title>Cone beam computed tomography in endodontics&#x2014;a review of the literature</article-title><source>Int Endod J</source><year>2019</year><month>08</month><volume>52</volume><issue>8</issue><fpage>1138</fpage><lpage>1152</lpage><pub-id pub-id-type="doi">10.1111/iej.13115</pub-id><pub-id pub-id-type="medline">30868610</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Petersson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Axelsson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Davidson</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Radiological diagnosis of periapical bone tissue lesions in endodontics: a systematic review</article-title><source>Int Endod J</source><year>2012</year><month>09</month><volume>45</volume><issue>9</issue><fpage>783</fpage><lpage>801</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2591.2012.02034.x</pub-id><pub-id pub-id-type="medline">22429152</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mileman</surname><given-names>PA</given-names> </name><name name-style="western"><surname>van den Hout</surname><given-names>WB</given-names> </name></person-group><article-title>Evidence-based diagnosis and clinical decision making</article-title><source>Dentomaxillofac Radiol</source><year>2009</year><month>01</month><volume>38</volume><issue>1</issue><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.1259/dmfr/18200441</pub-id><pub-id pub-id-type="medline">19114417</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sutherland</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Matthews</surname><given-names>DC</given-names> </name></person-group><article-title>Conducting systematic reviews and creating clinical practice guidelines in dentistry: lessons learned</article-title><source>J Am Dent Assoc</source><year>2004</year><month>06</month><volume>135</volume><issue>6</issue><fpage>747</fpage><lpage>753</lpage><pub-id pub-id-type="doi">10.14219/jada.archive.2004.0301</pub-id><pub-id pub-id-type="medline">15270157</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>O</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><etal/></person-group><article-title>ChatGPT for shaping the future of dentistry: the potential of multi-modal large language model</article-title><source>Int J Oral Sci</source><year>2023</year><month>07</month><day>28</day><volume>15</volume><issue>1</issue><fpage>29</fpage><pub-id pub-id-type="doi">10.1038/s41368-023-00239-y</pub-id><pub-id pub-id-type="medline">37507396</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirosawa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kawamura</surname><given-names>R</given-names> </name><name name-style="western"><surname>Harada</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>ChatGPT-generated differential diagnosis lists for complex case-derived clinical vignettes: diagnostic accuracy evaluation</article-title><source>JMIR Med Inform</source><year>2023</year><month>10</month><day>9</day><volume>11</volume><fpage>e48808</fpage><pub-id pub-id-type="doi">10.2196/48808</pub-id><pub-id pub-id-type="medline">37812468</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mago</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>M</given-names> </name></person-group><article-title>The potential usefulness of ChatGPT in oral and maxillofacial radiology</article-title><source>Cureus</source><year>2023</year><month>07</month><volume>15</volume><issue>7</issue><fpage>e42133</fpage><pub-id pub-id-type="doi">10.7759/cureus.42133</pub-id><pub-id pub-id-type="medline">37476297</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Danesh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Danesh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Danesh</surname><given-names>F</given-names> </name></person-group><article-title>Innovating dental diagnostics: ChatGPT&#x2019;s accuracy on diagnostic challenges</article-title><source>Oral Dis</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>911</fpage><lpage>917</lpage><pub-id pub-id-type="doi">10.1111/odi.15082</pub-id><pub-id pub-id-type="medline">39039720</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdulrab</surname><given-names>S</given-names> </name><name name-style="western"><surname>Abada</surname><given-names>H</given-names> </name><name name-style="western"><surname>Mashyakhy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mostafa</surname><given-names>N</given-names> </name><name name-style="western"><surname>Alhadainy</surname><given-names>H</given-names> </name><name name-style="western"><surname>Halboub</surname><given-names>E</given-names> </name></person-group><article-title>Performance of 4 artificial intelligence chatbots in answering endodontic questions</article-title><source>J Endod</source><year>2025</year><month>05</month><volume>51</volume><issue>5</issue><fpage>602</fpage><lpage>608</lpage><pub-id pub-id-type="doi">10.1016/j.joen.2025.01.002</pub-id><pub-id pub-id-type="medline">39814135</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>B&#x00FC;ker</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mercan</surname><given-names>G</given-names> </name></person-group><article-title>Readability, accuracy and appropriateness and quality of AI chatbot responses as a patient information source on root canal retreatment: a comparative assessment</article-title><source>Int J Med Inform</source><year>2025</year><month>09</month><volume>201</volume><fpage>105948</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.105948</pub-id><pub-id pub-id-type="medline">40288015</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jalali</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mohammad-Rahimi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>FM</given-names> </name><etal/></person-group><article-title>Performance of 7 artificial intelligence chatbots on board-style endodontic questions</article-title><source>J Endod</source><year>2025</year><month>10</month><volume>51</volume><issue>10</issue><fpage>1413</fpage><lpage>1419</lpage><pub-id pub-id-type="doi">10.1016/j.joen.2025.06.014</pub-id><pub-id pub-id-type="medline">40581328</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aljamani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hassona</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fansa</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Saadeh</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Dafi Jamani</surname><given-names>K</given-names> </name></person-group><article-title>Evaluating large language models in addressing patient questions on endodontic pain: a comparative analysis of accessible chatbots</article-title><source>J Endod</source><year>2025</year><month>11</month><volume>51</volume><issue>11</issue><fpage>1617</fpage><lpage>1624</lpage><pub-id pub-id-type="doi">10.1016/j.joen.2025.04.015</pub-id><pub-id pub-id-type="medline">40334976</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ar&#x0131;l&#x0131; &#x00D6;zt&#x00FC;rk</surname><given-names>E</given-names> </name><name name-style="western"><surname>Turan G&#x00F6;kduman</surname><given-names>C</given-names> </name><name name-style="western"><surname>&#x00C7;anak&#x00E7;i</surname><given-names>BC</given-names> </name></person-group><article-title>Evaluation of the performance of ChatGPT-4 and ChatGPT-4o as a learning tool in endodontics</article-title><source>Int Endod J</source><year>2026</year><month>06</month><volume>59</volume><issue>6</issue><fpage>1057</fpage><lpage>1069</lpage><pub-id pub-id-type="doi">10.1111/iej.14217</pub-id><pub-id pub-id-type="medline">40025853</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>International statistical classification of diseases and related health problems</article-title><source>World Health Organization</source><access-date>2026-06-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://icd.who.int/browse10/2019/en">https://icd.who.int/browse10/2019/en</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>Z</given-names> </name></person-group><article-title>Comprehensiveness of large language models in patient queries on gingival and endodontic health</article-title><source>Int Dent J</source><year>2025</year><month>02</month><volume>75</volume><issue>1</issue><fpage>151</fpage><lpage>157</lpage><pub-id pub-id-type="doi">10.1016/j.identj.2024.06.022</pub-id><pub-id pub-id-type="medline">39147663</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cocci</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pezzoli</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lo Re</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Quality of information and appropriateness of ChatGPT outputs for urology patients</article-title><source>Prostate Cancer Prostatic Dis</source><year>2024</year><month>03</month><volume>27</volume><issue>1</issue><fpage>103</fpage><lpage>108</lpage><pub-id pub-id-type="doi">10.1038/s41391-023-00705-y</pub-id><pub-id pub-id-type="medline">37516804</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>Diagnostic test evaluation</article-title><source>Medcalc</source><access-date>2026-06-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.medcalc.org/en/calc/diagnostic_test.php">https://www.medcalc.org/en/calc/diagnostic_test.php</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eva</surname><given-names>KW</given-names> </name></person-group><article-title>What every teacher needs to know about clinical reasoning</article-title><source>Med Educ</source><year>2005</year><month>01</month><volume>39</volume><issue>1</issue><fpage>98</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2929.2004.01972.x</pub-id><pub-id pub-id-type="medline">15612906</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Loy</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Irwig</surname><given-names>L</given-names> </name></person-group><article-title>Accuracy of diagnostic tests read with and without clinical information: a systematic review</article-title><source>JAMA</source><year>2004</year><month>10</month><day>6</day><volume>292</volume><issue>13</issue><fpage>1602</fpage><lpage>1609</lpage><pub-id pub-id-type="doi">10.1001/jama.292.13.1602</pub-id><pub-id pub-id-type="medline">15467063</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hayashi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Neubig</surname><given-names>G</given-names> </name></person-group><article-title>Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing</article-title><source>ACM Comput Surv</source><year>2023</year><month>09</month><day>30</day><volume>55</volume><issue>9</issue><fpage>1</fpage><lpage>35</lpage><pub-id pub-id-type="doi">10.1145/3560815</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mesk&#x00F3;</surname><given-names>B</given-names> </name></person-group><article-title>Prompt engineering as an important emerging skill for medical professionals: tutorial</article-title><source>J Med Internet Res</source><year>2023</year><month>10</month><day>4</day><volume>25</volume><fpage>e50638</fpage><pub-id pub-id-type="doi">10.2196/50638</pub-id><pub-id pub-id-type="medline">37792434</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zaghir</surname><given-names>J</given-names> </name><name name-style="western"><surname>Naguib</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bjelogrlic</surname><given-names>M</given-names> </name><name name-style="western"><surname>N&#x00E9;v&#x00E9;ol</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tannier</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lovis</surname><given-names>C</given-names> </name></person-group><article-title>Prompt engineering paradigms for medical applications: scoping review</article-title><source>J Med Internet Res</source><year>2024</year><month>09</month><day>10</day><volume>26</volume><fpage>e60501</fpage><pub-id pub-id-type="doi">10.2196/60501</pub-id><pub-id pub-id-type="medline">39255030</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leypold</surname><given-names>T</given-names> </name><name name-style="western"><surname>Sch&#x00E4;fer</surname><given-names>B</given-names> </name><name name-style="western"><surname>Boos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Beier</surname><given-names>JP</given-names> </name></person-group><article-title>Can AI think like a plastic surgeon? Evaluating GPT-4&#x2019;s clinical judgment in reconstructive procedures of the upper extremity</article-title><source>Plast Reconstr Surg Glob Open</source><year>2023</year><month>12</month><volume>11</volume><issue>12</issue><fpage>e5471</fpage><pub-id pub-id-type="doi">10.1097/GOX.0000000000005471</pub-id><pub-id pub-id-type="medline">38093728</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grabb</surname><given-names>D</given-names> </name></person-group><article-title>The impact of prompt engineering in large language model performance: a psychiatric example</article-title><source>J Med Artif Intell</source><year>2023</year><volume>6</volume><pub-id pub-id-type="doi">10.21037/jmai-23-71</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tangadulrat</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sono</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tangtrakulwanich</surname><given-names>B</given-names> </name></person-group><article-title>Using ChatGPT for clinical practice and medical education: cross-sectional survey of medical students&#x2019; and physicians&#x2019; perceptions</article-title><source>JMIR Med Educ</source><year>2023</year><month>12</month><day>22</day><volume>9</volume><issue>1</issue><fpage>e50658</fpage><pub-id pub-id-type="doi">10.2196/50658</pub-id><pub-id pub-id-type="medline">38133908</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roos</surname><given-names>J</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kaczmarczyk</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating Bard Gemini Pro and GPT-4 Vision against student performance in medical visual question answering: comparative case study</article-title><source>JMIR Form Res</source><year>2024</year><month>12</month><day>17</day><volume>8</volume><issue>1</issue><fpage>e57592</fpage><pub-id pub-id-type="doi">10.2196/57592</pub-id><pub-id pub-id-type="medline">39714199</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suh</surname><given-names>PS</given-names> </name><name name-style="western"><surname>Shim</surname><given-names>WH</given-names> </name><name name-style="western"><surname>Suh</surname><given-names>CH</given-names> </name><etal/></person-group><article-title>Comparing diagnostic accuracy of radiologists versus GPT-4V and Gemini Pro Vision using image inputs from diagnosis please cases</article-title><source>Radiology</source><year>2024</year><month>07</month><volume>312</volume><issue>1</issue><fpage>e240273</fpage><pub-id pub-id-type="doi">10.1148/radiol.240273</pub-id><pub-id pub-id-type="medline">38980179</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>K&#x00FC;nzle</surname><given-names>P</given-names> </name><name name-style="western"><surname>Paris</surname><given-names>S</given-names> </name></person-group><article-title>Performance of large language artificial intelligence models on solving restorative dentistry and endodontics student assessments</article-title><source>Clin Oral Investig</source><year>2024</year><month>10</month><day>7</day><volume>28</volume><issue>11</issue><fpage>575</fpage><pub-id pub-id-type="doi">10.1007/s00784-024-05968-w</pub-id><pub-id pub-id-type="medline">39373739</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Durmazpinar</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Ekmekci</surname><given-names>E</given-names> </name></person-group><article-title>Comparing diagnostic skills in endodontic cases: dental students versus ChatGPT-4o</article-title><source>BMC Oral Health</source><year>2025</year><month>03</month><day>29</day><volume>25</volume><issue>1</issue><fpage>457</fpage><pub-id pub-id-type="doi">10.1186/s12903-025-05857-y</pub-id><pub-id pub-id-type="medline">40158110</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Dang</surname><given-names>HP</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>TL</given-names> </name><name name-style="western"><surname>Hoang</surname><given-names>V</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>VA</given-names> </name></person-group><article-title>Accuracy of latest large language models in answering multiple choice questions in dentistry: a comparative study</article-title><source>PLoS One</source><year>2025</year><volume>20</volume><issue>1</issue><fpage>e0317423</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0317423</pub-id><pub-id pub-id-type="medline">39879192</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baris</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Baris</surname><given-names>K</given-names> </name></person-group><article-title>Assessment of various artificial intelligence applications in responding to technical questions in endodontic surgery</article-title><source>BMC Oral Health</source><year>2025</year><month>05</month><day>22</day><volume>25</volume><issue>1</issue><fpage>763</fpage><pub-id pub-id-type="doi">10.1186/s12903-025-06149-1</pub-id><pub-id pub-id-type="medline">40405212</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Prompt engineering in consistency and reliability with the evidence-based guideline for LLMs</article-title><source>NPJ Digit Med</source><year>2024</year><month>02</month><day>20</day><volume>7</volume><issue>1</issue><fpage>41</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01029-4</pub-id><pub-id pub-id-type="medline">38378899</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt templates used for the 4 prompt types (Korean-doctor, Korean-patient, English-doctor, and English-patient) administered to all large language models.</p><media xlink:href="medinform_v14i1e86145_app1.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Evaluation criteria for symptom-based screening performance using the 0 to 2-point concordance scale.</p><media xlink:href="medinform_v14i1e86145_app2.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Pooled 2&#x00D7;2 contingency cell counts underlying the diagnostic metrics reported in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><media xlink:href="medinform_v14i1e86145_app3.docx" xlink:title="DOCX File, 20 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Evaluation criteria for treatment plan validity and relevance using the 5-point Likert scale.</p><media xlink:href="medinform_v14i1e86145_app4.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Post hoc pairwise comparisons of symptom-based assessment scores across large language models and human evaluator groups using the Dunn test with Bonferroni correction (all 28 pairwise comparisons).</p><media xlink:href="medinform_v14i1e86145_app5.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Representative examples of hallucinated and clinically inconsistent responses across the 4 large language models evaluated.</p><media xlink:href="medinform_v14i1e86145_app6.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Strict diagnostic accuracy metrics (sensitivity, specificity, positive predictive value, negative predictive value, and accuracy).</p><media xlink:href="medinform_v14i1e86145_app7.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material></app-group></back></article>