<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e94614</article-id><article-id pub-id-type="doi">10.2196/94614</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Evaluation of Prompt Design and Internal Reasoning in Chatbot-Based Medical History Taking: Simulation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Thawinwisan</surname><given-names>Nattawipa</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Chang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yamamoto</surname><given-names>Goshiro</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kishimoto</surname><given-names>Kazumasa</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mori</surname><given-names>Yukiko</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kuroda</surname><given-names>Tomohiro</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Graduate School of Medicine, Kyoto University</institution><addr-line>54 Shogoin-kawahara-cho, Sakyo-ku</addr-line><addr-line>Kyoto</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Preemptive Medicine and Lifestyle-Related Disease Research Center, Kyoto University Hospital</institution><addr-line>Kyoto</addr-line><country>Japan</country></aff><aff id="aff3"><institution>Division of Medical Information Technology and Administration Planning, Kyoto University Hospital</institution><addr-line>Kyoto</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Yadamani</surname><given-names>Abinav</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kang</surname><given-names>Hongyu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Song</surname><given-names>Lei</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Nattawipa Thawinwisan, MD, Graduate School of Medicine, Kyoto University, 54 Shogoin-kawahara-cho, Sakyo-ku, Kyoto, Japan, 81 75-366-7701; <email>nattawipa_th@kuhp.kyoto-u.ac.jp</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>8</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e94614</elocation-id><history><date date-type="received"><day>08</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>10</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>14</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Nattawipa Thawinwisan, Chang Liu, Goshiro Yamamoto, Kazumasa Kishimoto, Yukiko Mori, Tomohiro Kuroda. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 21.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e94614"/><abstract><sec><title>Background</title><p>A persistent discrepancy exists between patient-reported information and physician documentation. While conversational agents have been developed to collect medical histories prior to consultations, existing evaluations have largely focused on diagnostic accuracy or user satisfaction rather than on the completeness and clinical relevance of the information collected. There remains a need to assess the extent to which clinically relevant information is captured through chatbot-based interviews, and to understand how model configurations and instructional strategies influence this coverage.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the extent to which a chatbot can obtain clinically useful patient history information, and to examine how prompt detail and internal reasoning influence information coverage during chatbot-based medical interviews.</p></sec><sec sec-type="methods"><title>Methods</title><p>We developed a medical history-taking chatbot using the Qwen3-14B-Instruct model and evaluated 4 configurations in a 2&#x00D7;2 factorial design: detailed and thinking mode, detailed and nonthinking mode, minimal and thinking mode, and minimal and nonthinking mode. These configurations were compared against a rule-based system baseline (choice mode) using 66 standardized primary care clinical cases, with simulated patients interacting with the chatbot according to predefined case scripts. Information coverage (%) was assessed using a checklist inspired by Objective Structured Clinical Examination (OSCE) frameworks. Three physicians independently evaluated transcript coverage, with interrater agreement assessed using full agreement rates and Fleiss &#x03BA;. For the 4 LLM configurations, coverage was analyzed using 2-way repeated-measures ANOVA to examine the effects of prompt detail, internal reasoning, and their interaction. All 5 configurations, including the rule-based baseline, were additionally compared using 1-way repeated-measures ANOVA with post hoc 2-tailed paired <italic>t</italic> tests.</p></sec><sec sec-type="results"><title>Results</title><p>Interrater agreement was substantial (Fleiss &#x03BA;=0.75). Across all 66 simulated cases, information coverage differed significantly among configurations (<italic>P</italic>&#x003C;.001), with the detailed prompt with thinking (detailed and thinking) mode achieving the highest mean coverage (72.3%, SD 14.3%), compared with moderate coverage in configurations using either thinking or detailed prompts alone (approximately 60%) and lower coverage in minimal nonthinking and rule-based configurations (approximately 51%&#x2010;54%). In the factorial analysis of the 4 LLM configurations, both prompt detail and internal reasoning were significantly associated with improved information coverage, with a significant interaction between prompt detail and internal reasoning interaction. Mode-related differences were most pronounced for past medical and family history domains. Symptom-level analyses revealed substantial variability, with higher coverage for symptoms associated with well-defined diagnostic frameworks and lower coverage for multisystem presentations.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this controlled, simulated setting, the detailed prompt with a thinking mode achieved the highest overall checklist-based information coverage. The findings suggest that combining structured clinical prompts with internal reasoning may improve the completeness of chatbot-collected patient histories. Further research is needed to evaluate its impact on clinical documentation, workflow integration, and real-world usefulness.</p></sec></abstract><kwd-group><kwd>chatbots</kwd><kwd>medical history taking</kwd><kwd>large language models</kwd><kwd>prompt engineering</kwd><kwd>clinical documentation</kwd><kwd>preconsultation assessment</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Information about medical history and personal perceptions provided by patients is essential for effective clinical care [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. However, studies have consistently demonstrated a persistent discrepancy between what patients report and what physicians document during clinical encounters [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. This discrepancy often arises from time constraints, selective note-taking, and different communication styles [<xref ref-type="bibr" rid="ref8">8</xref>]. The absence of this information can potentially hinder accurate diagnoses, continuity of care, and patient-centered decision-making.</p><p>To bridge these informational gaps, various digital tools have been developed to collect patient data before consultation [<xref ref-type="bibr" rid="ref9">9</xref>]. These systems aim to improve the completeness and efficiency of clinical documentation. Studies have shown that computerized history-taking enhances data quality and consistency compared to physician-recorded histories alone [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. More recently, conversational agents or chatbots have emerged as promising tools in health care [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Evidence suggests that chatbots can capture clinically relevant information while maintaining patient engagement and satisfaction, offering continuous and automated data collection that has the potential to improve the completeness and accessibility of medical history-taking [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>Despite the growing accessibility of these technologies, physicians&#x2019; acceptance of AI tools remains limited due to concerns about transparency, interpretability, and accountability [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. These concerns underscore the importance of evaluating AI-based tools using clinically interpretable outcome measures, rather than relying solely on technical performance metrics.</p><p>Existing preconsultation or history-taking tools are usually choice-based and structured [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. Although these systems are interpretable and support standardized documentation, their predefined options may constrain the capture of individualized or nuanced patient information. These limitations highlight the need for more flexible and adaptive approaches that maintain interpretability while capturing richer patient narratives.</p><p>Recent advances in large language models (LLMs) have significantly enhanced their contextual understanding and multistep reasoning capabilities, facilitating more coherent and clinically relevant patient interactions. Modern LLMs can now use an internal reasoning process, frequently termed &#x201C;thinking,&#x201D; to decompose complex clinical prompts into a sequence of intermediate logical steps before generating a final response. This deliberative process has been shown to improve performance on multifaceted tasks, leading to the generation of more structured and clinically aligned medical inquiries [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. However, their outputs depend on prompt design, which is critical for guiding model behavior and reasoning structure [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. Carefully constructing prompts to follow clinical reasoning frameworks can improve history coverage, mitigate premature diagnostic closure, and make the model&#x2019;s underlying logic more transparent.</p><p>Evaluations of AI-based conversational agents remain limited and have primarily focused on diagnostic accuracy, report quality, and user experience rather than on the completeness or clinical relevance of the information collected [<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>Detailed patient histories contribute to accurate diagnoses, treatment plans, and follow-ups by enabling physicians to systematically reason through symptom onset, progression, and contextual factors [<xref ref-type="bibr" rid="ref27">27</xref>]. When AI-assisted systems gather more complete information, they provide richer data for clinical decision-making and are less likely to prematurely close or jump to conclusions without sufficient information. Evaluating AI tools based on detail and reasoning coverage provides additional clinically useful aspects beyond diagnosis or satisfaction alone.</p><p>To address this need, we adopted an evaluation inspired by the Objective Structured Clinical Examination (OSCE) framework, which is widely used and trusted by physicians to assess clinical competence [<xref ref-type="bibr" rid="ref28">28</xref>]. OSCEs use checklist-based assessments to measure specific skills, ensuring interpretability and transparency. Our study applies a similar, checklist-driven approach, developed from representative clinical cases, to evaluate the coverage and completeness of AI-assisted history taking in a structured and clinically interpretable manner. This study aims (1) to assess the extent to which the chatbot can obtain clinically useful information, as measured by coverage of a case-based history-taking checklist, and (2) to determine the effects of internal reasoning and prompt detail on the chatbot&#x2019;s performance in achieving comprehensive history coverage.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Chatbot</title><sec id="s2-1-1"><title>Chatbot Creation</title><p>An overview of the study workflow is presented in <xref ref-type="fig" rid="figure1">Figure 1</xref>. The chatbot prototype was developed with the primary aim of collecting patient medical histories prior to physician encounters, to support subsequent clinical documentation and decision-making.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Experimental flow: standardized clinical cases (n=66) were compiled from medical textbooks, case reports, and case studies. Physicians extracted clinically relevant history items from each case to create case-specific evaluation checklists. Simulated patients reviewed the case scenarios and interacted with 5 history-taking chatbots. The chatbot configurations included 4 large language model (LLM) conditions in a 2&#x00D7;2 factorial design: detailed prompt with thinking mode (DT), detailed prompt with nonthinking mode (DN), minimal prompt with thinking mode (MT), and minimal prompt with nonthinking mode (MN), as well as a non-LLM chatbot baseline. Conversation transcripts were independently evaluated by physicians using a case-specific checklist. Information coverage was calculated and statistically compared across chatbot configurations.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e94614_fig01.png"/></fig><p>The chatbot prototype was developed in Python (Python Software Foundation) using the Qwen3-14B-Instruct model as the core LLM. Qwen3-14B is an open-source, transformer-based model that demonstrates competitive reasoning and instruction-following capabilities while remaining computationally efficient for local deployment [<xref ref-type="bibr" rid="ref29">29</xref>]. The model integrates &#x201C;thinking&#x201D; and &#x201C;nonthinking&#x201D; modes within a single framework, allowing flexible switching to examine how reasoning effort affects the completeness and interpretability of conversational outputs.</p><p>The system was implemented using Python 3.13.3, Streamlit 1.44.1 (Snowflake Inc), LangChain Core 0.3.51 (LangChain Inc), and LangChain Ollama 0.3.1 (LangChain Inc), with the Ollama backend managing model inference for the Qwen3-14B. Inference was conducted locally via Ollama, which handled model loading and execution without relying on external API calls. The conversational workflow was orchestrated through LangChain, which managed modular prompt structures and response generation. The Streamlit framework was used to build the graphical user interface, enabling real-time interaction with simulated patients (SPs) and structured data logging for subsequent analysis. Model inference used the following hyperparameters: temperature=0.6, top_<italic>P</italic>=.95, top_k=20, repeat_penalty=1, and num_predict=128.</p></sec><sec id="s2-1-2"><title>Types of Comparison</title><p>We evaluated 4 LLM conditions in a 2&#x00D7;2 factorial design and compared them with an existing choice-based, rule-driven chatbot as a non-LLM baseline.</p><p>The arms were defined as follows: detailed prompt with thinking mode (DT), detailed prompt with nonthinking mode (DN), minimal prompt with thinking mode (MT), and minimal prompt with nonthinking mode (MN); choice, the baseline condition represented a conventional rule-based chatbot using predefined options without generative language modeling.</p></sec><sec id="s2-1-3"><title>Operational Definitions</title><sec id="s2-1-3-1"><title>Prompt Design (Detailed vs Minimal)</title><p>The chatbot was guided by a detailed prompt, adapted from standard medical history-taking frameworks described in clinical textbooks [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref30">30</xref>], instructing it to act as a virtual medical assistant conducting comprehensive patient interviews. The prompt instructed the chatbot to generate only 1 question per response, avoid repeated questions, and maintain a smooth and natural conversation flow. It specified a structured interview sequence beginning with clarification of the reason for the visit and the main symptom, followed by symptom characterization, including onset, location, quality, severity, duration, triggers, relieving factors, recurrence, and prior treatments when clinically applicable. The chatbot was then instructed to ask about associated symptoms (ASs) across organ systems, to consider multiple differential diagnoses, to ask follow-up questions to support or rule out diagnostic possibilities, and to explore relevant external causes and risk factors. After sufficient information about the present illness had been gathered, the prompt instructed the chatbot to proceed in a fixed order to past medical history (PH), treatment and compliance, family history (FH), smoking and alcohol use, and conversation closure.</p><p>The minimal prompt instructed the chatbot to act as an AI-powered virtual medical assistant conducting a patient interview in English, generate 1 question per response, and to end the conversation when appropriate. It did not include explicit instructions regarding diagnostic reasoning, symptom characterization, AS, history-taking sequence, risk factors, or domain transitions.</p><p>The full text of both prompts is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-1-3-2"><title>Internal Reasoning Configuration (Thinking vs Nonthinking)</title><p>In this study, &#x201C;internal reasoning&#x201D; refers to the generation of intermediate reasoning steps during response generation and was operationalized using the model&#x2019;s built-in &#x201C;thinking mode.&#x201D; In this mode, complex, multistep deliberation occurred as intermediate reasoning steps before the model produced the final response, which were intended to guide response generation [<xref ref-type="bibr" rid="ref29">29</xref>]. In practical terms, this configuration was expected to influence how the model selected follow-up questions and organized the next conversational step during the interview. The model&#x2019;s internal reasoning process was configured in one of the two ways: (1) in the thinking mode, the model internally generated intermediate reasoning steps, while the user interface displayed only the final questions or responses; or (2) in the nonthinking mode, in which internal reasoning was disabled via a prompt instruction (/no_think), and questions were generated directly without intermediate deliberation.</p></sec></sec></sec><sec id="s2-2"><title>Baseline Description</title><p>Choice baseline is a commercially available symptom checker operating in Japan (Ubie, Inc) that guides users through fixed, multiple-choice questions based on branching clinical logic. This system is representative of current rule-based history-taking tools. Ubie was accessed in its English version through the publicly available web interface at the time of the study (which was operating in August 2025) [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec><sec id="s2-3"><title>End-of-Conversation Determination</title><p>Each LLM chatbot interaction was designed to terminate under specific conditions reflecting the intended prompt structure and conversation control strategy. The termination criteria differed across configurations to reflect the varying degrees of reasoning autonomy and prompt guidance.</p><sec id="s2-3-1"><title>Detailed Prompt (End When Sufficient Information Is Obtained)</title><p>For chatbots provided with detailed prompts, conversation termination was guided by an instruction to conclude the interview after sufficient information had been collected. Specifically, the prompt instructed the chatbot to politely inform the patient that all necessary information had been collected, thank them for their cooperation, let them know they would see the physician next, and ask about their expectations or any other questions they wanted to ask the physician. This instruction was intended to guide the chatbot to end the information-gathering phase after completing the structured history-taking sequence.</p></sec><sec id="s2-3-2"><title>Minimal Prompt (End When Appropriate)</title><p>In this mode, the chatbot autonomously decided when sufficient information had been gathered. The conversation ended once the system determined that the patient&#x2019;s responses provided adequate information without explicit external cues.</p></sec><sec id="s2-3-3"><title>Force-End Condition</title><p>Regardless of the configuration, all conversations were capped at 30 total messages (15 turns). This limit was selected to approximate a feasible preconsultation history-taking interaction while balancing information completeness with interaction burden for SPs. It also served to prevent excessively long or repetitive conversations. The same turn limit was applied to all LLM configurations to ensure comparability across modes. If this limit was reached before a natural conclusion, the conversation was automatically terminated, and the final message was recorded as the endpoint. The chatbot was not explicitly informed of this turn limit in the prompt; rather, the limit was implemented externally by the system.</p><p>This setup incorporated both prompt-guided and model-driven termination strategies to accommodate different levels of conversational control and ensure consistent stopping conditions across configurations.</p></sec></sec><sec id="s2-4"><title>Case and Checklist</title><sec id="s2-4-1"><title>Case Selection</title><p>Symptom selection was guided by 3 reference sources covering primary care curricula from different countries, which were chosen to capture common clinical presentations encountered in general practice: the Royal College of General Practitioners curriculum [<xref ref-type="bibr" rid="ref32">32</xref>], the Japanese Primary Care curriculum [<xref ref-type="bibr" rid="ref33">33</xref>], and the symptom list is from the textbook <italic>Primary Care: A Case-Based Office Evaluation</italic> [<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>Two distinct clinical cases were created or selected for each symptom, each corresponding to a different final diagnosis. This resulted in a total of 66 standardized cases. Case selection prioritized clinical variety while maintaining presentation patterns representative of outpatient encounters.</p><p>The cases were primarily adapted from 4 core textbooks emphasizing case-based reasoning in general and family medicine: <italic>100 Cases in Clinical Medicine</italic> [<xref ref-type="bibr" rid="ref35">35</xref>], <italic>Bates&#x2019; Guide to Physical Examination and History Taking: Case Studies</italic> [<xref ref-type="bibr" rid="ref36">36</xref>], <italic>The Patient History: An Evidence-Based Approach</italic> [<xref ref-type="bibr" rid="ref30">30</xref>], and <italic>Case Files: Family Medicine</italic> [<xref ref-type="bibr" rid="ref37">37</xref>]</p><p>When suitable examples were insufficient in the selected textbooks, supplementary cases were adapted from journal case reports or published case studies, yielding a total of 9 cases [<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref46">46</xref>]. All cases were reviewed by physicians to ensure that they contained the key elements of history-taking typically expected of a clinician when making a diagnosis. See <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for the complete list of 33 target symptoms and 2 representative diagnoses for each symptom. Each case was further summarized into a structured checklist for the evaluation of chatbot outputs.</p></sec><sec id="s2-4-2"><title>Checklist Creation</title><p>A structured checklist was developed to assess whether the chatbot asked clinically relevant questions for each SP case. The checklist was designed to capture the completeness and clinical appropriateness of the chatbot&#x2019;s biomedical inquiry, based on typical history-taking expectations in primary care.</p><p>Each checklist was organized into six domains as follows:</p><list list-type="order"><list-item><p>History of presenting complaint (HPC): onset, location, duration, character or quality, severity, progression, aggravating and relieving factors, treatments and their effectiveness, and any related trauma or accidents.</p></list-item><list-item><p>ASs: other symptoms mentioned in the case, whether present or specifically denied.</p></list-item><list-item><p>PH: existing conditions, comorbidities, and ongoing treatments.</p></list-item><list-item><p>FH: familial diseases or hereditary risks.</p></list-item><list-item><p>Social history (SH): history of smoking, alcohol consumption, lifestyle, and occupation</p></list-item><list-item><p>Other contextual information: additional relevant findings or case-specific details, including elaboration on AS when applicable.</p></list-item></list><p>Two physicians (a general practitioner and a family medicine resident) independently extracted checklist components from each reference case following a written protocol. The protocol specified that items were to be marked only if explicitly or implicitly mentioned in the case description, ensuring per case standardization rather than reliance on general expectations of history-taking. Discrepancies were reviewed, and final decisions were reached by consensus. An example of the created checklists is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-5"><title>Settings</title><p>The study was conducted in a controlled simulation environment using 6 SPs.</p><p>Six Thai nonmedical SPs with advanced proficiency in conversational English participated in the simulations. Each SP was assigned a fixed subset of 11 clinical cases, and each assigned case was portrayed by the same SP across all chatbot configurations. This approach was used to maintain consistent case portrayal across configurations while distributing the workload among SPs. Before data collection, SPs received standardized written instructions and a short orientation session. They were instructed to read the assigned case scenarios carefully, respond strictly in accordance with the provided case information, and avoid adding information not included in the scenario. For questions not explicitly addressed in the case script, SPs were instructed to avoid unsupported improvisation and to use &#x201C;unknown/not specified&#x201D; as the default response. If an &#x201C;unknown/not specified&#x201D; response seemed clinically implausible or inconsistent with the assigned case scenario, SPs were instructed to consult the researcher before responding. In such cases, the researcher confirmed whether the SP should use the default &#x201C;unknown/not specified&#x201D; response or provide a brief answer consistent with the general clinical features of the assigned case, without adding new case-specific information.</p><p>To reduce order effects, the order of chatbot configurations was randomized using a balanced rotation scheme generated by the researcher. The rotation was designed so that each configuration appeared in different ordinal positions across cases. SPs were not informed of the chatbot configuration labels during the simulations.</p><p>All simulations were performed locally on a MacBook Pro (Apple M4 Max, 128 GB of unified memory) to ensure consistent performance across all chatbot configurations.</p><p>Each clinical case was run once for each chatbot configuration, resulting in one transcript per case-configuration pair.</p></sec><sec id="s2-6"><title>Evaluation</title><sec id="s2-6-1"><title>Primary Evaluation</title><p>All chatbot-SP conversations were exported as structured text logs for independent physician evaluation. For the choice baseline, each fixed multiple-choice interaction was converted into a chronological text transcript containing the system questions, available response options, and the SP-selected option. The selected response was explicitly marked, and the resulting transcript was evaluated using the same case-specific checklist used for the LLM transcripts.</p><p>Before physician evaluation, each transcript was saved using a randomly assigned 6-digit transcript ID. The correspondence among the transcript ID, chatbot configuration, and case number was stored separately and not provided to physician raters. Therefore, physician raters were blinded to the LLM configuration labels during transcript review. However, the choice baseline could not be fully blinded because its fixed multiple-choice interaction format differed visibly from the open-ended LLM transcripts. This limitation was considered when interpreting comparisons involving the choice baseline.</p><p>Three physicians with prior OSCE evaluation experience independently reviewed the transcripts to assess coverage completeness, defined as the proportion of items on from the reference checklist that were appropriately elicited by the chatbot.</p><p>Each checklist item was marked as covered or not covered. When disagreements occurred, the final score for each item was determined by majority consensus (2 of the 3 raters). Coverage percentage for each case was then calculated as:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtext>Coverage</mml:mtext><mml:mtext>&#x00A0;</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="normal">%</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mtext>Number of covered items</mml:mtext><mml:mtext>Total items in checklist</mml:mtext></mml:mfrac><mml:mo>&#x00D7;</mml:mo><mml:mn>100</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>For each chatbot configuration, coverage percentages were macroaveraged across the 66 cases.</p></sec><sec id="s2-6-2"><title>Subgroup Analysis by Question Type</title><p>Coverage was further examined by question type, corresponding to the checklist domains of HPC, ASs, PH, FH, and SH.</p><p>For the subgroup analysis by question type, each domain-specific analysis included only cases containing at least one checklist item in that domain. Therefore, the number of cases differed across domains because not all reference cases included PH, FH, SH, or AS items. Domain-specific coverage was calculated as the number of covered items divided by the total number of checklist items within that domain for each eligible case. The mean and SD of these case-level coverage percentages were then calculated for each chatbot mode within each domain.</p></sec><sec id="s2-6-3"><title>Symptom-Based Analysis</title><p>To explore variation across clinical presentations, coverage results were further summarized by symptom. We identified the highest and lowest macroaverage coverage percentages per symptom across all modes, and the largest advantage (&#x0394; coverage) of the DT mode compared with the average of the other 3 LLM configurations (DN, MT, and MN).</p><p>This approach allowed the examination of symptom-level performance patterns and the identification of clinical scenarios in which the combination of detailed prompting and the thinking mode had the greatest impact.</p></sec><sec id="s2-6-4"><title>Repetitive Questioning Analysis</title><p>To further characterize chatbot behavior beyond checklist coverage, 1 researcher conducted an exploratory descriptive review of repetitive questioning in the 4 open-ended LLM configurations. Repetitive questioning was defined as asking the same or substantially similar questions more than once despite a clear prior answer from the SP. Repeated questioning was not counted when the SP had not clearly answered the previous question, such as when a multipart question was only partially answered. This analysis was descriptive and exploratory.</p></sec><sec id="s2-6-5"><title>Interrater Agreement</title><p>Agreement among the three physician evaluators on item-level coverage was assessed using 2 complementary metrics: (1) the proportion of items with full agreement across all raters and (2) Fleiss &#x03BA;, calculated with the <italic>statsmodels</italic> library in Python (version 3.13.3).</p></sec></sec><sec id="s2-7"><title>Statistical Analysis</title><p>To formally examine the factorial effects among the 4 LLM configurations, a 2-way repeated-measures ANOVA was conducted with prompt detail and internal reasoning as within-case factors. The choice chatbot was excluded from this factorial analysis. To compare all evaluated configurations, including the choice baseline, a 1-way repeated-measures ANOVA was additionally conducted across the 5 configurations, followed by post hoc 2-tailed paired <italic>t</italic> tests with Bonferroni correction. Sphericity was assessed using the Mauchly test, and for repeated-measures ANOVA with more than 2 levels, Greenhouse&#x2013;Geisser-corrected <italic>P</italic>-values were used when the sphericity assumption was violated. Effect sizes reported for the factorial 2-way repeated-measures ANOVA and the overall 1-way repeated-measures ANOVA were expressed as generalized eta squared (&#x03B7;&#x00B2;<sub>G</sub>), and pairwise effect sizes were reported as Hedge <italic>g</italic>.</p><p>All analyses were conducted in a Jupyter Notebook (Project Jupyter) using Python 3.13.3 and <italic>Pingouin</italic> package, with statistical significance set at <italic>P</italic>&#x003C;.05.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>This study was considered exempt from formal ethics review under the Ethical Guidelines for Medical and Biological Research Involving Human Subjects in Japan [<xref ref-type="bibr" rid="ref47">47</xref>] because it used standardized fictional or publicly available case scenarios that did not contain information relating to identifiable individuals and because it involved SPs who were nonmedical volunteers recruited outside clinical care settings and who role-played predefined cases without providing personal health information. All SPs were informed of the study&#x2019;s purpose and procedures before participating and agreed to participate voluntarily. Transcript data were labeled using randomly assigned study IDs, and no personally identifiable information was included in the evaluation files, figures, or supplementary materials. SPs received a small honorarium for their participation. No identifiable participant images or personal information is included in the manuscript or supplementary materials.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Interrater Agreement</title><p>To assess the reliability of grading across evaluators, interrater agreement was calculated based on all checklist items. The evaluators achieved an overall full agreement rate of 81.90%, indicating a high level of consistency in their assessments. The Fleiss &#x03BA; value was 0.753, representing substantial agreement according to the Landis and Koch Benchmark Scale [<xref ref-type="bibr" rid="ref48">48</xref>]. These results suggest that the evaluation criteria were well-defined and consistently applied among raters.</p></sec><sec id="s3-2"><title>Comparison of Coverage Across Chatbot Modes</title><p>Across all 66 SP cases, coverage rates varied substantially among the 5 chatbot configurations (<xref ref-type="table" rid="table1">Table 1</xref>). The DT mode achieved the highest mean coverage (72.3%, SD 14.3%), followed by the MT mode (60.5%, SD 15.0%), the DN mode (59.8%, SD 13.6%), the MN mode (54.0%, SD 15.3%), and the choice chatbot (51.1%, SD 13.3%).</p><p>To formally examine the factorial effects among the 4 LLM configurations, a 2-way repeated-measures ANOVA was conducted with prompt detail and internal reasoning as within-case factors. This analysis showed significant main effects of prompt detail (<italic>F</italic><sub>1, 65</sub>=56.74; <italic>P</italic>&#x003C;.001; &#x03B7;&#x00B2;<sub>G</sub>=0.085) and internal reasoning (<italic>F</italic><sub>1, 65</sub>=47.45; <italic>P</italic>&#x003C;.001; &#x03B7;&#x00B2;<sub>G</sub>=0.096). A significant interaction between prompt detail and internal reasoning was also observed (<italic>F</italic><sub>1, 65</sub>=6.32; <italic>P</italic>=.01; &#x03B7;&#x00B2;<sub>G</sub>=0.011), indicating that the effect of internal reasoning differed according to prompt detail. The DT configuration achieved the highest overall coverage among the 4 LLM configurations.</p><p>To compare all evaluated configurations, including the choice baseline, a 1-way repeated-measures ANOVA was additionally conducted across the 5 configurations. Sphericity was assessed using Mauchly test and was not violated. Therefore, uncorrected results are reported. This analysis revealed a significant effect of chatbot mode on coverage (<italic>F</italic><sub>4, 260</sub>=36.88; <italic>P</italic>&#x003C;.001; &#x03B7;&#x00B2;<sub>G</sub>=0.21), indicating that the completeness of gathered information differed significantly across chatbot modes.</p><p>Post hoc pairwise <italic>t</italic> tests with Bonferroni correction (<xref ref-type="table" rid="table2">Table 2</xref>) showed that the DT mode significantly outperformed all other configurations. Among the remaining configurations, DN and MT did not differ significantly (<italic>P</italic>&#x003E;.99). However, both DN and MT achieved higher coverage than MN and choice. The MN and choice modes did not differ significantly (<italic>P</italic>&#x003E;.99).</p><p>Overall, these findings demonstrate that both prompt detail and internal reasoning are associated with improved information completeness, with their combination in the DT mode producing the greatest coverage.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Mean coverage percentages (SD) across 5 chatbot configurations in 66 simulated patient interviews. Each configuration represents a combination of prompt detail (detailed vs minimal) and internal reasoning (thinking vs nonthinking).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Chatbot mode</td><td align="left" valign="bottom">Prompt type</td><td align="left" valign="bottom">Reasoning mode</td><td align="left" valign="bottom" colspan="2">Average coverage (%), mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">DT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">Detailed</td><td align="left" valign="top">Thinking</td><td align="left" valign="top" colspan="2">72.3 (14.3)</td></tr><tr><td align="left" valign="top">DN<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">Detailed</td><td align="left" valign="top">Nonthinking</td><td align="left" valign="top" colspan="2">59.8 (13.6)</td></tr><tr><td align="left" valign="top">MT<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">Minimal</td><td align="left" valign="top">Thinking</td><td align="left" valign="top" colspan="2">60.5 (15.0)</td></tr><tr><td align="left" valign="top">MN<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">Minimal</td><td align="left" valign="top">Nonthinking</td><td align="left" valign="top" colspan="2">54.0 (15.3)</td></tr><tr><td align="left" valign="top" colspan="3">Choice</td><td align="left" valign="top" colspan="2">51.1 (13.3)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>DT: detailed prompt with thinking mode.</p></fn><fn id="table1fn2"><p><sup>b</sup>DN: detailed prompt with nonthinking mode.</p></fn><fn id="table1fn3"><p><sup>c</sup>MT: minimal prompt with thinking mode.</p></fn><fn id="table1fn4"><p><sup>d</sup>MN: minimal prompt with nonthinking mode.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Post hoc pairwise <italic>t</italic> test comparisons between chatbot configurations using a repeated-measures design. Bonferroni correction was applied for multiple comparisons.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison (A)</td><td align="left" valign="bottom">Comparison (B)</td><td align="left" valign="bottom"><italic>2-tailed t</italic> test (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic> (Bonferroni-corrected)</td><td align="left" valign="bottom">Hedge <italic>g</italic></td></tr></thead><tbody><tr><td align="left" valign="top">DT<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">DN<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">7.20 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.88</td></tr><tr><td align="left" valign="top">DT</td><td align="left" valign="top">MT<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">7.05 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.80</td></tr><tr><td align="left" valign="top">DT</td><td align="left" valign="top">MN<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">9.40 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.22</td></tr><tr><td align="left" valign="top">DT</td><td align="left" valign="top">Choice</td><td align="left" valign="top">9.71 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">1.53</td></tr><tr><td align="left" valign="top">DN</td><td align="left" valign="top">MT</td><td align="left" valign="top">&#x2212;0.39 (65)</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">&#x2212;0.05</td></tr><tr><td align="left" valign="top">DN</td><td align="left" valign="top">MN</td><td align="left" valign="top">3.50 (65)</td><td align="left" valign="top">.008</td><td align="left" valign="top">0.40</td></tr><tr><td align="left" valign="top">DN</td><td align="left" valign="top">Choice</td><td align="left" valign="top">4.52 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.65</td></tr><tr><td align="left" valign="top">MT</td><td align="left" valign="top">MN</td><td align="left" valign="top">3.41 (65)</td><td align="left" valign="top">.01</td><td align="left" valign="top">0.42</td></tr><tr><td align="left" valign="top">MT</td><td align="left" valign="top">Choice</td><td align="left" valign="top">4.31 (65)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.66</td></tr><tr><td align="left" valign="top">MN</td><td align="left" valign="top">Choice</td><td align="left" valign="top">1.48 (65)</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">0.21</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>DT: detailed prompt with thinking mode.</p></fn><fn id="table2fn2"><p><sup>b</sup>DN: detailed prompt with nonthinking mode.</p></fn><fn id="table2fn3"><p><sup>c</sup>MT: minimal prompt with thinking mode.</p></fn><fn id="table2fn4"><p><sup>d</sup>MN: minimal prompt with nonthinking mode.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Subgroup Analysis by Question Type</title><p>Coverage was further examined across 5 question types based on mean coverage: HPC, ASs, PH, FH, and SH (<xref ref-type="table" rid="table3">Table 3</xref>). The DT mode achieved the highest coverage in most question categories, including HPC, AS, FH, and SH; however, the choice baseline showed slightly higher coverage for PH. The DN and MT modes showed intermediate performance, while MN and the choice chatbot demonstrated the lowest coverage overall, except for the high PH coverage observed in the choice baseline.</p><p>Marked variation was observed across question types: HPC and AS achieved relatively high coverage in most modes, whereas PH, FH, and SH showed greater discrepancies among configurations.</p><p>A series of 1-way repeated-measures ANOVAs revealed significant differences in coverage among chatbot modes for all question types. Sphericity was assessed using Mauchly test, and Greenhouse-Geisser-corrected <italic>P</italic>-values were used when sphericity was violated. Significant differences were observed for all question types: HPC (n=66; <italic>F</italic><sub>4, 260</sub>=12.06; <italic>P</italic>&#x003C;.001), AS (n=64; <italic>F</italic><sub>4, 252</sub>=7.69; <italic>P</italic>&#x003C;.001), PH (n=57; <italic>F</italic><sub>4, 224</sub>=34.39; <italic>P</italic>&#x003C;.001), FH (n=28; <italic>F</italic><sub>4, 108</sub>=50.12; <italic>P</italic>&#x003C;.001), and SH (n=57; <italic>F</italic><sub>4, 224</sub>=13.47; <italic>P</italic>&#x003C;.001).</p><p>Significant mode-related differences were consistent across all question types, with the largest effects observed for FH and PH.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Mean coverage percentages (SD) for each chatbot configuration by question type<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question type</td><td align="left" valign="bottom">Cases, n</td><td align="left" valign="bottom">DT (%), mean (SD)<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom">DN (%), mean (SD)<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="bottom">MT (%), mean (SD)<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="bottom">MN (%), mean (SD)<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="bottom">Choice (%), mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">HPC<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="left" valign="top">66</td><td align="left" valign="top">79.0 (19.7)</td><td align="left" valign="top">77.3 (20.2)</td><td align="left" valign="top">69.5 (26.0)</td><td align="left" valign="top">60.4 (28.7)</td><td align="left" valign="top">62.6 (22.0)</td></tr><tr><td align="left" valign="top">AS<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">64</td><td align="left" valign="top">80.1 (22.9)</td><td align="left" valign="top">76.0 (22.5)</td><td align="left" valign="top">78.7 (25.0)</td><td align="left" valign="top">75.8 (23.2)</td><td align="left" valign="top">64.7 (26.2)</td></tr><tr><td align="left" valign="top">PH<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="left" valign="top">57</td><td align="left" valign="top">93.0 (19.9)</td><td align="left" valign="top">36.0 (48.0)</td><td align="left" valign="top">53.5 (48.1)</td><td align="left" valign="top">45.6 (46.6)</td><td align="left" valign="top">94.7 (15.5)</td></tr><tr><td align="left" valign="top">FH<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">28</td><td align="left" valign="top">98.2 (9.4)</td><td align="left" valign="top">7.1 (26.2)</td><td align="left" valign="top">58.9 (49.2)</td><td align="left" valign="top">21.4 (41.8)</td><td align="left" valign="top">3.6 (18.9)</td></tr><tr><td align="left" valign="top">SH<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup></td><td align="left" valign="top">57</td><td align="left" valign="top">45.3 (38.4)</td><td align="left" valign="top">12.0 (25.7)</td><td align="left" valign="top">19.3 (31.8)</td><td align="left" valign="top">21.1 (35.1)</td><td align="left" valign="top">11.4 (27.3)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Sample sizes differ because the analyses were restricted to cases containing at least one checklist item in each domain.</p></fn><fn id="table3fn2"><p><sup>b</sup>DT: detailed prompt with thinking mode.</p></fn><fn id="table3fn3"><p><sup>c</sup>DN: detailed prompt with nonthinking mode.</p></fn><fn id="table3fn4"><p><sup>d</sup>MT: minimal prompt with thinking mode.</p></fn><fn id="table3fn5"><p><sup>e</sup>MN: minimal prompt with nonthinking mode.</p></fn><fn id="table3fn6"><p><sup>f</sup>HPC: history of presenting complaint.</p></fn><fn id="table3fn7"><p><sup>g</sup>AS: associated symptom.</p></fn><fn id="table3fn8"><p><sup>h</sup>PH: past medical history.</p></fn><fn id="table3fn9"><p><sup>i</sup>FH: family history.</p></fn><fn id="table3fn10"><p><sup>j</sup>SH: social history.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Symptom-Based Analysis</title><p>The highest and lowest macroaveraged coverage percentages per symptom across all modes were examined. To explore variation across clinical presentations, coverage was further summarized by symptoms (<xref ref-type="table" rid="table4">Table 4</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Top 3 and bottom 3 symptoms ranked by mean macroaverage coverage across all chatbot modes (n=33 symptoms).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Rank</td><td align="left" valign="bottom">Symptom</td><td align="left" valign="bottom">Average coverage (%), mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">Highest coverage</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">Palpitations</td><td align="left" valign="top">77.60 (14.50)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">Joint pain</td><td align="left" valign="top">75.23 (6.77)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">Abdominal pain</td><td align="left" valign="top">72.83 (8.78)</td></tr><tr><td align="left" valign="top" colspan="3">Lowest coverage</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>31</td><td align="left" valign="top">Fever</td><td align="left" valign="top">47.79 (9.13)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>32</td><td align="left" valign="top">Impaired vision</td><td align="left" valign="top">46.11 (5.76)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>33</td><td align="left" valign="top">Nausea and vomiting</td><td align="left" valign="top">46.07 (6.34)</td></tr></tbody></table></table-wrap><p>Across all chatbot modes, the symptoms with the highest average coverage were palpitations (77.60%, SD 14.50%), joint pain (75.23%, SD 6.77%), and abdominal pain (72.83%, SD 8.78%).</p><p>In contrast, the lowest-coverage symptoms were nausea and vomiting (46.07%, SD 6.34%), impaired vision (46.11%, SD 5.76%), and fever (47.79%, SD 9.13%).</p><p>These results highlight substantial variation in information completeness across presenting complaints.</p><p>The full list of 33 symptoms is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref></p><p>We then identified the largest advantage (&#x0394; coverage) of the DT mode over the average of the other 3 LLM configurations (DN, MT, and MN).</p><p>To identify the symptom types with the greatest improvement under the DT configuration, we calculated the difference in mean coverage between DT and the average of the other 3 LLM modes (DN, MT, and MN).</p><p>The 5 symptoms showing the largest relative advantage of DT were weakness, headache, dysphagia, palpitations, and dizziness. Among these 5 symptoms with the largest DT advantage, coverage improvements ranged from approximately 22.4%-30.6% points compared with the average of the other LLM configurations. Pairwise differences between DT and individual LLM configurations are shown in <xref ref-type="table" rid="table5">Table 5</xref>.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Top 5 symptoms showing the largest relative advantage of the detailed prompt with the thinking mode (DT) configuration in coverage percentage compared with other large language model (LLM) configurations.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Symptom</td><td align="left" valign="bottom">&#x0394; coverage (DT<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> - Mean of DN, MT, and MN) (%)</td><td align="left" valign="bottom">DT-DN<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup> (%)</td><td align="left" valign="bottom">DT-MT<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> (%)</td><td align="left" valign="bottom">DT-MN<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup> (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Weakness</td><td align="left" valign="top">30.6</td><td align="left" valign="top">30.3</td><td align="left" valign="top">21.2</td><td align="left" valign="top">40.4</td></tr><tr><td align="left" valign="top">Headache</td><td align="left" valign="top">30.0</td><td align="left" valign="top">30.0</td><td align="left" valign="top">35.0</td><td align="left" valign="top">25.0</td></tr><tr><td align="left" valign="top">Dysphagia</td><td align="left" valign="top">28.2</td><td align="left" valign="top">25.8</td><td align="left" valign="top">14.6</td><td align="left" valign="top">44.2</td></tr><tr><td align="left" valign="top">Palpitations</td><td align="left" valign="top">26.8</td><td align="left" valign="top">33.2</td><td align="left" valign="top">15.4</td><td align="left" valign="top">31.7</td></tr><tr><td align="left" valign="top">Dizziness</td><td align="left" valign="top">22.4</td><td align="left" valign="top">22.6</td><td align="left" valign="top">17.9</td><td align="left" valign="top">26.8</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>DT: detailed prompt with thinking mode.</p></fn><fn id="table5fn2"><p><sup>b</sup>DN: detailed prompt with nonthinking mode.</p></fn><fn id="table5fn3"><p><sup>c</sup>MT: minimal prompt with thinking mode.</p></fn><fn id="table5fn4"><p><sup>d</sup>MN: minimal prompt with nonthinking mode.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Repetitive Questioning Analysis</title><p>In an exploratory review of repetitive questioning, repeated questions were observed in 4 of 66 (6.1%) DT transcripts, 7 of 66 (10.6%) DN transcripts, 3 of 66 (4.5%) MT transcripts, and 10 of 66 (15.2%) MN transcripts. Repetitive questioning was uncommon overall but occurred more often in nonthinking configurations than in thinking configurations, most frequently in the MN configuration.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study demonstrates that LLM-based chatbot interviews are capable of collecting patient-reported medical history across multiple clinical areas in a controlled, simulated setting. Across the evaluated settings, LLM-based chatbots achieved overall information coverage comparable to or exceeding that of the existing choice-based system, supporting their feasibility as flexible tools for collecting baseline patient history. Notably, the MN configuration performed at a level comparable to the choice-based baseline. This indicates that the use of an LLM alone, without structured prompting or reasoning support, does not inherently provide an advantage. Instead, the completeness and clinical relevance of the collected information, as reflected by case-based checklist coverage, were strongly influenced by prompt design and the presence of internal reasoning mechanisms.</p><p>Among the evaluated configurations, the DT mode achieved the highest overall coverage. Compared with other LLM configurations and the choice-based baseline, DT achieved an absolute improvement in information coverage of approximately 12&#x2010;21 percentage points at a fixed interaction length. The factorial analysis showed additional significant main effects of prompt detail and internal reasoning, as well as a significant interaction between prompt detail and internal reasoning. These findings suggest that detailed prompting and internal reasoning each contributed to improved coverage, and that their combination was associated with the greatest information capture in this controlled simulated setting.</p><p>Prompt detail appeared to play an important role in establishing a clinically coherent interview structure. While the underlying language model could collect relevant patient information, detailed prompts improved the consistency and depth of history taking. Configurations lacking sufficient prompt structure tended to produce less detailed information in several history components, resulting in an overall reduction in coverage. These findings suggest that prompt detail functions as a clinical framework, aligning the model&#x2019;s flexible behavior with established medical interviewing frameworks and supporting more reliable information collection.</p><p>Internal reasoning mechanisms may have further contributed by enabling prioritization and adaptive follow-up questioning. When internal reasoning was absent, chatbots tended to remain within the same history domain that was emphasized early in the prompt, resulting in limited progression to subsequent components. In contrast, reasoning-enabled configurations appeared better able to identify clinically relevant information and pursue follow-up questions that extended beyond the initial domain, such as PH, FH, and SH. These findings suggest that internal reasoning may support dynamic decision-making within the interview, complementing structured prompts by guiding when and how to transition between history components. In the exploratory review of repetitive questioning, repeated questions were observed slightly more often in nonthinking configurations than in thinking configurations. This pattern may suggest that internal reasoning helped the chatbot make more consistent use of prior conversational context in some cases. However, because the internal reasoning process itself was not directly analyzed, this interpretation should be considered a possible explanation rather than a confirmed mechanism. Further qualitative error analysis of transcript-level behavior would be needed to better examine how reasoning-mode configurations influence question selection, domain transitions, and use of prior conversational context.</p><p>The low coverage observed in SH, even in the DT configuration, should be interpreted in relation to the alignment between the prompt&#x2019;s scope and the checklist criteria. To prevent the chatbot from drifting into overly broad or nonclinical lifestyle discussions, social history questioning was limited to key clinically relevant risk factors, specifically smoking and alcohol consumption. Consequently, lower SH coverage may partly reflect a mismatch between the prompt&#x2019;s scope and the grading checklist, rather than a general inability of the chatbot to collect social history. While this design choice supported a focused interview, it resulted in limited coverage of other SH subcategories included in the grading checklist that were not explicitly prompted. This limitation reflects broader challenges in capturing social history information, as SH sections can contain a wide range of social, behavioral, and environmental determinants, yet they are variably documented in practice [<xref ref-type="bibr" rid="ref49">49</xref>]. In real-world use, the scope of social history questioning may need to be adapted according to the clinical scenario, specialty, or intended use of the chatbot in clinical care.</p><p>Performance varied across clinical scenarios. Symptoms linked to specific organ systems and common diagnostic pathways, such as palpitations, joint pain, and abdominal pain, were associated with higher coverage, whereas systemic or nonspecific presentations, such as fever and nausea and vomiting, were more challenging because they may require broader and less linear exploration across multiple organ systems. This suggests that a single static interviewing strategy is unlikely to perform optimally across all contexts and that adaptation to the presenting complaint may be required. The largest gains with the DT mode were observed in cases containing distinctive diagnostic cues (eg, transient visual loss in suspected transient ischemic attack or thunderclap onset in subarachnoid hemorrhage). In these scenarios, the combination of detailed prompting and internal reasoning may have helped the chatbot ask more targeted follow-up questions, although this mechanism was not directly tested.</p><p>Taken together, these results suggest that chatbot-based systems may be capable of collecting clinically relevant patient history information in controlled, simulated settings, supporting their potential role in preconsultation data gathering. The findings further suggest that the reliability and completeness of this information depend on the integration of multiple design elements, including sufficiently detailed prompts to provide clinical structure and internal reasoning mechanisms to support prioritization and progression through the interview. In practice, such systems are best positioned as supportive tools that augment baseline information collection alongside clinician-led interviews.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Consistent with previous studies [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], our results support the feasibility of using chatbots to collect patient medical histories prior to physician encounters. Beyond demonstrating feasibility, this study extends prior work by quantitatively evaluating information coverage across multiple clinical domains and symptom types. The findings further suggest that structured prompt design and the inclusion of internal reasoning mechanisms, as explored in earlier studies [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref25">25</xref>], can support clinically oriented questioning, leading to more comprehensive and diagnostically relevant information collection.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, the fixed 15-turn limit may have truncated deeper questioning, potentially underestimating the full capabilities of the chatbot configurations, particularly for later history domains such as PH, FH, and social history. Because this limit was implemented externally and was not explicitly stated in the chatbot prompts, the models could not anticipate the remaining number of turns when prioritizing questions. Lower coverage in these later domains may therefore partly reflect the predefined conversation-length limit rather than the model&#x2019;s full potential capabilities. In addition, this study did not formally analyze how conversation length or stopping point affected performance.</p><p>Second, the results are based solely on the Qwen3-14B-Instruct model. While this model has demonstrated strong reasoning capabilities, the findings regarding &#x201C;thinking mode&#x201D; may vary across different model architectures or larger-scale models.</p><p>Third, only a single detailed prompt design was evaluated, and the results may not generalize to other prompt styles or instructional formulations.</p><p>Fourth, the symptom-level analyses were based on a limited number of simulated cases (2 cases per symptom), which may not fully capture the heterogeneity of real-world clinical presentations.</p><p>Fifth, the evaluation was conducted using SP cases in a controlled setting, and chatbot performance may differ in real-world clinical environments.</p><p>Sixth, the low SH coverage should be interpreted as a limitation in prompt-rubric alignment. It may partly reflect the evaluation design rather than an inherent inability of the model to collect social history information.</p><p>Seventh, the choice chatbot differed from the LLM configurations in its design objectives, interface format, branching logic, and stopping rules. Therefore, it should be interpreted as a pragmatic baseline reference rather than a fully equivalent comparator. In addition, although physician raters were blinded to LLM configuration labels, the choice baseline could not be fully blinded during evaluation because its fixed multiple-choice format was visibly different from the open-ended format of the LLM transcripts.</p><p>Finally, the outcome in this study was checklist-based information coverage, which remains a surrogate measure. Because the checklists were derived from standardized case information, the coverage score primarily reflected whether the chatbot elicited expected case-specific information, rather than whether it performed an optimal or fully individualized clinical interview. Higher coverage does not necessarily indicate better diagnostic accuracy, clinical safety, patient experience, conversational efficiency, or real-world usefulness.</p></sec><sec id="s4-4"><title>Future Directions</title><p>Future work should evaluate chatbot performance in real clinical settings, where patient perceptions, communication styles, and time constraints vary more widely. Integrating chatbot-collected histories with clinician documentation workflows and examining the impact of this integration on documentation quality and patient-physician information alignment will be important next steps. Future studies should also examine how chatbot-collected histories can be summarized and presented to clinicians, whether findings generalize to other LLMs, including larger frontier models, and how conversation length, adaptive stopping strategies, and transcript-level error patterns, such as repetitive questioning, missed transitions, and irrelevant follow-up questions, influence information coverage and conversational quality. Future simulation-based studies should also consider analytical approaches that account for simulated-patient-level variability, as well as clinical safety outcomes, such as hallucination-related errors and potentially unsafe responses.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study evaluated the effects of prompt detail and internal reasoning on checklist-based information coverage in LLM-based chatbot medical history taking using standardized primary care cases. In this controlled simulated setting, the LLM chatbot was able to collect patient history information across multiple clinical domains, with the detailed prompt and thinking mode achieving the highest overall coverage among the evaluated configurations. The results indicate that both structured clinical prompt design and internal reasoning configuration influence the completeness of chatbot-collected patient history information, with performance varying across symptom types. These findings support the feasibility of LLM-based chatbots for preconsultation history collection within the scope of checklist-based evaluation in a controlled simulated setting.</p></sec></sec></body><back><ack><p>During manuscript preparation, ChatGPT (OpenAI) was used to assist with language editing and to improve the readability of the author-drafted text. The authors have reviewed, edited, and verified all AI-assisted text and take full responsibility for the manuscript&#x2019;s final content.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Cross-Ministerial Strategic Innovation Promotion Program (SIP) on &#x201C;Integrated Health Care System&#x201D; Grant Number JPJ012425.</p></sec><sec><title>Data Availability</title><p>The chatbot implementation code used at the time of the study is available at the GitHub repository [<xref ref-type="bibr" rid="ref50">50</xref>]. The analysis-ready aggregate dataset generated and analyzed during this study is available from the corresponding author upon reasonable request. Full chatbot-SP transcripts, case scripts, and the complete set of detailed checklist items are not publicly available because they may contain participant-generated wording and case-specific details derived from copyrighted or licensed source materials.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: NT, CL, GY, KK, YM</p><p>Formal Analysis: NT</p><p>Funding Acquisition: TK</p><p>Methodology: NT, CL, GY, KK</p><p>Project Administration: TK</p><p>Software: NT</p><p>Supervision: CL, GY</p><p>Validation: NT</p><p>Writing &#x2013; original draft: NT</p><p>Writing &#x2013; review &#x0026; editing: NT, CL, GY, KK, YM, TK</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AS</term><def><p>associated symptom</p></def></def-item><def-item><term id="abb2">DN</term><def><p>detailed prompt with nonthinking mode</p></def></def-item><def-item><term id="abb3">DT</term><def><p>detailed prompt with thinking mode</p></def></def-item><def-item><term id="abb4">FH</term><def><p>family history</p></def></def-item><def-item><term id="abb5">HPC</term><def><p>history of presenting complaint</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">MN</term><def><p>minimal prompt with nonthinking mode</p></def></def-item><def-item><term id="abb8">MT</term><def><p>minimal prompt with thinking mode</p></def></def-item><def-item><term id="abb9">OSCE</term><def><p>Objective Structured Clinical Examination</p></def></def-item><def-item><term id="abb10">PH</term><def><p>past medical history</p></def></def-item><def-item><term id="abb11">SH</term><def><p>social history</p></def></def-item><def-item><term id="abb12">SP</term><def><p>simulated patient</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Petrie</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Jago</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Devcich</surname><given-names>DA</given-names> </name></person-group><article-title>The role of illness perceptions in patients with medical conditions</article-title><source>Curr Opin Psychiatry</source><year>2007</year><month>03</month><volume>20</volume><issue>2</issue><fpage>163</fpage><lpage>167</lpage><pub-id pub-id-type="doi">10.1097/YCO.0b013e328014a871</pub-id><pub-id pub-id-type="medline">17278916</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Asanad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Asanad</surname><given-names>K</given-names> </name><name name-style="western"><surname>Karanjia</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sadun</surname><given-names>AA</given-names> </name></person-group><article-title>Value of medical history in ophthalmology: a study of diagnostic accuracy</article-title><source>J Curr Ophthalmol</source><year>2018</year><month>12</month><volume>30</volume><issue>4</issue><fpage>359</fpage><lpage>364</lpage><pub-id pub-id-type="doi">10.1016/j.joco.2018.09.001</pub-id><pub-id pub-id-type="medline">30555971</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bickley</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Szilagyi</surname><given-names>PG</given-names> </name><name name-style="western"><surname>Hoffman</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Soriano</surname><given-names>RP</given-names> </name></person-group><source>Bates&#x2019; Guide to Physical Examination and History Taking</source><year>2023</year><edition>13</edition><publisher-name>Lippincott Williams &#x0026; Wilkins</publisher-name><pub-id pub-id-type="other">9781975216054</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Loy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kowalsky</surname><given-names>R</given-names> </name></person-group><article-title>Narrative medicine: the power of shared stories to enhance inclusive clinical care, clinician well-being, and medical education</article-title><source>Perm J</source><year>2024</year><month>06</month><day>14</day><volume>28</volume><issue>2</issue><fpage>93</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.7812/TPP/23.116</pub-id><pub-id pub-id-type="medline">38225914</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weiner</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kelly</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>A</given-names> </name></person-group><article-title>How accurate is the medical record? A comparison of the physician&#x2019;s note with a concealed audio recording in unannounced standardized patient encounters</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>05</month><day>1</day><volume>27</volume><issue>5</issue><fpage>770</fpage><lpage>775</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa027</pub-id><pub-id pub-id-type="medline">32330258</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hardy</surname><given-names>V</given-names> </name><name name-style="western"><surname>Usher-Smith</surname><given-names>J</given-names> </name><name name-style="western"><surname>Archer</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Agreement between patient&#x2019;s description of abdominal symptoms of possible upper gastrointestinal cancer and general practitioner consultation notes: a qualitative analysis of video-recorded UK primary care consultation data</article-title><source>BMJ Open</source><year>2023</year><month>01</month><day>5</day><volume>13</volume><issue>1</issue><fpage>e058766</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2021-058766</pub-id><pub-id pub-id-type="medline">36604136</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weiner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Flanagan</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Ernst</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Accuracy, thoroughness, and quality of outpatient primary care documentation in the U.S. Department of Veterans Affairs</article-title><source>BMC Prim Care</source><year>2024</year><month>07</month><day>18</day><volume>25</volume><issue>1</issue><fpage>262</fpage><pub-id pub-id-type="doi">10.1186/s12875-024-02501-6</pub-id><pub-id pub-id-type="medline">39026167</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thawinwisan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yamamoto</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kishimoto</surname><given-names>K</given-names> </name><name name-style="western"><surname>Mori</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kuroda</surname><given-names>T</given-names> </name></person-group><article-title>Comparing patient questionnaires and physician documentation in a tertiary hospital setting: a retrospective analysis of subjective information</article-title><source>Inform Med Unlocked</source><year>2025</year><volume>59</volume><fpage>101707</fpage><pub-id pub-id-type="doi">10.1016/j.imu.2025.101707</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Berdahl</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Henreid</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Pevnick</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nuckols</surname><given-names>TK</given-names> </name></person-group><article-title>Digital tools designed to obtain the history of present illness from patients: scoping review</article-title><source>J Med Internet Res</source><year>2022</year><month>11</month><day>17</day><volume>24</volume><issue>11</issue><fpage>e36074</fpage><pub-id pub-id-type="doi">10.2196/36074</pub-id><pub-id pub-id-type="medline">36394945</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zakim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Braun</surname><given-names>N</given-names> </name><name name-style="western"><surname>Fritz</surname><given-names>P</given-names> </name><name name-style="western"><surname>Alscher</surname><given-names>MD</given-names> </name></person-group><article-title>Underutilization of information and knowledge in everyday medical practice: evaluation of a computer-based solution</article-title><source>BMC Med Inform Decis Mak</source><year>2008</year><month>11</month><day>5</day><volume>8</volume><issue>1</issue><fpage>50</fpage><pub-id pub-id-type="doi">10.1186/1472-6947-8-50</pub-id><pub-id pub-id-type="medline">18983684</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zakim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Brandberg</surname><given-names>H</given-names> </name><name name-style="western"><surname>El Amrani</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Computerized history-taking improves data quality for clinical decision-making-Comparison of EHR and computer-acquired history data in patients with chest pain</article-title><source>PLOS ONE</source><year>2021</year><volume>16</volume><issue>9</issue><fpage>e0257677</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0257677</pub-id><pub-id pub-id-type="medline">34570811</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Albrink</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hummers</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Concordance of data collected by an app for medical history taking and in-person interviews from patients in primary care</article-title><source>JAMIA Open</source><year>2024</year><month>12</month><volume>7</volume><issue>4</issue><fpage>ooae102</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooae102</pub-id><pub-id pub-id-type="medline">39386064</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cho</surname><given-names>J</given-names> </name><name name-style="western"><surname>Han</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>HY</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name></person-group><article-title>Enhancing clinical history taking through the implementation of a streamlined electronic questionnaire system at a pediatric headache clinic: development and evaluation study</article-title><source>JMIR Med Inform</source><year>2024</year><month>11</month><day>8</day><volume>12</volume><issue>1</issue><fpage>e54415</fpage><pub-id pub-id-type="doi">10.2196/54415</pub-id><pub-id pub-id-type="medline">39622694</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tudor Car</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dhinagaran</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Kyaw</surname><given-names>BM</given-names> </name><etal/></person-group><article-title>Conversational agents in health care: scoping review and conceptual analysis</article-title><source>J Med Internet Res</source><year>2020</year><month>08</month><day>7</day><volume>22</volume><issue>8</issue><fpage>e17158</fpage><pub-id pub-id-type="doi">10.2196/17158</pub-id><pub-id pub-id-type="medline">32763886</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Milne-Ives</surname><given-names>M</given-names> </name><name name-style="western"><surname>de Cock</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>E</given-names> </name><etal/></person-group><article-title>The effectiveness of artificial intelligence conversational agents in health care: systematic review</article-title><source>J Med Internet Res</source><year>2020</year><month>10</month><day>22</day><volume>22</volume><issue>10</issue><fpage>e20346</fpage><pub-id pub-id-type="doi">10.2196/20346</pub-id><pub-id pub-id-type="medline">33090118</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hindelang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sitaru</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zink</surname><given-names>A</given-names> </name></person-group><article-title>Transforming health care through chatbots for medical history-taking and future directions: comprehensive systematic review</article-title><source>JMIR Med Inform</source><year>2024</year><month>08</month><day>29</day><volume>12</volume><fpage>e56628</fpage><pub-id pub-id-type="doi">10.2196/56628</pub-id><pub-id pub-id-type="medline">39207827</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spinazze</surname><given-names>P</given-names> </name><name name-style="western"><surname>Aardoom</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chavannes</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kasteleyn</surname><given-names>M</given-names> </name></person-group><article-title>The computer will see you now: overcoming barriers to adoption of computer-assisted history taking (CAHT) in primary care</article-title><source>J Med Internet Res</source><year>2021</year><month>02</month><day>24</day><volume>23</volume><issue>2</issue><fpage>e19306</fpage><pub-id pub-id-type="doi">10.2196/19306</pub-id><pub-id pub-id-type="medline">33625360</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Negash</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gundlack</surname><given-names>J</given-names> </name><name name-style="western"><surname>Buch</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Physicians&#x2019; attitudes and acceptance towards artificial intelligence in medical care: a qualitative study in Germany</article-title><source>Front Digit Health</source><year>2025</year><volume>7</volume><fpage>1616827</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2025.1616827</pub-id><pub-id pub-id-type="medline">40727617</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Albrink</surname><given-names>K</given-names> </name><name name-style="western"><surname>Joos</surname><given-names>C</given-names> </name><name name-style="western"><surname>Schr&#x00F6;der</surname><given-names>D</given-names> </name><name name-style="western"><surname>M&#x00FC;ller</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hummers</surname><given-names>E</given-names> </name><name name-style="western"><surname>Noack</surname><given-names>EM</given-names> </name></person-group><article-title>Obtaining patients&#x2019; medical history using a digital device prior to consultation in primary care: study protocol for a usability and validity study</article-title><source>BMC Med Inform Decis Mak</source><year>2022</year><month>07</month><day>19</day><volume>22</volume><issue>1</issue><fpage>189</fpage><pub-id pub-id-type="doi">10.1186/s12911-022-01928-0</pub-id><pub-id pub-id-type="medline">35854290</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hauber</surname><given-names>R</given-names> </name><name name-style="western"><surname>Schirm</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lukas</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Computer-assisted medical history taking prior to patient consultation in the outpatient care setting: a prospective pilot project</article-title><source>BMC Health Serv Res</source><year>2024</year><month>12</month><day>18</day><volume>24</volume><issue>1</issue><fpage>1616</fpage><pub-id pub-id-type="doi">10.1186/s12913-024-12043-3</pub-id><pub-id pub-id-type="medline">39696381</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>SZY</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name></person-group><article-title>Critique of impure reason: unveiling the reasoning behaviour of medical large language models</article-title><source>Elife</source><year>2025</year><month>10</month><day>28</day><volume>14</volume><fpage>e106187</fpage><pub-id pub-id-type="doi">10.7554/eLife.106187</pub-id><pub-id pub-id-type="medline">41150728</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mo&#x00EB;ll</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sand Aronsson</surname><given-names>F</given-names> </name><name name-style="western"><surname>Akbar</surname><given-names>S</given-names> </name></person-group><article-title>Medical reasoning in LLMs: an in-depth analysis of DeepSeek R1</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1616145</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1616145</pub-id><pub-id pub-id-type="medline">40607450</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Prompt engineering in consistency and reliability with the evidence-based guideline for LLMs</article-title><source>NPJ Digit Med</source><year>2024</year><month>02</month><day>20</day><volume>7</volume><issue>1</issue><fpage>41</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01029-4</pub-id><pub-id pub-id-type="medline">38378899</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nayak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rangan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>JH</given-names> </name></person-group><article-title>Diagnostic reasoning prompts reveal the potential for large language model interpretability in medicine</article-title><source>NPJ Digit Med</source><year>2024</year><month>01</month><day>24</day><volume>7</volume><issue>1</issue><fpage>20</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01010-1</pub-id><pub-id pub-id-type="medline">38267608</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sonoda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kurokawa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hagiwara</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Structured clinical reasoning prompt enhances LLM&#x2019;s diagnostic capabilities in diagnosis please quiz cases</article-title><source>Jpn J Radiol</source><year>2025</year><month>04</month><volume>43</volume><issue>4</issue><fpage>586</fpage><lpage>592</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01712-2</pub-id><pub-id pub-id-type="medline">39625594</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Palepu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="medline">40205050</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paley</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zornitzki</surname><given-names>T</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Friedman</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kozak</surname><given-names>N</given-names> </name><name name-style="western"><surname>Schattner</surname><given-names>A</given-names> </name></person-group><article-title>Utility of clinical examination in the diagnosis of emergency department patients admitted to the department of medicine of an academic hospital</article-title><source>Arch Intern Med</source><year>2011</year><month>08</month><day>8</day><volume>171</volume><issue>15</issue><fpage>1394</fpage><lpage>1396</lpage><pub-id pub-id-type="doi">10.1001/archinternmed.2011.340</pub-id><pub-id pub-id-type="medline">21824956</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gormley</surname><given-names>G</given-names> </name></person-group><article-title>Summative OSCEs in undergraduate medical education</article-title><source>Ulster Med J</source><year>2011</year><month>09</month><volume>80</volume><issue>3</issue><fpage>127</fpage><lpage>132</lpage><pub-id pub-id-type="medline">23526843</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><access-date>2026-03-02</access-date><comment>Preprint posted online on  May 14, 2025</comment><comment><ext-link ext-link-type="uri" xlink:href="http://10.48550/arXiv.2505.09388">10.48550/arXiv.2505.09388</ext-link></comment></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Henderson</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Tierney Jr</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Smetana</surname><given-names>GW</given-names> </name></person-group><source>The Patient History: An Evidence-Based Approach to Differential Diagnosis</source><year>2012</year><edition>2</edition><publisher-name>McGraw-Hill Medical</publisher-name><pub-id pub-id-type="other">978-0-07-180420-2</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="web"><source>Ubie</source><year>2025</year><month>08</month><day>1</day><access-date>2026-07-31</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://ubiehealth.com">https://ubiehealth.com</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="web"><article-title>The clinical topic guides</article-title><source>Royal College of General Practitioners (RCGP)</source><access-date>2025-06-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.rcgp.org.uk/mrcgp-exams/gp-curriculum/clinical-topic-guides">https://www.rcgp.org.uk/mrcgp-exams/gp-curriculum/clinical-topic-guides</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="web"><article-title>Specialized training program [Article in Japanese]</article-title><source>Japan Primary Care Association</source><access-date>2025-06-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.primary-care.or.jp/nintei_tr/kouki_touroku.php">https://www.primary-care.or.jp/nintei_tr/kouki_touroku.php</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Goroll</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Mulley</surname><given-names>AG</given-names> </name></person-group><source>Primary Care Medicine: Office Evaluation and Management of the Adult Patient</source><year>2014</year><edition>7</edition><publisher-name>Wolters Kluwer</publisher-name><pub-id pub-id-type="other">978-1-4511-5149-7</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rees</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pattison</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kosky</surname><given-names>C</given-names> </name></person-group><source>100 Cases In Clinical Medicine</source><year>2014</year><edition>3</edition><publisher-name>CRC Press/Taylor &#x0026; Francis Group</publisher-name><pub-id pub-id-type="other">978-1-4441-7430-4</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Prabhu</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Bickley</surname><given-names>LS</given-names> </name></person-group><source>Case Studies to Accompany Bates&#x2019; Guide to Physical Examination and History Taking</source><year>2007</year><edition>9</edition><publisher-name>Lippincott Williams &#x0026; Wilkins</publisher-name><pub-id pub-id-type="other">9780781792219</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Toy</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Briscoe</surname><given-names>D</given-names> </name><name name-style="western"><surname>Britton</surname><given-names>B</given-names> </name><name name-style="western"><surname>Heidelbaugh</surname><given-names>JJ</given-names> </name></person-group><source>Case Files Family Medicine</source><year>2021</year><edition>5</edition><publisher-name>McGraw-Hill</publisher-name><pub-id pub-id-type="other">978-1-260-46860-1</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><person-group person-group-type="author"><collab>Institute of Medicine</collab></person-group><article-title>Case study 52: behavioral and audiologic manifestations of noise-induced hearing loss</article-title><source>Environmental Medicine: Integrating a Missing Element into Medical Education</source><year>1995</year><publisher-name>National Academies Press</publisher-name><fpage>868</fpage><lpage>871</lpage><pub-id pub-id-type="doi">10.17226/4795</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>French</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Benseler</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Birken</surname><given-names>CS</given-names> </name></person-group><article-title>Case 2: a teenage boy with epistaxis</article-title><source>Paediatr Child Health</source><year>2009</year><month>02</month><volume>14</volume><issue>2</issue><fpage>99</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1093/pch/14.2.99a</pub-id><pub-id pub-id-type="medline">19436560</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chowdhury</surname><given-names>R</given-names> </name><name name-style="western"><surname>Turkdogan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Silver</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Approach to epistaxis</article-title><source>J Otorhinolaryngol Hear Balance Med</source><year>2024</year><month>12</month><day>23</day><volume>5</volume><issue>2</issue><fpage>21</fpage><pub-id pub-id-type="doi">10.3390/ohbm5020021</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Trottier</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Massoud</surname><given-names>E</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>T</given-names> </name></person-group><article-title>A case of hoarseness and vocal cord immobility</article-title><source>CMAJ</source><year>2013</year><month>11</month><day>19</day><volume>185</volume><issue>17</issue><fpage>1520</fpage><lpage>1524</lpage><pub-id pub-id-type="doi">10.1503/cmaj.112112</pub-id><pub-id pub-id-type="medline">23695603</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Morrison</surname><given-names>J</given-names> </name></person-group><article-title>Persistent hoarseness: a case study</article-title><year>2011</year><access-date>2026-07-31</access-date><publisher-name>The Royal Australian College of General Practitioners (RACGP)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.racgp.org.au/getattachment/8a5cb549-e73a-436b-90f0-7b685a2f28cb/Persistent-hoarseness.aspx">https://www.racgp.org.au/getattachment/8a5cb549-e73a-436b-90f0-7b685a2f28cb/Persistent-hoarseness.aspx</ext-link></comment></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="web"><source>BMJ Best Practice</source><year>2025</year><month>07</month><day>10</day><access-date>2026-07-31</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://bestpractice.bmj.com/topics/en-gb/252/case-history">https://bestpractice.bmj.com/topics/en-gb/252/case-history</ext-link></comment></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="web"><article-title>Case 7: achalasia</article-title><source>American Society for Gastrointestinal Endoscopy (ASGE)</source><year>2025</year><month>07</month><day>10</day><access-date>2026-07-31</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.asge.org/home/resources/key-resources/blog/view/practical-solutions/2023/09/27/case-7--achalasia">https://www.asge.org/home/resources/key-resources/blog/view/practical-solutions/2023/09/27/case-7--achalasia</ext-link></comment></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mahmad</surname><given-names>AI</given-names> </name><name name-style="western"><surname>Jehangir</surname><given-names>W</given-names> </name><name name-style="western"><surname>Littlefield</surname><given-names>JM</given-names> </name><name name-style="western"><surname>John</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yousif</surname><given-names>A</given-names> </name></person-group><article-title>Cannabis hyperemesis syndrome: a case report review of treatment</article-title><source>Toxicol Rep</source><year>2015</year><volume>2</volume><fpage>889</fpage><lpage>890</lpage><pub-id pub-id-type="doi">10.1016/j.toxrep.2015.05.015</pub-id><pub-id pub-id-type="medline">28962425</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shin</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name></person-group><article-title>Case 10: a 30-year-old woman with breast mass and family history of cancer</article-title><source>J Korean Med Sci</source><year>2023</year><month>05</month><day>8</day><volume>38</volume><issue>18</issue><fpage>e138</fpage><pub-id pub-id-type="doi">10.3346/jkms.2023.38.e138</pub-id><pub-id pub-id-type="medline">37158774</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="web"><article-title>Life science and medical research involving human subjects [Article in Japanese]</article-title><source>Ministry of Education, Culture, Sports, Science and Technology (MEXT)</source><access-date>2026-05-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.mext.go.jp/a_menu/lifescience/bioethics/seimeikagaku_igaku.html">https://www.mext.go.jp/a_menu/lifescience/bioethics/seimeikagaku_igaku.html</ext-link></comment></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Manaktala</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>IN</given-names> </name><name name-style="western"><surname>Melton</surname><given-names>GB</given-names> </name></person-group><article-title>A multi-site content analysis of social history information in clinical notes</article-title><source>AMIA Annu Symp Proc</source><year>2011</year><volume>2011</volume><fpage>227</fpage><lpage>236</lpage><pub-id pub-id-type="medline">22195074</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="web"><article-title>thw-nattaw/chatbot_qwen_history_taking_paper</article-title><source>GitHub</source><year>2026</year><access-date>2026-05-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/thw-nattaw/chatbot_qwen_history_taking_paper">https://github.com/thw-nattaw/chatbot_qwen_history_taking_paper</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary materials comprising the full detailed prompt (Section A), the complete list of 33 target symptoms, and 2 representative diagnoses for each symptom (Section B), an example of a case-specific checklist developed for evaluation (Section C), and the full ranking of the 33 symptoms by macroaverage coverage percentage across all chatbot modes (Section D).</p><media xlink:href="medinform_v14i1e94614_app1.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material></app-group></back></article>