<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e87887</article-id><article-id pub-id-type="doi">10.2196/87887</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Decision Support Framework for Quality Assurance and Enhancement of Therapeutic Artificial Intelligence Systems: Mixed Methods Pilot Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Kang</surname><given-names>Boyoung</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kwon</surname><given-names>Kyungmin</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huilin</surname><given-names>Piao</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hong</surname><given-names>Seohyeon</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Choi</surname><given-names>Seoin</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Oh</surname><given-names>Hayoung</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Applied Artificial Intelligence, Sungkyunkwan University</institution><addr-line>25-2, Sungkyunkwan-Ro, Jongno-gu</addr-line><addr-line>Seoul</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Patel</surname><given-names>Birjukumar</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Adegoke</surname><given-names>Kola</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Palama</surname><given-names>Valentina</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Hayoung Oh, PhD, Department of Applied Artificial Intelligence, Sungkyunkwan University, 25-2, Sungkyunkwan-Ro, Jongno-gu, Seoul, Republic of Korea, 82 1053895996; <email>hyoh79@skku.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e87887</elocation-id><history><date date-type="received"><day>16</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>12</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Boyoung Kang, Kyungmin Kwon, Piao Huilin, Seohyeon Hong, Seoin Choi, Hayoung Oh. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e87887"/><abstract><sec><title>Background</title><p>Therapeutic chatbots are increasingly deployed across digital mental health services, yet most evaluation efforts remain diagnostic rather than actionable. Organizations lack structured pathways to translate evaluation findings into validated quality improvements aligned with health care quality assurance requirements.</p></sec><sec><title>Objective</title><p>This study aims to introduce EvaluationPlus, a decision support framework that operationalizes a reproducible evaluation-to-enhancement loop for therapeutic artificial intelligence systems. We aimed to demonstrate its feasibility through expert-guided diagnosis, multi-large language model (LLM) enhancement mapping, and within-subject validation.</p></sec><sec sec-type="methods"><title>Methods</title><p>Using the bilingual mental health chatbot Dr. CareSam (GPT 4.0&#x2013;based), we conducted 3 iterative enhancement cycles. Two licensed clinical psychologists performed structured diagnostic reviews using think-aloud protocols to identify competency-specific deficits across a 7-dimension therapeutic competency rubric. Three LLMs (GPT 4.0, Claude 4.0 Sonnet, Gemini 2.5 Flash) generated prescriptive enhancement strategies aligned with identified gaps. A participant-blinded, within-subject A/B validation study with Korean graduate students from the Department of Applied Artificial Intelligence, Sungkyunkwan University (N=15; 16 recruited, 1 excluded; IRB-approved) compared baseline and enhanced versions across standardized clinical scenarios spanning mild anxiety, to crisis-level presentations.</p></sec><sec sec-type="results"><title>Results</title><p>The enhanced system demonstrated substantial improvement in overall therapeutic quality, with mean scores increasing from 5.40 to 7.63 (&#x0394; =+2.23 points, 41%; dz=0.881; 95% bootstrap CI [0.32-2.20]). Prespecified target dimensions &#x2014; active listening and appropriate questions, personalization, and complex thinking &#x2014; showed large-effect improvements (mean gain +3.04; dz range 0.96&#x2010;1.08), significantly exceeding gains in nontargeted dimensions (+1.62; targeting differential +1.42 points). Directional improvement was observed in 13 of 15 participants (86.7%). User preference strongly favored the enhanced system (13/15, 86.7%), and expert clinical evaluation confirmed maintained safety and therapeutic appropriateness across four scenario severity levels (preference rate 75%; 3 of 4 scenarios). Cross-participant rating consistency improved substantially (coefficient of variation: 20% &#x2192; 8.1%).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>EvaluationPlus demonstrates feasibility as a structured framework for iterative quality assurance of therapeutic artificial intelligence systems. By linking expert diagnostic procedures with prescriptive multi-LLM enhancement mapping and multistakeholder validation, the framework supports reproducible improvement cycles relevant to organizational oversight of digital mental health tools. Limitations include a small pilot sample, single-culture focus, and simulated crisis scenarios; future work should extend validation to diverse clinical populations and longitudinal outcome assessment.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>chatbots</kwd><kwd>mental health services</kwd><kwd>medical informatics</kwd><kwd>quality assurance</kwd><kwd>large language model</kwd><kwd>human-computer interaction</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Digital mental health tools, including therapeutic chatbots, are increasingly deployed to address rising psychological distress among young adults and the limited availability of traditional counseling services [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. In South Korea, the proportion of young adults aged 20&#x2010;30 at risk for depression reached approximately 30% in 2021, representing a sixfold increase compared to 2018 [<xref ref-type="bibr" rid="ref4">4</xref>]; similarly, anxiety and depression symptoms among college students in the United States increased by 75% between 2019 and 2021 [<xref ref-type="bibr" rid="ref5">5</xref>]. Despite this growing need, treatment-seeking rates remain critically low, with fewer than 36% of distressed students using available counseling services. Evidence suggests that conversational agents can provide accessible and scalable support, demonstrating small but meaningful effects in university populations [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. As deployment expands, health care organizations face growing pressure to maintain clinical safety, therapeutic appropriateness, and systematic quality assurance processes comparable to other clinical decision support tools.</p><p>However, a persistent evaluation&#x2013;enhancement gap limits responsible iterative development. Although therapeutic chatbots are routinely evaluated through user studies, expert reviews, or large language model (LLM)-based assessments, these evaluations rarely translate into structured and governable improvement processes. Existing workflows often rely on ad hoc prompt adjustments or informal fine-tuning, leaving organizations without systematic methods to act on diagnostic insights, verify improvements, or document decision pathways for organizational oversight. This gap is particularly concerning in mental health contexts, where deficits in personalization, probing skills, or crisis handling can directly affect user safety. Although recent advances in multidimensional evaluation, think-aloud protocols, and LLM-assisted scoring have improved diagnostic precision, these frameworks remain primarily diagnostic and offer limited operational guidance on how identified weaknesses should be systematically enhanced and validated.</p><p>Our previous study [<xref ref-type="bibr" rid="ref6">6</xref>] established a comprehensive baseline evaluation of Dr.CareSam, a GPT 4.0&#x2013;based bilingual mental health chatbot. That study identified strong relational competencies, including positivity and support, empathy, and active listening, while revealing systematic deficits in professionalism, content complexity, and personalization. However, the evaluation concluded without providing operational pathways for systematic improvement, exemplifying the evaluation&#x2013;enhancement gap in therapeutic artificial intelligence (AI) development.</p></sec><sec id="s1-2"><title>Related Work</title><sec id="s1-2-1"><title>Evaluation of Mental Health Chatbots</title><p>Evaluating therapeutic chatbots requires capturing complex communicative qualities, empathy, boundary maintenance, probing skills, and clinical appropriateness that traditional natural language processing (NLP) metrics cannot reliably assess. Studies consistently demonstrate weak correlations between metrics such as BLEU, ROUGE, and BERTScore and human judgment in conversational or clinical settings. Chu et al [<xref ref-type="bibr" rid="ref7">7</xref>] showed that these metrics substantially underperform on coherence, relevance, and appropriateness, underscoring the need for human-aligned evaluation approaches. Meta-analytic evidence on digital mental health interventions further highlights both their modest effectiveness and the substantial heterogeneity in evaluation methodology [<xref ref-type="bibr" rid="ref2">2</xref>], reinforcing the need for standardized, clinically meaningful quality assurance frameworks.</p></sec><sec id="s1-2-2"><title>Advances in LLM-Based Evaluation</title><p>LLMs have enabled scalable clinical assessment, but most efforts remain diagnostic rather than enhancement-oriented. Large-scale benchmarking illustrates this limitation: Hager et al [<xref ref-type="bibr" rid="ref8">8</xref>] found that state-of-the-art LLMs, despite high test performance, frequently failed guideline adherence and clinical reasoning on 2400 real patient cases. Park et al [<xref ref-type="bibr" rid="ref9">9</xref>] showed that LLM ensembles can approximate expert ratings (<italic>r</italic>&#x003E;0.80), yet their framework focuses exclusively on scoring accuracy rather than on pathways for systematic enhancement.</p><p>Hybrid human&#x2013;AI evaluation methods follow similar patterns. Think-aloud protocols, originally introduced by Ericsson and Simon [<xref ref-type="bibr" rid="ref10">10</xref>], have been adapted for LLM-supported assessment [<xref ref-type="bibr" rid="ref7">7</xref>] and achieve high alignment with human ratings. Choo et al [<xref ref-type="bibr" rid="ref11">11</xref>] proposed a 3-bot evaluation system comprising simulated patient, provider, and evaluator roles, demonstrating efficiency and expert concordance while noting that more detailed scoring standards are needed to support actionable improvement. Frameworks such as QUEST [<xref ref-type="bibr" rid="ref12">12</xref>] provide structured evaluation procedures but remain primarily oriented toward assessment design rather than prescriptive quality improvement.</p></sec><sec id="s1-2-3"><title>Human&#x2013;LLM Collaborative Evaluation Frameworks</title><p>Newer frameworks integrate clinical expertise with LLM capabilities to improve evaluation fidelity. Park et al [<xref ref-type="bibr" rid="ref9">9</xref>] introduced a benchmark structure using expert-written ideal responses and guideline-based prompts, creating a scalable evaluation mechanism for mental health scenarios. Louie et al [<xref ref-type="bibr" rid="ref13">13</xref>] developed Roleplay-doh, enabling iterative refinement of simulated AI patients and improving authenticity and training readiness. Moilanen et al [<xref ref-type="bibr" rid="ref14">14</xref>] examined personality-driven engagement differences in chatbot interactions, while advances in chain-of-thought prompting [<xref ref-type="bibr" rid="ref15">15</xref>] offer more interpretable reasoning pathways for clinical decision support.</p><p>These approaches represent meaningful progress in evaluation methodology but remain focused on assessment fidelity&#x2014;not on how identified weaknesses should be translated into validated system enhancements.</p></sec><sec id="s1-2-4"><title>The Evaluation&#x2013;Enhancement Gap</title><p>Across the literature, the primary limitation is the absence of structured quality assurance frameworks that operationalize evaluation findings into actionable enhancements. When evaluations flag deficits, such as weak personalization or inadequate crisis response, existing methods provide limited guidance on diagnostic specificity, targeted intervention design, or safety validation. Contemporary therapeutic AI development often relies on trial-and-error prompt refinement or irregular updates that lack documented quality assurance principles. Even advanced evaluation systems, including automated triads [<xref ref-type="bibr" rid="ref11">11</xref>], personality impact analyses [<xref ref-type="bibr" rid="ref14">14</xref>], and safety audits [<xref ref-type="bibr" rid="ref16">16</xref>], remain nonprescriptive and do not offer structured pathways for continuous improvement.</p><p>Emerging evidence suggests that workflow-level quality assurance, not model-level scoring alone, will be necessary for responsible clinical integration. Gaber et al [<xref ref-type="bibr" rid="ref17">17</xref>] demonstrated this through evaluation of LLM-based triage, referral, and diagnosis pipelines, emphasizing the need for systematic QA processes that extend beyond performance metrics alone. This persistent evaluation&#x2013;enhancement gap motivates the framework developed in this study.</p></sec><sec id="s1-2-5"><title>Study Aims</title><p>Building on these findings, this study introduces EvaluationPlus, a decision support framework designed to operationalize the evaluation-to-enhancement loop. The framework integrates expert-guided diagnosis, multi-LLM&#x2013;assisted enhancement mapping, and controlled within-subject validation within a unified quality assurance workflow. Using the same system and evaluation rubric as the prior study ensures methodological consistency and enables direct comparison of enhancement effectiveness.</p><p>The aim of this study is to develop and validate EvaluationPlus, a structured decision support framework that translates diagnostic evaluation findings into systematic, reproducible enhancement processes for therapeutic AI systems.</p></sec></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>We conducted a within-subject, participant-blinded A/B pilot validation study of the EvaluationPlus framework. Each participant interacted with both the baseline system (version A) and the enhanced system (version B) of Dr. CareSam, a bilingual GPT 4.0&#x2013;based therapeutic chatbot (<xref ref-type="fig" rid="figure1">Figure 1</xref>). Sessions comprised 5&#x2010;6 conversational turns per scenario; participants accessed both versions independently via an online platform, and presentation order was self-determined rather than experimentally controlled. The 3-stage enhancement and validation process is illustrated in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Within-subject participant-blinded A/B validation workflow. Steps: (1) landing page; (2) safety instructions and crisis resources; (3) participant-blinded interaction across three scenarios; presentation order was self-determined by each participant; (4) survey submission; (5a) standardized 10-point Likert evaluation across seven dimensions; (5b) open-ended qualitative feedback.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e87887_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>EvaluationPlus three-stage workflow. Stage 1 (Diagnostic review): structured think-aloud review by two licensed clinical practitioners. Stage 2 (gap&#x2013;solution mapping): multilarge language model consultation using GPT 4.0 (OpenAI), Claude 4.0 Sonnet (Anthropic), and Gemini 2.5 Flash (Google). Stage 3 (validation and implementation): participant-blinded within-subject A/B validation (N=15). AI: artificial intelligence; LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e87887_fig02.png"/></fig></sec><sec id="s2-2"><title>Ethical Considerations</title><p>The study was approved by the Sungkyunkwan University Institutional Review Board (IRB number 2023-02-043-007). All participants provided written informed consent and were informed that Dr. CareSam is a research prototype. Transcripts were fully deidentified per Korea&#x2019;s Personal Information Protection Act (PIPA). A structured debriefing protocol was implemented after each session; participants engaging with the suicidal ideation scenario received written materials and national crisis helpline resources. No participants required clinical referral. Participants were compensated &#x20A9;20,000 (~USD 15) for approximately 60 minutes.</p></sec><sec id="s2-3"><title>Participants</title><p>Korean graduate students were recruited at the Department of Applied Artificial Intelligence, Sungkyunkwan University. Inclusion criteria were Korean fluency and self-reported comfort with mental health content; participants in active crisis care were excluded. Sixteen participants enrolled; one was excluded under a prespecified dual criterion: statistical extremity (z=&#x2212;2.60, |z| &#x003E; 2.5 SD) and qualitative&#x2013;quantitative inconsistency indicating task misunderstanding (full documentation in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The final cohort was N=15 (<xref ref-type="table" rid="table1">Table 1</xref>). Data collection was conducted in August 2025.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographic and academic characteristics of the validation cohort (N=15).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variable and categories</td><td align="left" valign="bottom">Values</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Age (years)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mean (SD)</td><td align="left" valign="top">28.3 (7.8)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>20 s, n (%)</td><td align="left" valign="top">13 (86.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>30 s, n (%)</td><td align="left" valign="top">1 (6.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>50 s, n (%)</td><td align="left" valign="top">1 (6.7)</td></tr><tr><td align="left" valign="top" colspan="2">Sex, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">7 (46.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">8 (53.3)</td></tr><tr><td align="left" valign="top" colspan="2">Education, n (%)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Graduate student</td><td align="left" valign="top">14 (93.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Postdoctoral researcher</td><td align="left" valign="top">1 (6.7)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Mean age computed with one participant&#x2019;s age estimated as 25 (midpoint of reported range); range 23&#x2013;53.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-4"><title>Therapeutic Scenarios</title><p>Four standardized scenarios spanned a clinical severity spectrum: presentation anxiety (mild), academic dropout ideation (moderate), panic symptoms (moderate&#x2013;high), and suicidal ideation (crisis) [<xref ref-type="bibr" rid="ref18">18</xref>] (<xref ref-type="fig" rid="figure3">Figure 3</xref>). User validation used three scenarios (presentation anxiety, academic dropout, and suicidal ideation); participants completed a minimum of 2 per version. The panic scenario was reserved for expert evaluation only. Across both versions, 13 of 15 participants (86.7%) engaged with Scenario 4 (suicidal ideation), 9 (60%) with Scenario 1 (presentation anxiety), and 3 (20%) with Scenario 2 (academic dropout). Bilingual scenario scripts are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Therapeutic scenario severity spectrum from mild anxiety to crisis-level suicidal ideation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e87887_fig03.png"/></fig></sec><sec id="s2-5"><title>EvaluationPlus Framework</title><p>EvaluationPlus integrates 3 sequential stages (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><sec id="s2-5-1"><title>Stage 1 (Diagnostic Review)</title><p>Two licensed clinical psychologists conducted structured think-aloud reviews of baseline transcripts using a seven-dimension therapeutic competency rubric (Table A1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). The 2 practitioners reached consensus through iterative discussion rather than independent numeric scoring; formal interrater reliability was not computed and is acknowledged as a limitation.</p></sec><sec id="s2-5-2"><title>Stage 2 (Gap&#x2013;Solution Mapping)</title><p>Diagnostic findings were translated into prescriptive enhancement strategies through structured consultation with three LLMs: GPT 4.0 (OpenAI), Claude 4.0 Sonnet (Anthropic), and Gemini 2.5 Flash (Google). Each LLM received identical structured prompts presenting the diagnostic findings and requesting enhancement recommendations aligned with the rubric dimensions. Converging recommendations were directly incorporated; divergent recommendations were adjudicated by the supervising clinical psychologist based on clinical appropriateness, implementation feasibility, and safety criteria. Full prompt texts, outputs, and the adjudication log are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec><sec id="s2-5-3"><title>Stage 3 (Implementation and Validation)</title><p>Enhancements were implemented across 3 iterative cycles (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>) and validated through within-subject A/B comparison. A single licensed clinical psychologist &#x2014; independent of the stage 1 reviewers &#x2014; conducted blinded evaluation of matched transcript pairs; the use of a single evaluator precludes interrater reliability estimation and is acknowledged as a limitation.</p></sec></sec><sec id="s2-6"><title>Outcome Measures</title><p>A multistakeholder framework integrated 3 evaluator perspectives (<xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>). Full evaluation instruments are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Outcome measures and data collection methods for clinical validation.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome category</td><td align="left" valign="bottom">Measurement approach</td><td align="left" valign="bottom">Scale or method</td><td align="left" valign="bottom">Evaluator</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Primary outcome</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Therapeutic communication score (TCS)</td><td align="left" valign="top">Average across 7 competency dimensions</td><td align="left" valign="top">10-point Likert scale</td><td align="left" valign="top">Users (N=15)</td></tr><tr><td align="left" valign="top" colspan="4">Dimension-level outcomes</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Seven therapeutic dimensions<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">Individual dimension scores and paired differences (A &#x2192; B)</td><td align="left" valign="top">10-point Likert per dimension</td><td align="left" valign="top">Users (N=15)</td></tr><tr><td align="left" valign="top" colspan="4">Process and behavioral features</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Conversation analytics</td><td align="left" valign="top">Response length, sentence count, question rate, emoji usage</td><td align="left" valign="top">Automated extraction</td><td align="left" valign="top">System</td></tr><tr><td align="left" valign="top" colspan="4">Expert clinical assessment</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Independent expert review</td><td align="left" valign="top">Blinded transcript evaluation across 7 dimensions for 4 scenarios</td><td align="left" valign="top">3-point rubric (1=Poor, 2=Adequate, 3=Excellent)</td><td align="left" valign="top">Clinical psychologist (blinded)</td></tr><tr><td align="left" valign="top" colspan="4">AI-based triangulation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Multi-LLM<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> assessment</td><td align="left" valign="top">Structured rubric-aligned evaluation with quantitative scores and rationales</td><td align="left" valign="top">Standardized prompts (temp=0.3)</td><td align="left" valign="top">GPT 4.0, Claude 4.0 Sonnet, Gemini 2.5 Flash</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Seven dimensions: (1) empathy, (2) accuracy/usefulness, (3) complex thinking and emotions, (4) active listening and appropriate questions, (5) positivity and support, (6) professionalism, (7) personalization.</p></fn><fn id="table2fn2"><p><sup>b</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Multistakeholder validation protocol: evaluator roles and assessment procedures.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluator type</td><td align="left" valign="bottom">Assessment method</td><td align="left" valign="bottom">Measurement scale</td><td align="left" valign="bottom">Validation purpose</td></tr></thead><tbody><tr><td align="left" valign="top">End users (N=15)</td><td align="left" valign="top">Postsession evaluation forms after each chatbot interaction</td><td align="left" valign="top">10-point Likert across 7 dimensions</td><td align="left" valign="top">User experience quality and perceived therapeutic effectiveness</td></tr><tr><td align="left" valign="top">Clinical expert</td><td align="left" valign="top">Blinded transcript review across severity spectrum</td><td align="left" valign="top">3-point clinical rubric per dimension</td><td align="left" valign="top">Clinical appropriateness, safety, and professional quality</td></tr><tr><td align="left" valign="top">Multi-LLM<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> assessment</td><td align="left" valign="top">Structured rubric prompts (temperature=0.3)</td><td align="left" valign="top">Quantitative scores + qualitative rationales</td><td align="left" valign="top">Scalable complementary validation perspective</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-7"><title>Statistical Analysis</title><p>All analyses followed a prespecified within-subject plan emphasizing effect size estimation over null-hypothesis testing, consistent with pilot validation practice. As a pilot feasibility study, the sample size was determined to permit intensive multiscenario evaluation (each participant generated 14 independent dimension-level ratings [7 dimensions&#x00D7;2 versions]; where participants engaged with multiple scenarios, dimension ratings were averaged across scenarios prior to analysis to yield a single composite score per dimension per version) rather than to achieve formal statistical power.</p><p>Descriptive statistics (mean&#x00B1;SD) were computed for both versions across all seven dimensions. The primary effect size was paired Cohen dz = (M&#x1D2E; &#x2212; M&#x1D2C;) / SD&#x0394;. Enhancement targeting validity was assessed as &#x0394;_targets &#x2212; &#x0394;_nontargets (target dimensions: active listening, personalization, and complex thinking). Performance consistency was assessed via coefficient of variation (CV), Shannon entropy across dimension profiles, and the targeting differential. Bias-corrected bootstrap confidence intervals (B=5000 iterations) were computed for all effect size estimates. All <italic>P</italic> values are reported as exact values per <italic>Journal of Medical Internet Research</italic> (<italic>JMIR</italic>) guidelines.</p><p>The prespecified dual-criterion exclusion rule required both statistical extremity (|z|&#x003E;2.5 SD on the participant-level overall mean) and qualitative evidence of task invalidity. Participant P16 satisfied both criteria: the Version B participant-level mean (1.29) produced a standardized score of z=&#x2212;2.60, and open-ended feedback indicated that the participant was evaluating the general training characteristics of the underlying language model rather than the chatbot interaction itself, rendering the data substantively invalid rather than merely variable. As a sensitivity analysis, the primary within-subject effect size was recomputed including P16 (n=16): the mean gain was approximately 1.64 points and the effect size was dz=0.49 (small-to-medium), compared with the prespecified analytic result of dz=0.881 (large) for the N=15 cohort. The primary analysis is retained as prespecified; the sensitivity result is reported in the interest of transparency.</p><p>One retained participant (P5) assigned identical scores of 5 to all 7 dimensions for version B, consistent with central tendency bias &#x2014; a well-documented psychometric phenomenon reflecting a valid but conservative rating style. This pattern is qualitatively distinct from the P16 exclusion criterion, which was based on convergent statistical extremity and direct qualitative evidence of task misunderstanding. Retention of P5 is conservative with respect to the primary comparison, as uniform midpoint ratings reduce rather than inflate the reported mean gain.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Analytical Approach</title><p>The analytic cohort comprised 15 participants (demographic characteristics are reported in the Methods section). All analyses followed the prespecified within-subject analysis plan described in the Methods section, emphasizing effect size estimation and consistency metrics. Each participant evaluated both the baseline system (version A) and the enhanced system (version B) across seven therapeutic dimensions using validated 10-point Likert scales. Bias-corrected bootstrap CIs (B=5000 iterations) are reported alongside all effect size estimates.</p><p>Participants accessed both chatbot versions independently via an online platform; presentation order was self-determined and was not experimentally controlled. Accordingly, formal order effect testing was not conducted, and potential sequence effects cannot be ruled out. This represents a methodological limitation acknowledged in the Limitations section.</p><p>Mean conversational turns were 8.5 (SD 4.5, range 4&#x2010;20) for version A and 7.9 (SD 3.3, range 4&#x2010;15) for Version B, exceeding the 5&#x2010;6 turn threshold in 13 of 15 (version A) and 14 of 15 (version B) participants. Two participants in version A completed 4 turns, representing a minor deviation from the protocol threshold. Of the 15 participants, 8 (53.3%) engaged with exactly two scenarios across both versions combined, 1 (6.7%) engaged with all three scenarios, and 6 (40%) engaged primarily with one scenario per version, representing a deviation from the stated minimum of 2 scenarios per version; this is acknowledged as a protocol limitation in the Limitations section. Dimensional analyses were conducted across all available scenario&#x2013;version pairings.</p></sec><sec id="s3-2"><title>Primary Outcome: Overall Therapeutic Performance</title><p>The enhanced system (version B) consistently outperformed the baseline system (version A) across all seven therapeutic dimensions. The overall mean therapeutic communication score (TCS) increased from 5.40 (SD 2.11) for Version A to 7.63 (SD 1.31) for Version B, representing a mean gain of +2.23 points (41%) with a within-subject effect size of dz=0.881 (95% bootstrap CI [0.32-2.20]). The bootstrap CI is deliberately wide, with the lower bound crossing into small-effect territory, reflecting appropriate uncertainty for a pilot cohort of N=15; readers should interpret effect size estimates as exploratory rather than confirmatory.</p><p>Enhancement effects were largest in the three prespecified target dimensions identified through Stage 1 expert diagnosis. Large effect improvements (targeted dimensions) were observed for Active Listening and Appropriate Questions (+3.20 points; dz=1, 95% CI [0.42-2.26]), Personalization (+2.93 points; dz=1.08, 95% CI [0.51-2.44]), and Complex Thinking and Emotions (+3.00 points; dz=0.96, 95% CI [0.42-2.16]). Medium effect improvements (secondary dimensions) were observed for empathy (+2.00; dz=0.66, 95% CI [0.17-1.46]), accuracy and usefulness (+1.87; dz=0.65, 95% CI [0.14-1.62]), and professionalism (+1.54; dz=0.55, 95% CI [0.06-1.25]). A small effect improvement was observed for the nontargeted dimension of positivity and support (+1.07; dz=0.39, 95% CI [&#x2212;0.09 to 0.95]). The 95% bootstrap CI for this dimension (&#x2212;0.09 to 0.95) includes zero, and this result does not exclude a null effect; the improvement should accordingly be interpreted as preliminary rather than definitive. Full dimensional comparisons are presented in <xref ref-type="table" rid="table4">Table 4</xref>.</p><p>The targeting differential &#x2014; mean improvement in prespecified target dimensions (+3.04) minus nontargeted dimensions (+1.62) &#x2014; was+ 1.42 points, confirming that enhancement was concentrated in clinically intended competency areas rather than distributed uniformly across all dimensions.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Dimensional improvements in therapeutic competencies from baseline (version A) to enhanced system (version B), with paired effect sizes and bias-corrected bootstrap 95% CIs (N=15). All dimensions favor Version B. Bootstrap CIs based on B=5000 iterations (seed=42); see <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for full documentation. Interpretation thresholds: small dz&#x003C;0.5; medium 0.5-0.8; large&#x003E;0.8.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">A<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> Mean (SD)</td><td align="left" valign="bottom">B<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> Mean (SD)</td><td align="left" valign="bottom">&#x0394;(B&#x2212;A)<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="bottom">dz<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="bottom">95% Bootstrap CI</td><td align="left" valign="bottom">Interpretation</td></tr></thead><tbody><tr><td align="left" valign="top">Empathy</td><td align="left" valign="top">5.60 (2.77)</td><td align="left" valign="top"><bold>7.60 (1.18)</bold><sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">+2</td><td align="left" valign="top">0.66</td><td align="left" valign="top">0.17-1.46</td><td align="left" valign="top">Medium</td></tr><tr><td align="left" valign="top">Accuracy and usefulness</td><td align="left" valign="top">5.80 (2.54)</td><td align="left" valign="top"><bold>7.67 (1.23)</bold></td><td align="left" valign="top">+1.87</td><td align="left" valign="top">0.65</td><td align="left" valign="top">0.14-1.62</td><td align="left" valign="top">Medium</td></tr><tr><td align="left" valign="top">Complex thinking and emotions</td><td align="left" valign="top">4.53 (2.64)</td><td align="left" valign="top"><bold>7.53 (1.88)</bold></td><td align="left" valign="top">+3.00</td><td align="left" valign="top">0.96</td><td align="left" valign="top">0.42-2.16</td><td align="left" valign="top">Large</td></tr><tr><td align="left" valign="top">Active listening and appropriate questions</td><td align="left" valign="top">5.13 (2.90)</td><td align="left" valign="top"><bold>8.33 (1.59)</bold></td><td align="left" valign="top">+3.20</td><td align="left" valign="top">1</td><td align="left" valign="top">0.42-2.26</td><td align="left" valign="top">Large</td></tr><tr><td align="left" valign="top">Positivity and support</td><td align="left" valign="top">7.40 (2.44)</td><td align="left" valign="top"><bold>8.47 (1.36)</bold></td><td align="left" valign="top">+1.07</td><td align="left" valign="top">0.39</td><td align="left" valign="top">&#x2212;0.09 to 0.95</td><td align="left" valign="top">Small</td></tr><tr><td align="left" valign="top">Professionalism</td><td align="left" valign="top">5.33 (2.41)</td><td align="left" valign="top"><bold>6.87 (2.17)</bold></td><td align="left" valign="top">+1.54</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.06-1.25</td><td align="left" valign="top">Medium</td></tr><tr><td align="left" valign="top">Personalization</td><td align="left" valign="top">4 (2.17)</td><td align="left" valign="top"><bold>6.93 (2.15)</bold></td><td align="left" valign="top">+2.93</td><td align="left" valign="top">1.08</td><td align="left" valign="top">0.51-2.44</td><td align="left" valign="top">Large</td></tr><tr><td align="left" valign="top"><bold>Overall</bold> (<bold>participant-level mean</bold>)</td><td align="left" valign="top">5.40 (2.11)</td><td align="left" valign="top"><bold>7.63 (1.31)</bold></td><td align="left" valign="top"><bold>+2.23</bold></td><td align="left" valign="top"><bold>0.881</bold></td><td align="left" valign="top"><bold>0.32 to 2.20</bold></td><td align="left" valign="top"><bold>Large</bold></td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>A: baseline (pretuning) version.</p></fn><fn id="table4fn2"><p><sup>b</sup>B: enhanced (posttuning) version.</p></fn><fn id="table4fn3"><p><sup>c</sup>&#x0394;(B&#x2212;A)=mean difference.</p></fn><fn id="table4fn4"><p><sup>d</sup>Cohen dz is the within-subject effect size (mean of paired differences divided by the SD of those differences) and is not directly comparable to between-subject Cohen <italic>d</italic>.</p></fn><fn id="table4fn5"><p><sup>e</sup>dz: paired effect size.</p></fn><fn id="table4fn6"><p><sup>f</sup>These values indicate higher mean.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Individual Trajectories and Conversational Mechanisms</title><p>Individual trajectory analysis showed that 13 of 15 participants (86.7%) demonstrated improvement from Version A to Version B, with gains ranging from +0.43 to+5.86 points. Two participants showed declines (&#x2212;3.86 and &#x2212;1.29 points). The larger decline (&#x2212;3.86) occurred in the participant with the highest baseline score (version A=8.86), consistent with a ceiling or contrast effect rather than system deterioration. The cohort-level mean improvement was &#x0394; =+2.23 points (dz=0.881, 95% CI [0.32-2.20]), representing a large within-subject effect (<xref ref-type="fig" rid="figure4">Figure 4</xref>, Panel a).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Comprehensive comparison of therapeutic effectiveness and conversational patterns across the clinical validation cohort (N=15). Panel (A): individual participant score trajectories from version A to version B (<italic>&#x0394;</italic>=+2.23, dz=0.881, 95% CI [0.32-2.20]); Panel (B): average response length reduction (&#x2212;44.9%, from 381 to 210 characters); panel (C): sentence count reduction (&#x2212;50.5%, from 10.9 to 5.4 sentences per response); panel (D): question-based response rate increase (+27.4 percentage points, from 42.3% to 69.7%).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e87887_fig04.png"/></fig><p>Behavioral analysis of conversational logs revealed three concurrent changes in the enhanced system: average character count decreased by 44.9% (from 381 to 210 characters per response); sentence count decreased by 50.5% (from 10.9 to 5.4 sentences per response); and the proportion of question-based responses increased from 42.3% to 69.7% (+27.4 percentage points) (<xref ref-type="fig" rid="figure4">Figure 4</xref>, Panels b&#x2013;d).</p><p>These three changes co-occurred as a result of the same enhancement protocol and cannot be fully disentangled statistically within the current pilot design. Observed rating improvements may reflect the increase in interactive questioning, the reduction in cognitive load from shorter responses, or an interaction of both. We treat the interpretation that questioning rather than length reduction is the primary driver of improvement as a hypothesis to be tested in future experimental designs with orthogonally manipulated conditions, rather than a confirmed finding of this study.</p><p>Scenario-level disaggregation was not prespecified, and full stratified reanalysis is beyond the scope of this pilot study. However, descriptive performance for the crisis scenario (scenario 4: suicidal ideation) is noted: the expert evaluator assigned the highest possible scores to Version B across all 7 dimensions in this scenario (total score 21/21), compared with 14/21 for version A, indicating that the enhanced system performed strongly on the highest-severity scenario. This descriptive finding should be interpreted cautiously given single-expert evaluation and a single scenario instance; it does not preclude the possibility that composite averaging across severity levels may mask scenario-specific variation in other subgroups. This limitation is acknowledged in the Limitations section.</p></sec><sec id="s3-4"><title>Qualitative Validation: Participant Perspectives</title><p>Open-ended feedback from all 15 participants revealed three primary themes consistent with the quantitative improvements. Response structure: participants described baseline responses as &#x201C;too long and metaphorical,&#x201D; whereas enhanced responses were characterized as &#x201C;short, clear, and question-driven.&#x201D; Tone calibration: baseline responses were perceived as &#x201C;light and dismissive&#x201D; in high-severity situations, while the enhanced system was described as &#x201C;serious when needed, yet supportive.&#x201D; Contextual understanding: the baseline system was described as providing &#x201C;the same suggestions regardless of context,&#x201D; while the enhanced system felt &#x201C;more human-like, like talking to a friend who understands.&#x201D;</p><p>These themes align directly with the largest quantitative gains in active listening (+3.20), complex thinking (+3), and personalization (+2.93), providing convergent qualitative&#x2013;quantitative evidence for the enhancement mechanisms. One critical qualitative pattern also emerged: several participants noted that the enhanced system tended to repeat a structure of asking a question followed by a longer explanation, suggesting that the mandatory follow-up question rule may have replaced structural rigidity with a new perceivable cadence. This finding and its implications for future design iterations are discussed in the Discussion section. Full thematic synthesis and representative quotations are provided in Table S1 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p></sec><sec id="s3-5"><title>Expert Clinical Validation</title><p>A licensed clinical psychologist (PhD, 30+ y of clinical experience) conducted independent evaluation of matched transcript pairs across four mental health scenarios using a 3-point evidence-based scale (1=Poor; 2=Adequate; 3=Excellent). The evaluator was blinded to version assignment, participant identities, and scenario order.</p><p>The enhanced system was preferred in three of four scenarios (75% preference rate); the academic stress scenario (S2) favored the baseline version. Overall expert scores were 76 points (version B) versus 65 points (version A) across all scenarios and dimensions combined (<xref ref-type="table" rid="table5">Table 5</xref>). Given that a single expert evaluator was used, interrater reliability for this component cannot be estimated; expert preference findings should be interpreted as preliminary and hypothesis-generating rather than confirmatory.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Expert clinical evaluation scores across four therapeutic scenarios (3-point scale; 1=poor, 2=adequate, 3=excellent). Version B preferred in 3 of 4 scenarios (75%).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Scenario</td><td align="left" valign="bottom">Ver.</td><td align="left" valign="bottom">Emp<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="bottom">Acc<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="bottom">Cmp<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="bottom">Act<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="bottom">Pos<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup></td><td align="left" valign="bottom">Prof<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup></td><td align="left" valign="bottom">Pers<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup></td><td align="left" valign="bottom">Total</td></tr></thead><tbody><tr><td align="left" valign="top"><bold>S1: Anxiety (mild</bold>)<sup><xref ref-type="table-fn" rid="table5fn8">h</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table5fn9">i</xref></sup></td><td align="left" valign="top"><bold>B</bold><sup><xref ref-type="table-fn" rid="table5fn10">j</xref></sup></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>19</bold></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">A<sup><xref ref-type="table-fn" rid="table5fn12">l</xref></sup></td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">15</td></tr><tr><td align="left" valign="top"><bold>S2: Academic (moderate</bold>)<sup><xref ref-type="table-fn" rid="table5fn11">k</xref></sup></td><td align="left" valign="top"><bold>B</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>2</bold></td><td align="left" valign="top"><bold>15</bold></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">A</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">21</td></tr><tr><td align="left" valign="top"><bold>S3: Panic (mod&#x2013;high</bold>)<sup><xref ref-type="table-fn" rid="table5fn13">m</xref></sup></td><td align="left" valign="top"><bold>B</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>21</bold></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">A</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">15</td></tr><tr><td align="left" valign="top"><bold>S4: Crisis (crisis)</bold><sup><xref ref-type="table-fn" rid="table5fn14">n</xref></sup></td><td align="left" valign="top"><bold>B</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>3</bold></td><td align="left" valign="top"><bold>21</bold></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">A</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">2</td><td align="left" valign="top">14</td></tr><tr><td align="left" valign="top"><bold>Overall</bold></td><td align="left" valign="top"><bold>B</bold></td><td align="left" valign="top"><bold>11</bold></td><td align="left" valign="top"><bold>12</bold></td><td align="left" valign="top"><bold>11</bold></td><td align="left" valign="top"><bold>11</bold></td><td align="left" valign="top"><bold>10</bold></td><td align="left" valign="top"><bold>11</bold></td><td align="left" valign="top"><bold>10</bold></td><td align="left" valign="top"><bold>76</bold></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">A</td><td align="left" valign="top">9</td><td align="left" valign="top">11</td><td align="left" valign="top">9</td><td align="left" valign="top">8</td><td align="left" valign="top">10</td><td align="left" valign="top">9</td><td align="left" valign="top">9</td><td align="left" valign="top">65</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Emp: empathy.</p></fn><fn id="table5fn2"><p><sup>b</sup>Acc: accuracy and usefulness.</p></fn><fn id="table5fn3"><p><sup>c</sup>Cmp: complex thinking.</p></fn><fn id="table5fn4"><p><sup>d</sup>Act: active listening.</p></fn><fn id="table5fn5"><p><sup>e</sup>Pos: positivity and support.</p></fn><fn id="table5fn6"><p><sup>f</sup>Prof: professionalism.</p></fn><fn id="table5fn7"><p><sup>g</sup>Pers: personalization.</p></fn><fn id="table5fn8"><p><sup>h</sup>S1: presentation anxiety (mild).</p></fn><fn id="table5fn9"><p><sup>i</sup>Bold indicates higher score per cell.</p></fn><fn id="table5fn10"><p><sup>j</sup>B: enhanced version.</p></fn><fn id="table5fn11"><p><sup>k</sup>S2: academic stress (moderate); version A preferred.</p></fn><fn id="table5fn12"><p><sup>l</sup>A: baseline version.</p></fn><fn id="table5fn13"><p><sup>m</sup>S3: panic symptoms (moderate&#x2013;high; expert evaluation only).</p></fn><fn id="table5fn14"><p><sup>n</sup>S4: suicidal ideation (crisis).</p></fn></table-wrap-foot></table-wrap><p>Notably, the expert assigned high scores to the enhanced system&#x2019;s crisis response, while participant feedback characterized the same response as professional but insufficiently warm. This divergence between expert-rated clinical appropriateness and user-experienced engagement reflects a clinically meaningful tension between adherence to safety protocols and perceived conversational warmth, discussed further in the Discussion section. Representative response comparisons for the presentation anxiety and crisis scenarios are provided in Figures S1 and S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> .</p></sec><sec id="s3-6"><title>Multistakeholder Convergence</title><p>Multistakeholder assessment integrated three evaluator perspectives: participants (86.7% preference, 13/15), expert clinician (75% preference, 3/4 scenarios), and multi-LLM evaluation using GPT 4.0 (OpenAI), Claude 4.0 Sonnet (Anthropic), and Gemini 2.5 Flash (Google). All three evaluator categories showed consistent preference for the enhanced system, with convergence ranging from 75% to 86.7%.</p><p>LLM evaluators consistently favored Version B but frequently added qualified preferences, noting that enhancement effects were context-dependent, and declined to make definitive clinical recommendations that human evaluators provided more readily. While multi-LLM evaluation surfaced recurring structural issues in the baseline system &#x2014; including overuse of metaphors and insufficient crisis inquiry &#x2014; it provided fewer novel diagnostic insights than the expert clinical review. This pattern underscores that LLM evaluation functions most effectively as a scalable complementary validator rather than a primary clinical judge. Full cross-evaluator comparison is provided in Table S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p></sec><sec id="s3-7"><title>Performance Consistency and Reliability</title><p>Three prespecified consistency indicators confirmed that EvaluationPlus produced reliable and systematically targeted improvements across the clinical validation cohort (<xref ref-type="table" rid="table6">Table 6</xref>).</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Performance consistency indicators for the clinical validation cohort (N=15). Shannon entropy computed from normalized per-dimension mean score profiles; H<sub>max</sub> = ln(7) = 1.946. Targeting differential = mean improvement in active listening, personalization, and complex thinking minus mean improvement in remaining four dimensions.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Indicators</td><td align="left" valign="bottom">Values</td><td align="left" valign="bottom">95% Bootstrap CI</td><td align="left" valign="bottom">Interpretations</td></tr></thead><tbody><tr><td align="left" valign="top">Within-subject residual uncertainty (SD&#x0394;)</td><td align="left" valign="top">2.53</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="top">Moderate-to-high individual variation in improvement magnitude</td></tr><tr><td align="left" valign="top">Overall effect size (Cohen dz)</td><td align="left" valign="top">0.881</td><td align="left" valign="top">0.32-2.20</td><td align="left" valign="top">Large within-subject effect</td></tr><tr><td align="left" valign="top">Directional improvement rate</td><td align="left" valign="top">13/15 (86.7%)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Broad directional consistency across participants</td></tr><tr><td align="left" valign="top">Individual gain range</td><td align="left" valign="top">+0.43 to+5.86</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">No extreme negative outliers post-exclusion</td></tr><tr><td align="left" valign="top">Cross-participant CV (Version A)<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup></td><td align="left" valign="top">20%</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">High preenhancement variability</td></tr><tr><td align="left" valign="top">Cross-participant CV (Version B)</td><td align="left" valign="top">8.1%</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Substantially improved rating consistency</td></tr><tr><td align="left" valign="top">Mean within-dimension SD (Version A)</td><td align="left" valign="top">2.55</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Higher spread; more variable participant ratings across dimensions</td></tr><tr><td align="left" valign="top">Mean within-dimension SD (Version B)</td><td align="left" valign="top">1.65</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Narrower spread; more uniform participant agreement</td></tr><tr><td align="left" valign="top">Shannon entropy &#x2014; Version A (H)</td><td align="left" valign="top">1.929</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Uneven cross-dimensional competency profile</td></tr><tr><td align="left" valign="top">Shannon entropy &#x2014; Version B (H)</td><td align="left" valign="top">1.943</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">More balanced competency profile (H<sub>max</sub>=1.946)</td></tr><tr><td align="left" valign="top">Targeting differential (&#x0394;<sub>targret</sub> &#x2212; &#x0394;<sub>nontargret</sub>)</td><td align="left" valign="top">+1.42 points</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Enhancement concentrated in prespecified clinical priorities</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>CV: coefficient of variation (SD/Mean &#x00D7; 100%)</p></fn><fn id="table6fn2"><p><sup>b</sup>Not available.</p></fn></table-wrap-foot></table-wrap><p>Within-subject residual uncertainty (SD&#x0394;=2.53) reflects moderate variability in individual improvement magnitude, expected given the complexity of therapeutic communication assessment in a heterogeneous pilot cohort. The corresponding overall effect size of dz=0.881 (95% bootstrap CI [0.32-2.20]) indicates that mean improvement was substantial relative to this variability.</p><p>Cross-participant rating consistency improved substantially. The coefficient of variation across the seven therapeutic dimensions decreased from 20% (version A) to 8.1% (version B), indicating that participant ratings converged markedly around the group mean in the enhanced system. Mean within-dimension SD similarly decreased from 2.55 to 1.65, reflecting more uniform participant agreement on enhanced system quality.</p><p>. Shannon entropy computed from normalized per-dimension mean score profiles increased from H=1.929 (version A) to H=1.943 (version B), approaching the theoretical maximum of H=1.946 for seven equally-weighted dimensions. This increase indicates that the enhanced system produced a more evenly distributed competency profile, reducing the concentration of performance in dominant dimensions that characterized the baseline system.</p><p>Targeting precision was confirmed by a differential of +1.42 points between prespecified target dimensions (mean gain =+3.04, SD 0.14) and nontargeted dimensions (mean gain = +1.62, SD 0.41), demonstrating that the expert-guided enhancement protocol produced gains concentrated in clinically intended areas rather than global undifferentiated improvement.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings and Framework Contributions</title><p>This pilot study demonstrates that EvaluationPlus provides a structured, reproducible quality assurance framework for therapeutic AI systems at the feasibility-validation stage. Unlike traditional evaluation workflows that conclude with diagnostic reporting, EvaluationPlus operationalizes a 3-stage improvement loop &#x2014; expert-guided diagnosis, multi-LLM gap&#x2013;solution mapping, and within-subject validation &#x2014; to support governed, clinically supervised enhancement. The substantial overall improvement (+41%; dz=0.881, 95% CI [0.32-2.20]) and large targeted gains in expert-prioritized dimensions confirm that structured quality-improvement protocols can selectively strengthen communication competencies without compromising nontargeted dimensions. Beyond average performance gains, improvements in cross-participant rating consistency (CV: 20% &#x2192; 8.1%), within-dimension score variability (mean SD: 2.55 &#x2192; 1.65), and cross-dimensional balance (Shannon entropy: H=1.929 &#x2192; 1.943) indicate that the framework promotes more stable and reliable therapeutic behavior. Together, these findings establish a proof of concept for evaluation-driven enhancement aligned with health care quality assurance principles, while acknowledging that generalization to clinical populations and real-world contexts requires further validation.</p></sec><sec id="s4-2"><title>Mechanisms Underlying Observed Improvements</title><p>Process-level analyses revealed coherent mechanisms through which the framework enhanced therapeutic quality. Strategic questioning, mandating at least one context-relevant follow-up question per response, increased question-led turns by 27.4 percentage points and was associated with improvements in active listening and personalization ratings. Severity-appropriate tone calibration reduced incongruent metaphors and over-validation in high-severity scenarios while preserving warmth in lower-severity interactions. Contextual personalization, through reflective summaries, use of prior-turn information, and reduction of generic advice, addressed baseline uniformity and yielded more &#x201C;human-like&#x201D; interactions reported by participants. These mechanisms align with counseling psychology evidence emphasizing focused inquiry, collaborative agenda-setting, and succinct communication [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>However, an important interpretive caveat applies: the three enhancement components co-occurred within the same protocol cycles and cannot be statistically disentangled in the current design. Specifically, response length decreased by 44.9% concurrently with the increase in questioning behavior, and improved ratings may reflect reduced cognitive load from shorter responses, higher-quality questioning, or an interaction of both. Future studies should orthogonally manipulate these variables to isolate their independent contributions to therapeutic quality.</p></sec><sec id="s4-3"><title>Clinical Implications and Scope of Application</title><p>The findings clarify both the potential and the boundaries of LLM-mediated therapeutic support. EvaluationPlus demonstrates that structured, expert-guided enhancement can produce meaningful communication improvements within a pilot validation context. The multi-LLM gap&#x2013;solution mapping component offers a scalable mechanism for translating expert diagnostic findings into prescriptive enhancement strategies, reducing reliance on informal prompt adjustment and supporting documentation of quality improvement decisions.</p><p>At the same time, this study assesses communication quality as rated by healthy university students, not clinical outcomes such as symptom reduction or engagement retention. These findings should not be interpreted as evidence of therapeutic efficacy. Any organizational adoption of the EvaluationPlus framework should be understood as a quality assurance methodology for iterative system improvement, not as a clinical effectiveness intervention.</p><p>The crisis scenario findings warrant particular attention. Although the enhanced system improved questioning and tone in simulated crisis interactions, it remained limited to generic referrals and could not conduct comprehensive risk assessments or individualized safety planning. LLMs lack legal accountability, lived empathy, and dynamic situational awareness essential for handling imminent risk. Crisis responses were evaluated only by healthy participants using simulated scenarios, which substantially limits the ecological validity of crisis-related findings. Accordingly, any integration of AI-based therapeutic systems in contexts where crisis presentations are possible must include hybrid safety workflows: automatic risk-cue detection, immediate human handoff options, clinician review of flagged transcripts, and monitoring of near-miss events. Institutions should codify these safeguards within governance protocols and integrate escalation pathways into electronic health record or customer relationship management systems.</p></sec><sec id="s4-4"><title>Methodological Considerations</title><p>Two methodological issues merit explicit discussion. First, the EvaluationPlus framework uses the same 7-dimension therapeutic competency rubric across all three stages: as the diagnostic instrument in stage 1, as the target specification in stage 2, and as the outcome measure in stage 3. This design, while intentional for methodological consistency, introduces a circularity risk: improvements measured by the rubric may partly reflect optimization toward the rubric&#x2019;s own criteria rather than broader, context-independent therapeutic quality. Future validation studies should include outcome measures independent of the rubric used to guide enhancement, such as standardized therapeutic alliance scales or blinded third-party clinical assessments.</p><p>Second, participant feedback identified a new conversational pattern in the enhanced system: a perceivable rhythm of question followed by extended explanation, suggesting that the mandatory follow-up question rule may have replaced one form of structural rigidity with another. This finding illustrates a general design challenge in rule-based enhancement of conversational AI: explicit behavioral rules improve targeted metrics but may introduce secondary rigidity. Future enhancement cycles should explore probabilistic rather than mandatory questioning rules, and evaluate the naturalness of dialogue as an independent outcome dimension.</p><p>Third, the same LLM platforms used for stage 2 gap&#x2013;solution mapping were also used as 1 of 3 evaluator perspectives in stage 3 multistakeholder assessment. Although the prompts were sufficiently distinct &#x2014; stage 2 prompts requested prescriptive enhancement recommendations, while stage 3 prompts requested rubric-aligned evaluation of transcripts &#x2014; this overlap represents a potential evaluator&#x2013;mapper conflict. Future studies should use independent LLM platforms for mapping and evaluation to eliminate this circularity concern.</p></sec><sec id="s4-5"><title>Limitations</title><sec id="s4-5-1"><title>Sample Size and Generalizability</title><p>The modest sample (N=15) is appropriate for pilot feasibility testing but limits statistical precision and generalizability. The cohort consisted predominantly of Korean graduate students in applied AI in their 20s, which restricts application to other age groups, educational backgrounds, cultural contexts, and clinical populations with severe or chronic psychiatric conditions. The cohort shared background in applied AI may have introduced domain-specific evaluative expectations not representative of general help-seeking populations. Future validation studies should incorporate stratified sampling across demographic groups and clinical settings.</p></sec><sec id="s4-5-2"><title>Protocol Deviation</title><p>The stated protocol required participants to complete a minimum of two scenarios per version. However, 6 of 15 participants (40%) engaged primarily with one scenario per version, representing a deviation from this threshold. Dimensional analyses were conducted across all available scenario&#x2013;version pairings, but the uneven scenario coverage may have introduced variability in per-dimension estimates that cannot be fully controlled post hoc.</p></sec><sec id="s4-5-3"><title>Temporal Confound</title><p>Absolute ratings obtained in the current within-subject assessment were lower than those reported in the prior between-subject evaluation conducted approximately 18 months earlier [<xref ref-type="bibr" rid="ref6">6</xref>], likely reflecting the rapid evolution of LLM capabilities and the corresponding elevation of user expectations over that period rather than a deterioration in chatbot performance per se. Although the within-subject design controls for individual differences and contemporaneous contextual factors, it cannot fully eliminate this temporal confound. Future replication studies conducted within a compressed timeframe would more cleanly isolate enhancement effects from temporal drift.</p></sec><sec id="s4-5-4"><title>Single Expert Evaluator</title><p>Expert clinical validation relied on a single licensed clinical psychologist. The use of one evaluator precludes estimation of inter-rater reliability for the expert component, and the 75% preference rate should accordingly be treated as preliminary and hypothesis-generating rather than confirmatory.</p></sec><sec id="s4-5-5"><title>Safety&#x2013;Engagement Tension</title><p>Expert evaluation and participant feedback diverged in the crisis scenario: the expert rated the enhanced response highly for clinical safety and protocol adherence, while participants described it as &#x201C;professional but dismissive.&#x201D; This tension between guideline-adherent crisis responses and perceived conversational warmth represents a substantive open question for therapeutic AI development.</p></sec><sec id="s4-5-6"><title>Ecological Validity</title><p>Scenario-based assessments enable controlled comparison but cannot capture the spontaneity, variability, and longitudinal dynamics of real-world use. Crisis responses in particular were evaluated only in simulated contexts by healthy participants, substantially limiting inference about genuine emergency effectiveness. Composite averaging across heterogeneous severity levels (mild to crisis) may mask scenario-specific performance patterns, including potential variation in crisis-condition responding; future studies should report scenario-stratified outcomes separately.</p></sec><sec id="s4-5-7"><title>Outcome Scope</title><p>The study assessed communication quality as perceived by participants and one expert evaluator, not long-term clinical outcomes such as symptom reduction, therapeutic alliance, or engagement retention.</p></sec><sec id="s4-5-8"><title>Sequence Effects</title><p>Presentation order was self-selected by participants via independent platform access rather than experimentally controlled, precluding formal sequence effect testing. The potential influence of version order on ratings cannot be excluded and represents an uncontrolled source of variability in the within-subject comparison.</p></sec><sec id="s4-5-9"><title>Interrater Reliability</title><p>Formal Cohen &#x03BA; for stage 1 diagnostic deficit identification was not computed. The 2 practitioners reached consensus through iterative think-aloud discussion rather than independent blind numeric scoring. Future implementations should incorporate independent blind ratings prior to consensus to enable formal reliability estimation.</p></sec><sec id="s4-5-10"><title>Future Research Directions</title><p>Future work should address 5 priorities. First, large-scale validation with diverse demographic and clinical populations is needed to establish the external validity of EvaluationPlus beyond the current pilot cohort. Second, longitudinal studies evaluating symptom-level, engagement, and functional outcomes would clarify whether communication quality improvements translate into clinically meaningful benefits. Third, hybrid human-AI crisis response protocols with clinician-supervised escalation should be developed and evaluated to address the ecological validity limitations of simulated crisis assessment. Fourth, reproducibility testing across AI architectures, therapeutic domains, and health care systems would establish the framework&#x2019;s generalizability beyond Dr.CareSam. Fifth, implementation-science evaluations of workflow integration, clinician acceptance, and organizational governance impact would inform practical adoption pathways.</p></sec></sec><sec id="s4-6"><title>Conclusions</title><p>This pilot study demonstrates the feasibility of EvaluationPlus as a structured, reproducible framework for quality assurance in therapeutic AI systems. By integrating expert-guided diagnosis, multi-LLM gap&#x2013;solution mapping, and within-subject validation, the framework produced substantial and targeted improvements in therapeutic communication quality (overall dz=0.881), with consistent preference from both user (86.7%) and expert evaluators (75%). Methodological contributions include the operationalization of the evaluation-to-enhancement loop, the use of multi-LLM triangulation for prescriptive gap mapping, and the development of consistency metrics appropriate for pilot-stage validation. Although the study is limited by sample size, cultural specificity, single expert evaluation, and simulated crisis scenarios, it establishes a clear proof-of-concept for governed, evaluation-driven quality improvement in therapeutic AI development. Future work extending validation to diverse populations, longitudinal clinical outcomes, and hybrid crisis-response protocols will be essential for establishing the external validity and clinical use of this approach.</p></sec></sec></body><back><ack><p>We thank Eunjin Lee, licensed counseling psychologist, for her expert consultation during the Stage 1 diagnostic review of this study.</p><p>The funders had no role in study design, data collection, analysis, interpretation of results, or the decision to submit the manuscript for publication.</p><p>The authors declare that generative AI tools (Claude 4.0 Sonnet, Anthropic) were used to assist with English language editing and manuscript preparation. The same model family was also used as one of three platforms in Stage 2 multi-LLM gap&#x2013;solution mapping; this analytic use involved structured rubric prompts for therapeutic quality assessment of chatbot transcripts and was entirely separate from the editorial assistance described above. All scientific content, data analysis, interpretation, and conclusions are the sole responsibility of the authors.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the following funding sources:(1) The Ministry of Science and ICT (MSIT), Korea, under the Graduate School of Virtual Convergence support program (IITP-2026-RS-2023-00254129), supervised by the Institute for Information &#x0026; Communications Technology Planning &#x0026; Evaluation (IITP).(2) The MSIT, Korea, under the Global Scholars Invitation Program (grant number: RS-2024-00459638), also supervised by the IITP.(3) The Sports and Tourism R&#x0026;D Program through the Korea Creative Content Agency (KOCCA), funded by the Ministry of Culture, Sports and Tourism in 2024, under the project titled "Development of game-based digital therapeutics technology for adolescent mental health (psychological and behavioral control) management" (grant number: RS-2024-00344893).(4) The IITP grant funded by the Korean government (MSIT) (No. RS-2025-25442569, AI Star Fellowship Support Program, Sungkyunkwan University).(5) "Regional Innovation System &#x0026; Education (RISE)" through the Seoul RISE Center, funded by the Ministry of Education (MOE) and the Seoul Metropolitan Government (grant 2026-RISE-01-018-04).(6) The KOITA grant funded by the MSIT (No. S-2025-1855-000).(7) "Regional Innovation System &#x0026; Education (RISE)" through the Seoul RISE Center, funded by the Ministry of Education (MOE) and the Seoul Metropolitan Government (grant 2026-RISE-01-018-05).(8) Industry collaboration with Emotionwave (<ext-link ext-link-type="uri" xlink:href="https://emotionwave.com/">https://emotionwave.com</ext-link>).(9) The IITP grant funded by the Korean government (MSIT) (No. RS-2026-25520944, Development of XR Content Agent Technology Based on Emotion and Sensibility Reasoning, Sungkyunkwan University).</p></sec><sec><title>Data Availability</title><p>The deidentified participant-level score data supporting the findings of this study are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>. Due to the sensitive nature of mental health-related conversational data, full transcript datasets are not publicly available. Researchers requesting access to additional data may contact the corresponding author. The EvaluationPlus framework documentation, including the therapeutic competency rubric, evaluation instruments, and bilingual scenario scripts, is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization, methodology, software (framework development and data analysis scripts), formal analysis, investigation, data curation, writing &#x2013; original draft, writing &#x2013; review and editing, visualization, project administration: BK</p><p>Software (prompt tuning implementation), investigation, resources (participant recruitment), data curation, validation: KK</p><p>Writing &#x2013; review and editing, visualization: PH, SH, SC</p><p>Supervision, writing &#x2013; review and critical feedback, funding acquisition, project administration: HO</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">CV</term><def><p>coefficient of variation</p></def></def-item><def-item><term id="abb3">GPT</term><def><p>generative pre-trained transformer</p></def></def-item><def-item><term id="abb4">IITP</term><def><p>Institute for Information &#x0026; Communications Technology Planning &#x0026; Evaluation</p></def></def-item><def-item><term id="abb5">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb6">JMIR</term><def><p><italic>Journal of Medical Internet Research</italic></p></def></def-item><def-item><term id="abb7">KOCCA</term><def><p>Korea Creative Content Agency</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">MSIT</term><def><p>Ministry of Science and ICT</p></def></def-item><def-item><term id="abb10">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb11">PIPA</term><def><p>Personal Information Protection Act</p></def></def-item><def-item><term id="abb12">QA</term><def><p>quality assurance</p></def></def-item><def-item><term id="abb13">TCS</term><def><p>therapeutic communication score</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Auerbach</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Alonso</surname><given-names>J</given-names> </name><name name-style="western"><surname>Axinn</surname><given-names>WG</given-names> </name><etal/></person-group><article-title>Mental disorders among college students in the World Health Organization world mental health surveys</article-title><source>Psychol Med</source><year>2016</year><month>10</month><volume>46</volume><issue>14</issue><fpage>2955</fpage><lpage>2970</lpage><pub-id pub-id-type="doi">10.1017/S0033291716001665</pub-id><pub-id pub-id-type="medline">27484622</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harrer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Adam</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Baumeister</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Internet interventions for mental health in university students: a systematic review and meta-analysis</article-title><source>Int J Methods Psychiatr Res</source><year>2019</year><month>06</month><volume>28</volume><issue>2</issue><fpage>e1759</fpage><pub-id pub-id-type="doi">10.1002/mpr.1759</pub-id><pub-id pub-id-type="medline">30585363</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lipson</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Lattie</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Eisenberg</surname><given-names>D</given-names> </name></person-group><article-title>Increased rates of mental health service utilization by U.S. college students: 10-year population-level trends (2007-2017)</article-title><source>Psychiatr Serv</source><year>2019</year><month>01</month><day>1</day><volume>70</volume><issue>1</issue><fpage>60</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1176/appi.ps.201800332</pub-id><pub-id pub-id-type="medline">30394183</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="book"><source>National Survey on Mental Health</source><year>2021</year><publisher-name>Ministry of Health and Welfare</publisher-name><pub-id pub-id-type="doi">10.30773/pi.2022.0307</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="book"><person-group person-group-type="author"><collab>National Center for Health Statistics</collab></person-group><source>Health, United States, 2020&#x2013;2021: Annual Perspective</source><year>2023</year><publisher-name>NCHS</publisher-name><pub-id pub-id-type="doi">10.15620/cdc:122044</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>M</given-names> </name></person-group><article-title>Development and evaluation of a mental health chatbot using ChatGPT 4.0: mixed methods user experience study with Korean users</article-title><source>JMIR Med Inform</source><year>2025</year><month>01</month><day>3</day><volume>13</volume><fpage>e63538</fpage><pub-id pub-id-type="doi">10.2196/63538</pub-id><pub-id pub-id-type="medline">39752663</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chu</surname><given-names>SY</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>MY</given-names> </name></person-group><article-title>Think together and work better: combining humans&#x2019; and llms&#x2019; think-aloud outcomes for effective text evaluation</article-title><year>2025</year><month>04</month><day>26</day><conf-name>Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems (CHI &#x2019;25)</conf-name><conf-date>Apr 26 to May 1, 2025</conf-date><conf-loc>Yokohama Japan</conf-loc><publisher-name>ACM</publisher-name><pub-id pub-id-type="doi">10.1145/3706598.3713181</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hager</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jungmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Holland</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title><source>Nat Med</source><year>2024</year><month>09</month><volume>30</volume><issue>9</issue><fpage>2613</fpage><lpage>2622</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="medline">38965432</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>JI</given-names> </name><name name-style="western"><surname>Abbasian</surname><given-names>M</given-names> </name><name name-style="western"><surname>Azimi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Bounds</surname><given-names>DT</given-names> </name><name name-style="western"><surname>Jun</surname><given-names>A</given-names> </name><name name-style="western"><surname>Han</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Building trust in mental health chatbots: safety metrics and LLM-based evaluation tools</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 3, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.04650</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Ericsson</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Simon</surname><given-names>HA</given-names> </name></person-group><source>Protocol Analysis: Verbal Reports as Data</source><year>1993</year><edition/><publisher-name>MIT Press</publisher-name><pub-id pub-id-type="doi">10.7551/mitpress/5657.001.0001</pub-id><pub-id pub-id-type="other">9780262272391</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Choo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Endo</surname><given-names>K</given-names> </name><name name-style="western"><surname>Truong</surname><given-names>B</given-names> </name><name name-style="western"><surname>Son</surname><given-names>MH</given-names> </name></person-group><article-title>Advancing clinical chatbot validation using AI-powered evaluation with a new 3-bot evaluation system: instrument validation study</article-title><source>JMIR Nurs</source><year>2025</year><month>02</month><day>27</day><volume>8</volume><fpage>e63058</fpage><pub-id pub-id-type="doi">10.2196/63058</pub-id><pub-id pub-id-type="medline">40014000</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tam</surname><given-names>TYC</given-names> </name><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>28</day><volume>7</volume><issue>1</issue><fpage>258</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id><pub-id pub-id-type="medline">39333376</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Louie</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nandi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Brunskill</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name></person-group><article-title>Roleplay-doh: enabling domain-experts to create LLM-simulated patients via eliciting and adhering to principles</article-title><year>2024</year><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP 2024)</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, Florida, USA</conf-loc><publisher-name>ACL</publisher-name><fpage>10570</fpage><lpage>10603</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.591</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Moilanen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Visuri</surname><given-names>A</given-names> </name><name name-style="western"><surname>Suryanarayana</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Alorwu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yatani</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hosio</surname><given-names>S</given-names> </name></person-group><article-title>Measuring the effect of mental health chatbot personality on user engagement</article-title><year>2022</year><month>11</month><day>27</day><conf-name>MUM 2022</conf-name><conf-date>Nov 27-30, 2022</conf-date><conf-loc>Lisbon Portugal</conf-loc><publisher-name>ACM</publisher-name><fpage>138</fpage><lpage>150</lpage><pub-id pub-id-type="doi">10.1145/3568444.3568464</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bosma</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ichter</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title><source>Adv Neural Inf Process Syst</source><year>2022</year><publisher-name>Curran Associates</publisher-name><pub-id pub-id-type="doi">10.5555/3600270.3602070</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bae</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Evaluating and auditing LLM-driven chatbots for psychiatric patients in clinical mental health settings</article-title><access-date>2026-07-10</access-date><conf-name>CHI 2024 Workshop on Human-centered Evaluation and Auditing of Language Models (HEAL)</conf-name><conf-date>May 12, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://younghokim.net/files/papers/mindfuldiary-kim-heal24.pdf">https://younghokim.net/files/papers/mindfuldiary-kim-heal24.pdf</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaber</surname><given-names>F</given-names> </name><name name-style="western"><surname>Shaik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allega</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>9</day><volume>8</volume><issue>1</issue><fpage>263</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="medline">40346344</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Association</surname><given-names>AP</given-names> </name></person-group><source>Diagnostic and Statistical Manual of Mental Disorders</source><year>2013</year><edition>5</edition><publisher-name>American Psychiatric Publishing</publisher-name><pub-id pub-id-type="doi">10.1176/appi.books.9780890425596</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Hill</surname><given-names>CE</given-names> </name></person-group><source>Helping Skills: Facilitating Exploration, Insight, and Action</source><year>2009</year><edition>3</edition><publisher-name>American Psychological Association</publisher-name><pub-id pub-id-type="other">9781433804519</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weger</surname><given-names>H</given-names> </name><name name-style="western"><surname>Castle Bell</surname><given-names>G</given-names> </name><name name-style="western"><surname>Minei</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>MC</given-names> </name></person-group><article-title>The relative effectiveness of active listening in initial interactions</article-title><source>Int J Listen</source><year>2014</year><month>01</month><day>2</day><volume>28</volume><issue>1</issue><fpage>13</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.1080/10904018.2013.813234</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Norcross</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Lambert</surname><given-names>MJ</given-names> </name></person-group><source>Psychotherapy Relationships That Work: Volume 1 &#x2014; Evidence-Based Therapist Contributions</source><year>2019</year><edition>3</edition><publisher-name>Oxford University Press</publisher-name><pub-id pub-id-type="doi">10.1093/med-psych/9780190843953.003.0001</pub-id><pub-id pub-id-type="other">9780190843953</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Participant-level score summary and statistical documentation in Excel format, including raw scores (N=15), dimension summary, outlier documentation for excluded participant (z=&#x2212;2.60), and bias-corrected bootstrap CIs (B=5000 iterations).</p><media xlink:href="medinform_v14i1e87887_app1.xlsx" xlink:title="XLSX File, 13 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Evaluation instruments, therapeutic competency rubric with practice-informed anchors, user evaluation survey, expert clinical evaluation protocol, and bilingual scenario scripts (Korean and English) for four standardized mental health scenarios.</p><media xlink:href="medinform_v14i1e87887_app2.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Multi-large language model gap&#x2013;solution mapping documentation including structured prompt design, large language model recommendation matrix by dimension and platform (GPT 4.0, Claude 4.0 Sonnet, and Gemini 2.5 Flash), adjudication decision log for divergent recommendations, and relative platform contribution analysis.</p><media xlink:href="medinform_v14i1e87887_app3.docx" xlink:title="DOCX File, 21 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Three-cycle iterative enhancement protocol including modification procedures, targeted competencies, expert validation outcomes per cycle, design principles derived from the enhancement process, and quality assurance documentation.</p><media xlink:href="medinform_v14i1e87887_app4.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Qualitative thematic synthesis of participant feedback (Table S1), cross-evaluator preference convergence (Table S2), and representative response comparisons for presentation anxiety (Figure S1) and suicidal ideation crisis scenarios (Figure S2).</p><media xlink:href="medinform_v14i1e87887_app5.docx" xlink:title="DOCX File, 2208 KB"/></supplementary-material></app-group></back></article>