<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e83216</article-id><article-id pub-id-type="doi">10.2196/83216</article-id><article-categories><subj-group subj-group-type="heading"><subject>Tutorial</subject></subj-group></article-categories><title-group><article-title>A Secure User Interface for Preclinical Evaluation of AI in Patient Portal Message Management: Tutorial</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Gleason</surname><given-names>Kelly</given-names></name><degrees>PhD, RN</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kidu</surname><given-names>Thomas</given-names></name><degrees>MSE</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Babu</surname><given-names>Vignesh</given-names></name><degrees>MSE</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hasselfeld</surname><given-names>Brian</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wolff</surname><given-names>Jennifer</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>School of Nursing, Johns Hopkins University</institution><addr-line>525 N. Wolfe Street</addr-line><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff2"><institution>Whiting School of Engineering, Johns Hopkins University</institution><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff3"><institution>Bloomberg School of Public Health, Johns Hopkins University</institution><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Hagenimana</surname><given-names>Fabien</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Odeniya</surname><given-names>Joshua</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Romagnoli</surname><given-names>Katrina M</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>ROLAND</surname><given-names>ABI</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Kelly Gleason, PhD, RN, School of Nursing, Johns Hopkins University, 525 N. Wolfe Street, Baltimore, MD, 21215, United States, 1 708334876; <email>kgleaso2@jhmi.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>20</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e83216</elocation-id><history><date date-type="received"><day>02</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Kelly Gleason, Thomas Kidu, Vignesh Babu, Brian Hasselfeld, Jennifer Wolff. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 20.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e83216"/><abstract><p>The growing use of AI to support patient portal message management requires rigorous preclinical evaluation. Directly testing AI within electronic health record (EHR) systems poses significant safety, workflow, and data-governance risks. Here, we present a technical feasibility report on a secure user interface (UI) sandbox designed to enable clinical and technical teams to experiment with AI for portal messaging before clinical integration. In this context, a &#x201C;sandbox&#x201D; refers to a controlled, nonproduction environment that allows safe testing, prompt iteration, and evaluation of AI outputs without impacting live EHR systems or patient care. We developed a web UI in Python 3 with a modular backend for data handling and AI task execution that operates entirely within the institutional firewall. The system runs in a secure research environment equipped with an NVIDIA GRID T4-1Q graphics processing unit (GPU) and institutional access controls. We designed a deidentification pipeline to remove or replace personal health identifiers and assessed its precision. The platform supports single-message and batch workflows and exposes example large language model (LLM)&#x2013;enabled tasks such as authorship identification, message categorization, criticality flagging, and response drafting using zero-shot, one-shot, and few-shot prompting. The system successfully executed end-to-end workflows to ingest messages, run individual or batch AI analyses, and present outputs for review. Personal health information partial masking was applied across the corpus using a deidentification pipeline validated against 110 manually adjudicated entities (sensitivity 95.1%, precision 82.1%). We ran use cases with an institutional review board&#x2013;approved corpus of a dementia-relevant subset of 6941 patient portal messages categorized as &#x201C;medical advice requests&#x201D; from 497 unique patients. With the support of the UI, we tested which prompting strategies yielded interpretable outputs for authorship identification, categorization, and criticality flagging, and whether response drafting produced editable clinician starting points. A token-based cost readout provided transparent operating estimates for LLM-backed tasks. This framework offers a practical, secure path to test AI behavior on real messages without affecting live EHR workflows and thus supports exploratory testing, prompt iteration, and comparative analyses, including LLM prompts versus baseline models, while preserving governance boundaries. We discuss design choices, safety controls, and the limits of a sandbox approach. A secure, UI-based sandbox enables health system teams to evaluate AI for patient portal messaging before clinical integration. The goal is not to assume benefit but to generate evidence about feasibility, risks, and fit to clinical needs in a controlled setting.</p></abstract><kwd-group><kwd>patient portal</kwd><kwd>management</kwd><kwd>electronic health records</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd><kwd>large language models</kwd><kwd>tutorial</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Patient portal messaging has expanded rapidly and now encompasses clinical questions, symptom updates, and coordination tasks that can affect care quality and timeliness [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. As message volumes rise, concerns about clinician workload and burnout have grown [<xref ref-type="bibr" rid="ref2">2</xref>]. In parallel, health systems are exploring AI to help manage message flow through triage support, authorship cues, categorization, and draft responses [<xref ref-type="bibr" rid="ref5">5</xref>]. Despite optimism that AI may help, its true impact on workload, safety, and equity remains uncertain [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>The increasing use of AI to support patient portal message management requires rigorous testing and evaluation. Patient portal messages convey critical health information, from appointment reminders and laboratory results to educational content, making clarity, accuracy, and accessibility essential. Even minor ambiguities or technical errors in poorly designed AI can lead to misunderstandings and compromise care quality [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>Developing AI tools for portal inbox management is particularly challenging within live electronic health record (EHR) systems. Direct integration at early stages can introduce significant risks, including threats to data integrity and barriers to agile development [<xref ref-type="bibr" rid="ref12">12</xref>]. The sensitive nature of health information necessitates a secure, isolated environment to test messaging functions without endangering patient safety or system stability. A dedicated testing platform is therefore critical because it must realistically simulate message flows, support the creation and evaluation of dynamic content, and enable real-time interactions with patients and clinicians. Importantly, this environment should remain separate from the EHR during development and testing to allow researchers to gather rich feedback, iterate on message design, and validate functionality in a controlled setting to ensure that only vetted, effective approaches reach the live clinical environment.</p><p>A sandbox is a controlled, nonproduction environment that allows safe testing, prompt iteration, and evaluation of AI outputs without impacting live EHR systems or patient care. A preclinical sandbox can enable secure experiments with real messages, which is particularly valuable given the risks and difficulties of direct experimentation inside live EHRs [<xref ref-type="bibr" rid="ref12">12</xref>]. A well-designed sandbox could separate front-end interactions from back-end processing, provide robust deidentification, allow plug-and-play AI components, and offer transparent telemetry estimates, including latency, token counts, and error traces. Above all, it should keep experimentation out of production until teams understand behavior, constraints, and risks [<xref ref-type="bibr" rid="ref13">13</xref>]. Building on this need, we present a technical feasibility report describing a user interface (UI)&#x2013;based sandbox developed within a secure, governed research environment.</p></sec><sec id="s2"><title>Technical Architecture and Development</title><sec id="s2-1"><title>Setting and Governance</title><p>All development and experiments occurred within a secure research environment following institutional review board (IRB) approval and access controls. No production EHR systems were touched. The environment was equipped with an NVIDIA GRID T4-1Q graphics processing unit (GPU) to accelerate local natural language processing (NLP) pipelines and baseline model experiments.</p></sec><sec id="s2-2"><title>Software Stack and Architecture</title><p>The system is implemented in Python (version 3.12; Python Software Foundation) and Streamlit (version 1.46.1) to design the web interface. The backend is modular, with a data layer for loading and validating input, a privacy layer for deidentification, and an AI layer for task execution. Core packages included pandas and numpy for data handling, spaCy and scispaCy for entity recognition [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], Faker for personal health information (PHI) replacement tokens, and the OpenAI Python software development kit (SDK) configured for Azure OpenAI (GPT-4) for large language model (LLM) tasks. The deployed model was <italic>gpt-4.1,</italic> version <italic>2025-04-14</italic>, accessed using API version <italic>2024-12-01-preview</italic>. All dependencies were pinned in a requirements file to support reproducibility. The UI and backend are decoupled, which makes it easy to update prompts, switch model end points, or add tasks without reworking the interface. To support reproducibility, the companion source code, pinned requirements file, and synthetic test data are available in the public repository described in the Data Availability section.</p></sec><sec id="s2-3"><title>Data and Cohort Construction</title><p>We started with an IRB-approved corpus of patient portal messages over several years. For early feasibility experiments, we focused on the medical advice request message type. After standard cleaning, the working set included 6941 messages from 497 unique patient accounts in a cohort of older adults, described elsewhere [<xref ref-type="bibr" rid="ref16">16</xref>]. This subset was chosen to capture a range of clinical questions and caregiver-authored notes, but it is not intended to represent all portal message types [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>].</p></sec><sec id="s2-4"><title>Deidentification Pipeline</title><p>To protect patient privacy, we built a deidentification pipeline that detects and transforms personal health identifiers. Deidentification was used as an additional safeguard and as a testable privacy-preserving component, not as the only privacy control. The UI stayed within the institutional firewall, with messages only accessible by the IRB-approved study team. The pipeline uses spaCy and scispaCy named entity recognition (NER) models and custom regex patterns to identify names, health care provider titles, dates, phone numbers, addresses, and medication mentions. Replacements are deterministic per document to preserve within-message coherence; for example, the same name is mapped to the same synthetic token throughout a message. Deterministic replacement was used to preserve coherence within a single message, but synthetic placeholders were not intended to serve as persistent cross-thread pseudonyms. Conversation reconstruction was handled in the secure UI layer using internal message metadata. This design prioritizes message-level interpretability while reducing the risk that repeated placeholders could be used as linkage keys across messages. Covering medications required biomedical entity models, which we supplemented with curated lists where model coverage was limited. We sought to ensure that the output would preserve clinical semantics while removing identifiers.</p></sec><sec id="s2-5"><title>AI Tasks and Prompting Strategy</title><p>The platform was developed to enable experimentation with different AI use cases relevant to patient portal messaging. We studied four representative tasks: (1) authorship identification, to classify message authorship among 3 categories (patient, care partner, or ambiguous); (2) message categorization, to separate clinical content from administrative requests; (3) criticality flagging, to highlight urgent versus routine communications; and (4) response generation, to create editable draft replies that clinicians can review and refine.</p><p>All of these use cases were implemented using LLMs, tested with zero-shot, one-shot, and few-shot prompting strategies (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In the one-shot setting, the model is given a single example before the target message, while in the few-shot setting, multiple examples are provided to help the model generalize the intended pattern [<xref ref-type="bibr" rid="ref18">18</xref>]. Outputs were always framed as drafts for human review, as is currently standard practice in the use of LLM-generated drafts of patient portal messages [<xref ref-type="bibr" rid="ref5">5</xref>]. <xref ref-type="fig" rid="figure1">Figure 1</xref> summarizes the AI task orchestration layer, showing how a patient message is routed through selected use cases, processed with zero-shot, one-shot, or few-shot prompting, and returned as structured outputs for UI display and cost telemetry.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Prompt templates. UI: user interface.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e83216_fig01.png"/></fig></sec><sec id="s2-6"><title>UI and Workflows</title><p>The UI was designed to have several complementary workflows within a single platform. The most basic workflow enables the creation of scenarios from the full dataset of 6941 messages. Once loaded, the deidentification engine is automatically applied. Users can then select an enterprise identifier, which links patient and clinician messages, to reconstruct conversations. This enables side-by-side comparison of a conventional scenario with its AI-augmented counterpart, in which suggested responses or flagged messages are surfaced for clinician review. This feature supports task-level testing and the evaluation of AI within the broader flow of communication.</p><p>A second workflow allows users to enter a single message manually or upload a batch file of messages. In single-message mode, users paste a deidentified message, select one or more tasks, and view results in a side-by-side panel with the original redacted text. In batch mode, users upload a CSV file of messages; the system runs the deidentification pipeline and then executes selected tasks, producing a downloadable results file.</p><p>Additional features include a criticality ordering tool, which prioritizes messages according to urgency, and a cost analysis panel, which aggregates token use and estimates operational expenses. Together, these workflows extend the platform beyond isolated message-level testing, making it possible to simulate how AI would function in real-world messaging streams.</p></sec><sec id="s2-7"><title>Experiment Setup and Evaluation Approach</title><p>Experiments were designed to validate end-to-end functionality rather than establish clinical performance. We executed the deidentification pipeline across the 6941-message subset, then ran exemplar AI tasks in single-message and batch modes to confirm that the system produced interpretable outputs, maintained deidentification, and recorded provenance (prompt choice and task configuration).</p><p>Informal expert walkthroughs with clinical stakeholders were conducted to identify usability issues and assess whether outputs were appropriately constrained and reviewable. Participants included 2 physicians in health system information technology leadership roles, 3 internal medicine physicians, 1 nurse who triages patient portal messages, and a coalition of care partners, patient advocates, and health information technology researchers. Each walkthrough lasted 30 to 60 minutes and included an interface review, a demonstration of core features, and time for questions and feedback. We received consistent positive feedback regarding how valuable it would be to extend access to this sandbox for various AI-driven use cases. We also received feedback that more information regarding the patient&#x2019;s context should be uploaded into the UI to enable users to assess the outputs. A primary usability challenge identified was that a basic level of comfort with Python was required to open the UI in the secure environment. We also incorporated insights from interviews with individuals with dementia and their care partners regarding their perceptions of AI-generated patient portal message responses [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>We instrumented the UI to display token counts and approximate request costs. Given the volatility of LLM pricing, the cost panel was used primarily to compare prompt strategies and model choices rather than predict exact operational spending. As the focus was on the platform itself, we did not conduct head-to-head accuracy studies, which are reserved for future validation.</p></sec></sec><sec id="s3"><title>System Performance and Illustrative Findings</title><p>The system executed the complete pipeline, from message ingestion to output presentation, without accessing live EHR systems. Across the 6941-message feasibility corpus, the interface supported single-message testing, batch processing, prompt selection, task execution, cost telemetry, and export of results. Because this manuscript reports technical feasibility rather than formal model evaluation, we do not present task-level output distributions or performance metrics.</p><p>AI-supported task execution was averaged approximately 5 seconds end-to-end (deidentification plus all 4 AI tasks) per message. The application included retry handling for API rate-limit errors, with up to 3 retry attempts for HTTP 429 responses. Average message length was approximately 150 to 200 words, with prompt-based processing estimated at 300 to 400 input tokens and 100 to 150 output tokens per message. Using conservative pricing assumptions, this corresponded to an estimated upper-end cost of approximately US $0.044 per message. These values are presented as operational feasibility estimates rather than production benchmarks.</p><p>We ran deidentification by replacing personal names, dates, and contact details with deterministic placeholders while preserving contextual meaning. Biomedical entity coverage for medication mentions was achieved by combining scispaCy models with a small, curated lexicon. Inspection of deidentified text confirmed that messages remained coherent and suitable for downstream analysis, although occasional overmasking of medication names was observed. To evaluate the NER pipeline used for deidentification, we tested performance on a message set. Manual review of pipeline output was conducted to identify false positives and false negatives, yielding 110 adjudicated entities. The pipeline achieved 82.1% precision, 80.9% accuracy, 95.1% sensitivity, and 39.3% specificity, with 4 false negatives identified. Because the deidentification did not achieve 100% sensitivity, all nonsynthetic messages were treated as containing PHI, including after they underwent deidentification.</p><p>In the use cases, we found that authorship prompts produced rational, auditable rationales (eg, references to third-person phrasing or caregiver self-identification) useful for reviewer feedback. We observed failure modes, such as the AI wrongly detecting care partner message authorship when in fact a patient authored the message and mentioned a spouse or family member. We also tested whether categorization and criticality prompts yielded stable labels on representative samples, and whether disagreements across prompting strategies were visible in the UI. Showing the output from the categorization and criticality prompts allowed us to collect feedback from key partners (informatics leads and physicians) that may have been otherwise hard to collect. Similarly, we were able to explore whether response drafting generated concise, editable text. We observed that the model occasionally produced overly generic and lengthy replies that did not directly address the specific clinical concern raised in the message. Batch runs were completed on the 6941-message subset and exported to a results table that combined input metadata, task outputs, and prompt provenance.</p><p>The token panel surfaced per-message and aggregate token counts, allowing users to see how different prompt templates affect inference cost. This made trade-offs between few-shot accuracy and budget more tangible during experimentation.</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>A core challenge for AI in patient portal messaging is that most tools reach clinicians long before they have been studied with realistic data and users [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Our approach prioritizes a preclinical testbed that allows health system teams to probe behavior, iterate prompts, and examine costs in a controlled setting. Several design choices were important. First, we enforced a privacy boundary: all messages were partially masked before any model access and treated as containing PHI throughout the entire process. Second, we made prompts first-class, versioned objects so that results are reproducible and modifiable. Third, we exposed batch processing alongside single-message testing to support both qualitative review and early quantitative exploration without implying clinical readiness. Our initial corpus was intentionally narrow (dementia-relevant medical advice requests from older adults), which constrains claims about performance across other message types and populations. The UI architecture, however, was built to be extensible: additional message datasets, specialized lexicons, and prompt templates can be slotted in without changing the UI.</p><p>The sandbox helps align stakeholders around evidence rather than assumptions by making experimental outputs visible and auditable. Engineers can use the environment to compare prompt templates, routing logic, and cost trade-offs; clinicians can preview which outputs reduce workload and which create new review burden; and governance teams can assess safety and equity risks before any integration decisions. The current implementation uses the OpenAI Python SDK configured for Azure OpenAI within the institutional environment. The AI layer was designed for extensibility, meaning future versions could evaluate baseline classifiers, alternative model end points, or local open-weight models as institutional policies and infrastructure evolve. However, this feasibility study did not demonstrate model substitution or formal head-to-head model comparisons. This manuscript reports a technical feasibility study rather than a formal evaluation. Future work will use this environment to conduct formal validation, comparisons with baseline classifiers and local open-weight models, turnaround times, clinician workflow impact, and staff satisfaction.</p></sec><sec id="s5"><title>Limitations</title><p>This work has limitations. We did not conduct a formal accuracy study or randomized usability trial; our evaluation focused on feasibility, safety controls, and instrumentation. Deidentification quality, while strong in bench tests, requires formal validation against gold standards. While the sensitivity was strong, we continued to treat the messages as containing PHI after the deidentification. The UI was developed and tested in an environment within the institution&#x2019;s firewall, and nonsynthetic messages were only accessed by IRB-approved team members.</p><p>Deidentification was optimized for sensitivity to minimize PHI disclosure risk, accepting lower specificity as a deliberate trade-off. This level of overmasking, in which non-PHI terms, including some medication names, were redacted, may have reduced the semantic richness available to the LLM during preproduction testing and could have affected the quality of generated responses. The extent of this impact was not formally evaluated and represents a limitation of the current study. Medication and identifier coverage depends on biomedical NER supplemented by regex-based detection, curated medication lists, text normalization, and fuzzy matching. Although these steps improve detection of common variants and misspellings, broader use across clinical settings will require periodic updates for new medications, local abbreviations, and institution-specific terminology. The pipeline was designed with HIPAA (Health Insurance Portability and Accountability Act) Safe Harbor identifier categories in mind, but it should not be interpreted as a certified Safe Harbor or Expert Determination implementation. In settings where data would leave the institutional security perimeter, formal validation of deidentification accuracy should be evaluated before using the workflow outside a similarly governed environment.</p><p>The UI also lacked automated safeguards, such as confidence thresholds or contradiction detection. Future work will require automated safeguards and explicit strategies for long-thread management, such as truncation, summarization, or controlled context-window selection, for use cases that require reasoning across full conversation histories. Token-based cost views are approximate and will shift with pricing and prompt designs. Finally, because the system deliberately avoids live EHR integration, we did not assess downstream operational effects such as turnaround times or staff satisfaction.</p><p>The UI does not yet allow the import of demographic data; thus, we were unable to assess differential performance across demographics. With the added feature of demographic data imports in future versions, this environment could enable assessment of differential performance across demographic characteristics, such as age, race, and ethnicity, to evaluate the presence of bias and equity considerations before deployment.</p><p>Despite these limitations, a secure UI-based sandbox is a pragmatic step toward responsible AI for patient messaging. It allows organizations to study real behaviors on real messages securely, identify failure modes early, and decide which use cases, if any, merit the cost and complexity of a full integration.</p></sec><sec id="s6" sec-type="conclusions"><title>Conclusions</title><p>AI for patient portal messaging is advancing quickly, but its real impact on clinicians, patients, and care partners remains uncertain [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. A secure sandbox with an accessible UI provides a practical way to test and refine AI use cases before they reach clinical workflows. Our implementation demonstrates end-to-end feasibility across the full pipeline: data ingestion, privacy preservation, configurable prompting, batch analysis, and transparent cost telemetry, all within a governed environment. The goal is not to assume benefit but to create the conditions under which benefit, risk, and fit can be studied with rigor.</p></sec></body><back><ack><p>The authors wish to acknowledge Dr Renee Blanding for her leadership of the internship program that led to this collaboration. The authors acknowledge using Claude (Anthropic) for grammar and style editing of the paper.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the following grants: National Institutes of Health (NIH) National Institute on Aging (NIA) grant R35AG072310, Consumer Health Information Technology to Engage and Support ADRD Caregivers: Research Program to Address ADRD Implementation Milestone 13; the Hopkins Artificial Intelligence and Technology Collaboratory (AITC), NIA grant P30AG073104; and the Hopkins Economics of Alzheimer&#x2019;s Disease and Services (HEADS) Center (NIH NIA P30AG066587).</p></sec><sec><title>Data Availability</title><p>The source code, pinned requirements file, and synthetic example dataset used to test the user interface are available in a public GitHub repository [<xref ref-type="bibr" rid="ref23">23</xref>]. The 6941-message corpus contains personal health information and is not publicly available due to institutional review board (IRB) restrictions. Researchers interested in accessing the dataset may submit a formal data sharing request to the corresponding author; any access would be subject to IRB approval and a data use agreement.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: KG</p><p>Methodology: KG, TK, VB, BH, JW</p><p>Supervision: JW</p><p>Validation: TK, VB</p><p>Writing&#x2014;original draft: KG, TK, VB</p><p>Writing&#x2014;review and editing: KG, TK, VB, BH, JW</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb2">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb3">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb4">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">NER</term><def><p>named entity recognition</p></def></def-item><def-item><term id="abb7">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb8">PHI</term><def><p>personal health information</p></def></def-item><def-item><term id="abb9">SDK</term><def><p>software development kit</p></def></def-item><def-item><term id="abb10">UI</term><def><p>user interface</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steitz</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Unertl</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Levy</surname><given-names>MA</given-names> </name></person-group><article-title>An analysis of electronic health record work to manage asynchronous clinical messages among breast cancer care teams</article-title><source>Appl Clin Inform</source><year>2021</year><month>08</month><volume>12</volume><issue>4</issue><fpage>877</fpage><lpage>887</lpage><pub-id pub-id-type="doi">10.1055/s-0041-1735257</pub-id><pub-id pub-id-type="medline">34528233</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lieu</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Altschuler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Weiner</surname><given-names>JZ</given-names> </name><etal/></person-group><article-title>Primary care physicians&#x2019; experiences with and strategies for managing electronic messages</article-title><source>JAMA Netw Open</source><year>2019</year><month>12</month><day>2</day><volume>2</volume><issue>12</issue><fpage>e1918287</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2019.18287</pub-id><pub-id pub-id-type="medline">31880798</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Crotty</surname><given-names>BH</given-names> </name><name name-style="western"><surname>Mostaghimi</surname><given-names>A</given-names> </name><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bajracharya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Safran</surname><given-names>C</given-names> </name><name name-style="western"><surname>Landon</surname><given-names>BE</given-names> </name></person-group><article-title>Prevalence and risk profile of unread messages to patients in a patient web portal</article-title><source>Appl Clin Inform</source><year>2015</year><volume>6</volume><issue>2</issue><fpage>375</fpage><lpage>382</lpage><pub-id pub-id-type="doi">10.4338/ACI-2015-01-CR-0006</pub-id><pub-id pub-id-type="medline">26171082</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borre</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Nicholas</surname><given-names>MW</given-names> </name></person-group><article-title>The disproportionate burden of electronic health record messages with image attachments in dermatology</article-title><source>J Am Acad Dermatol</source><year>2022</year><month>02</month><volume>86</volume><issue>2</issue><fpage>492</fpage><lpage>494</lpage><pub-id pub-id-type="doi">10.1016/j.jaad.2021.09.026</pub-id><pub-id pub-id-type="medline">34555485</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcia</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Artificial intelligence-generated draft replies to patient inbox messages</article-title><source>JAMA Netw Open</source><year>2024</year><month>03</month><day>4</day><volume>7</volume><issue>3</issue><fpage>e243201</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.3201</pub-id><pub-id pub-id-type="medline">38506805</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hristidis</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ruggiano</surname><given-names>N</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>EL</given-names> </name><name name-style="western"><surname>Ganta</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Stewart</surname><given-names>S</given-names> </name></person-group><article-title>ChatGPT vs Google for queries related to dementia and other cognitive decline: comparison of results</article-title><source>J Med Internet Res</source><year>2023</year><month>07</month><day>25</day><volume>25</volume><fpage>e48966</fpage><pub-id pub-id-type="doi">10.2196/48966</pub-id><pub-id pub-id-type="medline">37490317</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yalamanchili</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sengupta</surname><given-names>B</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Quality of large language model responses to radiation oncology patient care questions</article-title><source>JAMA Netw Open</source><year>2024</year><month>04</month><day>1</day><volume>7</volume><issue>4</issue><fpage>e244630</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.4630</pub-id><pub-id pub-id-type="medline">38564215</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chiarelli</surname><given-names>G</given-names> </name><name name-style="western"><surname>Stephens</surname><given-names>A</given-names> </name><name name-style="western"><surname>Finati</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Adequacy of prostate cancer prevention and screening recommendations provided by an artificial intelligence-powered large language model</article-title><source>Int Urol Nephrol</source><year>2024</year><month>08</month><volume>56</volume><issue>8</issue><fpage>2589</fpage><lpage>2595</lpage><pub-id pub-id-type="doi">10.1007/s11255-024-04009-5</pub-id><pub-id pub-id-type="medline">38564079</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dergaa</surname><given-names>I</given-names> </name><name name-style="western"><surname>Fekih-Romdhane</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hallit</surname><given-names>S</given-names> </name><etal/></person-group><article-title>ChatGPT is not ready yet for use in providing mental health assessment and interventions</article-title><source>Front Psychiatry</source><year>2024</year><volume>14</volume><fpage>1277756</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2023.1277756</pub-id><pub-id pub-id-type="medline">38239905</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Antaki</surname><given-names>F</given-names> </name><name name-style="western"><surname>Touma</surname><given-names>S</given-names> </name><name name-style="western"><surname>Milad</surname><given-names>D</given-names> </name><name name-style="western"><surname>El-Khoury</surname><given-names>J</given-names> </name><name name-style="western"><surname>Duval</surname><given-names>R</given-names> </name></person-group><article-title>Evaluating the performance of ChatGPT in ophthalmology: an analysis of its successes and shortcomings</article-title><source>Ophthalmol Sci</source><year>2023</year><volume>3</volume><issue>4</issue><fpage>100324</fpage><pub-id pub-id-type="doi">10.1016/j.xops.2023.100324</pub-id><pub-id pub-id-type="medline">37334036</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sendak</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Ratliff</surname><given-names>W</given-names> </name><name name-style="western"><surname>Sarro</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Real-world integration of a sepsis deep learning technology into routine clinical care: implementation study</article-title><source>JMIR Med Inform</source><year>2020</year><month>07</month><day>15</day><volume>8</volume><issue>7</issue><fpage>e15182</fpage><pub-id pub-id-type="doi">10.2196/15182</pub-id><pub-id pub-id-type="medline">32673244</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shortliffe</surname><given-names>EH</given-names> </name><name name-style="western"><surname>Sep&#x00FA;lveda</surname><given-names>MJ</given-names> </name></person-group><article-title>Clinical decision support in the era of artificial intelligence</article-title><source>JAMA</source><year>2018</year><month>12</month><day>4</day><volume>320</volume><issue>21</issue><fpage>2199</fpage><lpage>2200</lpage><pub-id pub-id-type="doi">10.1001/jama.2018.17163</pub-id><pub-id pub-id-type="medline">30398550</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Honnibal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Montani</surname><given-names>I</given-names> </name><name name-style="western"><surname>Van Landeghem</surname><given-names>S</given-names> </name><name name-style="western"><surname>Boyd</surname><given-names>A</given-names> </name></person-group><article-title>spaCy: industrial-strength natural language processing in Python</article-title><source>Zenodo</source><year>2020</year><access-date>2026-07-08</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/10009823">https://zenodo.org/records/10009823</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Neumann</surname><given-names>M</given-names> </name><name name-style="western"><surname>King</surname><given-names>D</given-names> </name><name name-style="western"><surname>Beltagy</surname><given-names>I</given-names> </name><name name-style="western"><surname>Ammar</surname><given-names>W</given-names> </name></person-group><article-title>ScispaCy: fast and robust models for biomedical natural language processing</article-title><source>Proceedings of the 18th BioNLP Workshop and Shared Task</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><pub-id pub-id-type="doi">10.18653/v1/W19-5034</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gleason</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Wec</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Use of the patient portal among older adults with diagnosed dementia and their care partners</article-title><source>Alzheimers Dement</source><year>2023</year><month>12</month><volume>19</volume><issue>12</issue><fpage>5663</fpage><lpage>5671</lpage><pub-id pub-id-type="doi">10.1002/alz.13354</pub-id><pub-id pub-id-type="medline">37354066</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gleason</surname><given-names>K</given-names> </name><name name-style="western"><surname>DeGennaro</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating ChatGPT in portal messaging for dementia care: identifying care partners and exploring acceptance</article-title><source>Innov Aging</source><year>2025</year><month>12</month><day>1</day><volume>9</volume><issue>Supplement_2</issue><fpage>igaf122.380</fpage><pub-id pub-id-type="doi">10.1093/geroni/igaf122.380</pub-id><pub-id pub-id-type="medline">40255279</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Mann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Ryder</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Language models are few-shot learners</article-title><source>NIPS &#x2019;20: Proceedings of the 34th International Conference on Neural Information Processing Systems</source><year>2020</year><publisher-name>Curran Associates Inc</publisher-name><fpage>1877</fpage><lpage>1901</lpage><pub-id pub-id-type="doi">10.5555/3495724.3495883</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>AP</given-names> </name><etal/></person-group><article-title>Leveraging large language models for generating responses to patient messages-a subjective analysis</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>05</month><day>20</day><volume>31</volume><issue>6</issue><fpage>1367</fpage><lpage>1379</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae052</pub-id><pub-id pub-id-type="medline">38497958</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shimada</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Petrakis</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Rothendler</surname><given-names>JA</given-names> </name><etal/></person-group><article-title>An analysis of patient-provider secure messaging at two Veterans Health Administration medical centers: message content and resolution through secure messaging</article-title><source>J Am Med Inform Assoc</source><year>2017</year><month>09</month><day>1</day><volume>24</volume><issue>5</issue><fpage>942</fpage><lpage>949</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocx021</pub-id><pub-id pub-id-type="medline">28371896</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Semere</surname><given-names>W</given-names> </name><name name-style="western"><surname>Crossley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karter</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Secure messaging with physicians by proxies for patients with diabetes: findings from the ECLIPPSE study</article-title><source>J Gen Intern Med</source><year>2019</year><month>11</month><volume>34</volume><issue>11</issue><fpage>2490</fpage><lpage>2496</lpage><pub-id pub-id-type="doi">10.1007/s11606-019-05259-1</pub-id><pub-id pub-id-type="medline">31428986</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>W</given-names>  <suffix>3rd</suffix></name><name name-style="western"><surname>Balyan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Karter</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Challenges and solutions to employing natural language processing and machine learning to measure patients&#x2019; health literacy and physician writing complexity: the ECLIPPSE study</article-title><source>J Biomed Inform</source><year>2021</year><month>01</month><volume>113</volume><fpage>103658</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2020.103658</pub-id><pub-id pub-id-type="medline">33316421</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>thom22/Patient-Portal-AI-Sandbox</article-title><source>GitHub</source><access-date>2026-07-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/thom22/Patient-Portal-AI-Sandbox">https://github.com/thom22/Patient-Portal-AI-Sandbox</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompt templates.</p><media xlink:href="medinform_v14i1e83216_app1.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material></app-group></back></article>