<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMI</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id>
      <journal-title>JMIR Medical Informatics</journal-title>
      <issn pub-type="epub">2291-9694</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v14i1e91151</article-id>
      <article-id pub-id-type="pmid">42809845</article-id>
      <article-id pub-id-type="doi">10.2196/91151</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Detecting Misspelled Drug Names Using Transformer-Based Language Models: Model Development and External Validation</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Vengadassalapathy</surname>
            <given-names>Srinivasan</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Klemen</surname>
            <given-names>Matej</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Patel</surname>
            <given-names>Anant</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Lu</surname>
            <given-names>Jiayu</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-1617-7267</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>McConeghy</surname>
            <given-names>Kevin W</given-names>
          </name>
          <degrees>PharmD, PhD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5056-0431</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Zullo</surname>
            <given-names>Andrew R</given-names>
          </name>
          <degrees>PharmD, PhD</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1673-4570</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Chen</surname>
            <given-names>Jinying</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Department of Medicine/Section of Preventive Medicine and Epidemiology</institution>
            <institution>Boston University Chobanian &#38; Avedisian School of Medicine</institution>
            <addr-line>72 E Concord St</addr-line>
            <addr-line>Boston, MA, 02118</addr-line>
            <country>United States</country>
            <phone>1 617 358 5838</phone>
            <email>jinychen@bu.edu</email>
          </address>
          <xref rid="aff5" ref-type="aff">5</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-7259-4301</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Department of Medicine/Section of Preventive Medicine and Epidemiology</institution>
        <institution>Boston University Chobanian &#38; Avedisian School of Medicine</institution>
        <addr-line>Boston, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Department of Health Services, Policy and Practice</institution>
        <institution>Brown University School of Public Health</institution>
        <addr-line>Providence, RI</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Transformative Health Systems Research to Improve Veteran Equity and Independence Center of Innovation (THRIVE COIN)</institution>
        <institution>Veterans Affairs Providence Health Care System</institution>
        <addr-line>Providence, RI</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>Department of Epidemiology</institution>
        <institution>Brown University School of Public Health</institution>
        <addr-line>Providence, RI</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff5">
        <label>5</label>
        <institution>Data Science Core</institution>
        <institution>Boston University Chobanian &#38; Avedisian School of Medicine</institution>
        <addr-line>Boston, MA</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Jinying Chen <email>jinychen@bu.edu</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>29</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>14</volume>
      <elocation-id>e91151</elocation-id>
      <history>
        <date date-type="received">
          <day>19</day>
          <month>1</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>15</day>
          <month>3</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>27</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>10</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Jiayu Lu, Kevin W McConeghy, Andrew R Zullo, Jinying Chen. Originally published in JMIR Medical Informatics (https://medinform.jmir.org), 29.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on https://medinform.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://medinform.jmir.org/2026/1/e91151" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Misspellings in medication names can compromise patient safety, reduce data utility, and impede large-scale data initiatives that integrate medication information from electronic health records (EHRs). Existing methods for detecting misspelled medical terms are mostly dictionary-based and can lead to high false-positive rates when correctly spelled but previously unseen (out-of-vocabulary) terms are encountered.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>We aimed to develop and validate domain-specific, transformer-based language models for detecting misspelled drug names, with an emphasis on performance for unseen terms.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>Using RxNorm as a standardized drug vocabulary, we created an RxNorm-augmented training corpus and developed two BERT (Bidirectional Encoder Representations from Transformers)–based models—BERTDrug and CharBERTDrug—for misspelling detection. Specifically, we randomly split 69,824 RxNorm drug names into training, development, and test sets (3:1:1) and generated k misspellings per name using text-perturbation techniques (k optimized for training; fixed at 1 for development and test sets). The models were fine-tuned on the training and development sets and evaluated using the RxNorm test set and 3586 drug names from the Long-Term Care Data Cooperative (LTCDC) database (external validation). The RxNorm test set and out-of-vocabulary LTCDC dataset (1922 terms), neither overlapping with the RxNorm training data, were used to evaluate performance on unseen terms. SpellChecker served as a dictionary-based baseline, while fastText<sub>ML</sub> and BioWordVec<sub>ML</sub>, which used different subword embeddings as inputs for machine learning, served as additional baselines. Additionally, we compared model performance with GPT-4o, a generative large language model (LLM), using 2200 randomly sampled test terms.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>On the RxNorm test set, BERTDrug and CharBERTDrug outperformed the baseline models across most metrics. BERTDrug achieved the best overall performance (<italic>F</italic><sub>1</sub>-score=0.859; area under the receiver operating characteristic curve [ROC-AUC]=0.947), followed by CharBERTDrug (<italic>F</italic><sub>1</sub>-score=0.833; ROC-AUC=0.906). Both models also outperformed the baseline models on the out-of-vocabulary LTCDC dataset across most metrics, with CharBERTDrug performing best (<italic>F</italic><sub>1</sub>-score=0.696; ROC-AUC=0.788), followed by BERTDrug (<italic>F</italic><sub>1</sub>-score=0.669; ROC-AUC=0.786). In the secondary analysis, both models exceeded GPT-4o on most metrics (except Recall) for 2000 RxNorm terms. BERTDrug performed best (ROC-AUC=0.951; <italic>F</italic><sub>1</sub>-score=0.855), followed by CharBERTDrug (ROC-AUC=0.911; <italic>F</italic><sub>1</sub>-score=0.831) and GPT-4o (ROC-AUC=0.856; <italic>F</italic><sub>1</sub>-score=0.721). In contrast, among 200 randomly selected LTCDC terms, GPT-4o performed best on most metrics except precision and specificity.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Domain-specific language models improved detection of misspellings in out-of-vocabulary drug names and outperformed baseline models in both internal and external evaluations. The comparison with a generative LLM suggests that domain shift may substantially reduce the advantages conferred by domain-specific training. With further fine-tuning on diverse data that capture the terminology, formatting conventions, and spelling patterns encountered across real-world clinical settings, these models could be adapted for use in other clinical databases and EHR systems to improve medication data quality for research and to support future safety-focused applications.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>misspelling detection</kwd>
        <kwd>transformer-based language models</kwd>
        <kwd>large language models</kwd>
        <kwd>medication names</kwd>
        <kwd>clinical repositories</kwd>
        <kwd>data augmentation using RxNorm</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Medication names and labeling conventions are complex and susceptible to misspellings and typographical errors [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. These errors can arise from several sources, including phonetic similarity and keyboard input errors (eg, adjacent key swaps, extra or missing characters, and accidental key presses). They may occur when health providers manually enter prescription orders or document treatment information in clinical notes within the electronic health record (EHR) system [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. These errors can compromise patient safety and impede the effective use of EHR data for clinical research. For example, a misspelled drug name could lead to confusion about what drug should be administered and may result in a necessary treatment being held while the issue is reconciled [<xref ref-type="bibr" rid="ref9">9</xref>]. Computerized provider order entry (CPOE) systems have been shown to significantly reduce medication prescribing errors, including typographical errors and misspellings [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. However, misspelled query terms can potentially lead to search inefficiencies and CPOE-related medication errors [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. Additionally, allowing the use of free-text instructions to enter orders within CPOE systems can further contribute to medication errors [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>].</p>
      <p>In addition to EHR systems, misspelled and nonstandard medication or drug names are commonly found in various other sources, including medical research publications [<xref ref-type="bibr" rid="ref14">14</xref>], online surveillance systems for adverse drug events [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], and social media postings related to adverse drug events [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Detecting these errors is crucial for enhancing the reliability and utility of these information resources to support pharmacovigilance and medication-related research.</p>
      <p>Furthermore, large data initiatives [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>] aimed at advancing patient care, enabling precision medicine, and supporting comparative effectiveness research often integrate medication information from various sources, including both structured and free-text fields from EHRs. As a result, misspelled medication names may be propagated into these databases without manual review. For example, within the database of the Long-Term Care Data Cooperative (LTCDC), we found spelling errors such as <italic>simeth</italic><bold><italic>a</italic></bold><italic>cone</italic> for <italic>simeth</italic><bold><italic>i</italic></bold><italic>cone</italic>, <italic>morphin sulfate</italic> for <italic>morphin</italic><bold><italic>e</italic></bold> <italic>sulfate</italic>, <italic>lev</italic><bold><italic>i</italic></bold><italic>s</italic><bold><italic>o</italic></bold><italic>n</italic> for <italic>Levs</italic><bold><italic>i</italic></bold><italic>n</italic>, and <italic>Zo</italic><bold><italic>rf</italic></bold><italic>an</italic> for <italic>Zo</italic><bold><italic>fr</italic></bold><italic>an</italic>. Misspelling detection is an important early step toward standardization and harmonization of such medication information.</p>
      <p>Existing approaches to detecting misspellings in medical terms are primarily dictionary-based [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Recent research has explored the application of deep learning methods to correct misspellings in free clinical text [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. However, none of them focused on medication names. Furthermore, previous studies have rarely addressed the challenge of distinguishing between misspelled medical terms and correctly spelled out-of-vocabulary terms, which can commonly arise when new medication products enter the market.</p>
      <p>We aimed to develop and validate a deep learning approach for detecting misspelled drug names in medication orders and administration records. Our approach leveraged 2 techniques: transformer-based language models and data augmentation using RxNorm. The problem of misspelling correction can be solved in 2 steps: error detection and error correction [<xref ref-type="bibr" rid="ref24">24</xref>]. We focused on error detection because it directly impacts the downstream task of error correction, and distinguishing between misspelled and correctly spelled out-of-vocabulary terms is itself a challenging task that warrants independent evaluation. Our results showed that RxNorm-augmented, domain-specific language models improved detection of misspellings in out-of-vocabulary drug names and outperformed both a dictionary-based baseline and 2 machine learning–based baseline models. It also outperformed GPT-4o, a generative large language model (LLM), on the internal evaluation dataset.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design</title>
        <p>We developed 2 transformer-based language models—BERTDrug and CharBERTDrug—to detect misspelled drug names (<xref rid="figure1" ref-type="fig">Figure 1</xref>). We fine-tuned the models on an augmented dataset derived from RxNorm [<xref ref-type="bibr" rid="ref25">25</xref>] and evaluated their performance on both a held-out RxNorm-derived test set (internal validation) and drug names from the LTCDC database [<xref ref-type="bibr" rid="ref20">20</xref>] (external validation). A dictionary-based method (SpellChecker) and 2 machine learning–based methods (fastText<sub>ML</sub> and BioWordVec<sub>ML</sub>) served as the baselines.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Study overview. The study was conducted in two phases: (1) development and internal validation of five models (CharBERTDrug, BERTDrug, SpellChecker, fastText<sub>ML</sub>, and BioWordVec<sub>ML</sub>) for detecting misspelled medication names, using an augmented dataset derived from RxNorm drug names, and (2) external validation using the LTCDC dataset. BERT: Bidirectional Encoder Representations from Transformers; LTCDC: Long-Term Care Data Cooperative.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e91151_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>The Institutional Review Board (IRB) of Boston University Chobanian &#38; Avedisian School of Medicine (H-44363) reviewed the protocol for this study and determined the analysis of secondary data to be human subjects research exempt from IRB review. The IRB granted a waiver of HIPAA (Health Insurance Portability and Accountability Act) Authorization for Research under 45 CFR 164.512(i)(2)(ii). All methods were carried out in accordance with relevant guidelines and regulations.</p>
      </sec>
      <sec>
        <title>Data</title>
        <sec>
          <title>Augmented Data for Model Development and Internal Validation</title>
          <p>We used RxNorm, a large database of clinical drug information [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], to create a dataset for model development and internal validation. RxNorm is a normalized naming system for generic and branded clinical drugs. It standardizes drug names and maps them to drug vocabularies commonly used in pharmacy management systems. Because RxNorm is intended to represent medications and medication-related concepts rather than arbitrary nondrug terms, unrelated nondrug concepts (eg, diseases, procedures, and anatomy terms) are typically not included in RxNorm. We generated the augmented dataset in 3 steps: extracting drug names from RxNorm, preprocessing the terms (eg, converting them to lowercase), and generating misspelled drug names, as detailed in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. We applied text perturbation techniques—comprising the following five operations described in Zhuang and Zuccon’s study [<xref ref-type="bibr" rid="ref27">27</xref>]—to generate synthetic positive instances with typographical errors: (1) insertion: randomly insert a letter into a word; (2) deletion: randomly delete a letter in a word; (3) substitution: randomly substitute a letter in a word; (4) swapping: randomly swap two adjacent characters in a word; and (5) adjacent keyboard character substitution: randomly replace a letter in a word with an adjacent character on the keyboard.</p>
        </sec>
        <sec>
          <title>External Validation Set</title>
          <p>We used medication names extracted from the LTCDC database [<xref ref-type="bibr" rid="ref20">20</xref>] to externally evaluate model performance. The LTCDC database harmonized nursing home medical records from divergent EHR vendors into a common data model, including medication administrations and orders. Although most medication names were drawn from standardized databases of these EHR vendors, less common drugs could be populated by free text that was pulled into the medication name field in the LTCDC database. To ensure adequate representation of misspelled medication names in this evaluation, we used 2 complementary sampling approaches—randomly sampling from low frequency and from all LTCDC medication names, respectively. We then annotated the LTCDC datasets using a 3-phase iterative process and a semiautomatic approach. <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> details the data sampling and annotation process.</p>
          <p>In total, 4000 medication names from two LTCDC datasets were annotated, with 3093 negative instances (annotated as “correct drug name,” “correct nondrug name,” or “correct short drug name”), and 907 positive instances (annotated as “misspelled drug name,” “misspelled nondrug name,” or “not sure or nonstandard”). We then eliminated duplicates for 197 terms (1 positive and 196 negative) shared between LTCDC dataset-1 and dataset-2, resulting in a final LTCDC dataset of 3803 unique terms. For the primary evaluation, we derived two evaluation sets from these terms: (1) a cleaned dataset (3586 terms), which excluded nondrug terms and terms classified as “not sure” or nonstandard, and (2) a cleaned out-of-vocabulary dataset (1922 terms), which was derived from the cleaned dataset by excluding terms from the RxNorm training and development sets.</p>
        </sec>
      </sec>
      <sec>
        <title>Transformer-Based Language Models for Misspelling Detection</title>
        <p>We fine-tuned 2 transformer-based language models, CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub> [<xref ref-type="bibr" rid="ref28">28</xref>], to the task of detecting misspelled drug names.</p>
        <p>Both CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub> models are variants of the BERT (Bidirectional Encoder Representations from Transformers) model [<xref ref-type="bibr" rid="ref29">29</xref>]. BERT is a state-of-the-art language model for natural language processing tasks such as question answering, text classification, and machine translation [<xref ref-type="bibr" rid="ref30">30</xref>]. It builds on the Transformer architecture and leverages the self-attention mechanism for efficient contextual understanding [<xref ref-type="bibr" rid="ref31">31</xref>]. BERT uses a subword vocabulary for tokenization and word/token embedding. It learns the embeddings bidirectionally, from both left to right and right to left of the input sequences of tokens. BERT was designed for transfer learning, where it was first pretrained on 2 unsupervised tasks, that is, masked language modeling (MLM) and next sentence prediction (NSP), before fine-tuning for specific downstream applications such as text classification [<xref ref-type="bibr" rid="ref29">29</xref>].</p>
        <p>Similar to BERT, both CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub> were pretrained on the MLM and NSP tasks, but using data from both general and biomedical domains [<xref ref-type="bibr" rid="ref28">28</xref>]. The general-domain corpus included 5.99 million Wikipedia documents and 1.56 million OpenWebText documents, whereas the biomedical-domain corpus included 2.09 million clinical notes from the MIMIC-III (Medical Information Mart for Intensive Care III) database [<xref ref-type="bibr" rid="ref32">32</xref>] and 2.33 million abstracts from PMC Open Access [<xref ref-type="bibr" rid="ref33">33</xref>]. Unlike BERT, CharacterBERT<sub>medical</sub> does not embed tokens (words or subwords) directly. Instead, it represents each character as a 16-dimensional real-valued vector and represents each token as a sequence of character embeddings. The character embedding sequence is passed through multiple parallel 1-dimensional convolutional neural networks (CNNs) with varying filter sizes. The outputs of these CNNs are max-pooled along the character dimension and concatenated to produce a unified representation for the sequence. The unified CNN output is then passed through 2 highway layers (which combine nonlinear transformations with residual connections) [<xref ref-type="bibr" rid="ref34">34</xref>] and projected into a 768-dimensional space, matching the size of BERT’s word piece representation. An evaluation on medical text classification tasks, including the detection of chemical-protein interactions (ChemProt [<xref ref-type="bibr" rid="ref35">35</xref>]) and drug-drug interactions (DDI [<xref ref-type="bibr" rid="ref36">36</xref>]) from free text, showed that CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub> consistently outperformed the original BERT model [<xref ref-type="bibr" rid="ref28">28</xref>].</p>
        <p>For this study, we developed CharBERTDrug and BERTDrug by fine-tuning CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub> on RxNorm-augmented data (see section Model Development and Validation Using RxNorm-Augmented Data). We included BERTDrug in this study despite the primarily noncontextual nature of the misspelling detection task for two main reasons. First, many drug names are multiword expressions (eg, <italic>abacavir-dolutegravir-lamivudine</italic> and <italic>Admelog SoloStar U-100 Insulin</italic>), and BERT-based models are better able to leverage this local lexical context than noncontextual methods. Second, BERT models use subword tokenization, which, similar to character <italic>n</italic>-gram approaches, can help identify plausible word structure even when terms contain rare or atypical character sequences.</p>
      </sec>
      <sec>
        <title>Experimental Settings</title>
        <sec>
          <title>Evaluation Metrics</title>
          <p>We evaluated model performance by accuracy, precision, recall, specificity, <italic>F</italic><sub>1</sub>-score, the area under the receiver operating characteristic curve (ROC-AUC), the area under the precision-recall curve (PR-AUC), and Brier score. <italic>F</italic><sub>1</sub>-score is the harmonic mean of precision and recall. The ROC-AUC score is calculated as the area under the ROC curve, which plots true positive rate (ie, recall) against false positive rate (ie, 1-specificity). The PR-AUC score is calculated as the area under the precision-recall curve, which plots precision against recall. Compared to ROC-AUC, <italic>F</italic><sub>1</sub>-score and PR-AUC are more sensitive to imbalanced data (eg, data with fewer positive instances than negative instances). In addition, PR-AUC provides a threshold-independent summary of performance across all classification thresholds and therefore offers a more informative assessment than <italic>F</italic><sub>1</sub>-score when class distributions vary across settings. The Brier score measures the accuracy of predicted probabilities by calculating the mean squared difference between predicted probabilities and the actual class label (0 or 1), with lower scores indicating better performance.</p>
        </sec>
        <sec>
          <title>Baseline Models</title>
          <p>SpellChecker, a dictionary-based approach implemented using the pyspellchecker library in Python, was used as the baseline model. The SpellChecker algorithm uses dynamic programming to calculate the minimum edit distance between an input word and words in its dictionary. In this study, words present in SpellChecker’s dictionary (edit distance=0) are labeled as correctly spelled; others as misspelled. SpellChecker has shown good performance in detecting misspelled medical terms in prior studies [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. To ensure a fair comparison, we extended the original dictionary of SpellChecker with the negative instances (ie, correctly spelled drug names) in the training and development splits of the RxNorm-derived dataset. Because SpellChecker does not output prediction probabilities, we did not report the Brier score for this method. As a binary classifier, SpellChecker produces only a single operating point in both the ROC and precision-recall curves. To generate multiple operating points for AUC estimation, we combined SpellChecker’s misspelling classification with term length to construct a graded decision score. Terms classified as correct by SpellChecker (ie, terms present in its dictionary) were assigned a score of 0. Among terms classified as misspelled by SpellChecker (ie, terms not present in its dictionary), longer terms were assigned higher scores, reflecting the assumption that longer terms were more likely to be misspelled. The resulting AUC scores reflect the combined measure rather than SpellChecker’s binary classification alone and should be interpreted with this limitation in mind.</p>
          <p>In addition, we implemented 2 machine learning–based baseline models: fastText<sub>ML</sub> and BioWordVec<sub>ML</sub>. Both models were based on the XGBoost classifier, with fastText embeddings [<xref ref-type="bibr" rid="ref39">39</xref>] and BioWordVec embeddings [<xref ref-type="bibr" rid="ref40">40</xref>] serving as their respective input features. FastText is a subword-based word embedding method that represents words using character <italic>n</italic>-grams and learns the embeddings of these subwords from a large Wikipedia-derived text corpus [<xref ref-type="bibr" rid="ref39">39</xref>]. BioWordVec builds on the fastText framework, but its subword embeddings were trained on biomedical text derived from PubMed titles and abstracts as well as MIMIC-III clinical notes [<xref ref-type="bibr" rid="ref40">40</xref>]. To build the fastText<sub>ML</sub> model, we used the fasttext python library to learn 200-dimensional embeddings from the original RxNorm training set. In contrast, we used the pretrained BioWordVec embeddings (200-dimensional) [<xref ref-type="bibr" rid="ref41">41</xref>] as input features for BioWordVec<sub>ML</sub>. For drug names containing multiple tokens, token-level embeddings were averaged element-wise using mean pooling to generate a single 200-dimensional feature vector as input to the XGBoost classifier.</p>
        </sec>
        <sec>
          <title>Model Development and Validation Using RxNorm-Augmented Data</title>
          <p>We created the training/development/test sets in two steps. First, we split the list of negative instances (ie, correctly spelled drug names from RxNorm) into a 3:1:1 ratio. Second, we created training sets of varying sizes by generating <italic>k</italic> positive instances (see <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) for each negative instance in the training set and oversampling the negative instances <italic>k</italic> times, where <italic>k</italic>=1,2,4,6,8,10.</p>
          <p>We trained CharBERTDrug and BERTDrug on each training set using 1 NVIDIA L40s GPU with an initial learning rate of 5e-5. The models were initialized using pretrained weights from CharacterBERT<sub>medical</sub> and BERT<sub>medical</sub>, respectively. Both models were trained using the AdamW optimizer with a linear weight decay of 0.01 over 20 epochs, with a batch size of 32. To reduce the risk of overfitting, we examined the trajectory of each model’s <italic>F</italic><sub>1</sub>-score on the development set across 0-20 epochs and selected the checkpoint with the best <italic>F</italic><sub>1</sub>-score for subsequent evaluation (see Figure S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). In addition, we evaluated model performance on the development set using training sets of varying sizes. For each model, the training set yielding the best performance was used in subsequent experiments. <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> provides further information on model training and computational costs.</p>
          <p>We trained fastText<sub>ML</sub> and BioWordVec<sub>ML</sub> using the original training set without oversampling. We evaluated multiple combinations of 4 XGBoost hyperparameters, including learning rate (0.03, 0.05, and 0.1), the number of trees (400 and 600), the maximum tree depth (4, 6, 8, and 10), and the subsampling rate (0.8 and 1.0), and selected the optimal configuration based on performance on the development set.</p>
          <p>Finally, we compared the performance of CharBERTDrug, BERTDrug, and the baseline models on the test set. All machine learning–based models used a classification threshold of 0.5. In addition, we evaluated the stability of model performance under repeated resampling of the training and test sets. Specifically, we generated 20 random subsamples from the training set by repeatedly sampling 80% of the training data without replacement. Each model was retrained on the training subsamples. We then used a 2-way nonparametric bootstrap approach to estimate uncertainty in model performance and pairwise differences between classifiers (see the Additional Information on Model Evaluation section in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>).</p>
        </sec>
        <sec>
          <title>External Validation Using LTCDC Data</title>
          <p>For the primary evaluation, we compared model performance on the cleaned LTCDC dataset and the cleaned out-of-vocabulary LTCDC dataset. For the secondary evaluation (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>), we assessed performance on (1) the uncleaned LTCDC dataset and (2) the cleaned LTCDC dataset stratified by term type (branded vs nonbranded drug names) and term frequency (high vs low). As detailed in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>, we sampled two LTCDC datasets (LTCDC dataset-1 and LTCDC dataset-2) from separate releases of the LTCDC database. High-frequency terms were defined as those prescribed &#62;1000 times in dataset 1 and prescribed to &#62;100 patients in dataset 2, while low-frequency terms were defined as those prescribed to ≤100 patients in both datasets.</p>
          <p>To maintain a strict external evaluation setting, we applied the classifiers trained on the RxNorm data directly to the three LTCDC test sets described above, using the default classification threshold of 0.5. We used a nonparametric bootstrap approach to estimate the uncertainty in model performance and pairwise differences between classifiers on each test set (see the Additional Information on Model Evaluation section in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p>
        </sec>
        <sec>
          <title>Error Analysis</title>
          <p>Using each model’s predictions on the cleaned LTCDC terms, we identified easy and difficult cases. A term was classified as easy if its label was predicted correctly by all models, and difficult if its label was predicted incorrectly by all models.</p>
        </sec>
        <sec>
          <title>Examining Factors Affecting Model Performance</title>
          <p>CharBERTDrug and BERTDrug rely on local textual information, such as character-level or subword-level spelling patterns, to detect misspellings. Term length may affect model performance because longer terms can provide additional contextual information for the models. Term type, such as generic versus branded drug names, may also affect performance because branded names may lack consistent morphological structure and contain atypical character sequences. To quantify the impact of these factors, we used multivariable logistic regression to assess the associations of term length and term type with model performance. The outcome variable was term-level prediction accuracy, coded as 1 for a correct prediction and 0 for an incorrect prediction.</p>
        </sec>
        <sec>
          <title>Comparison With an Open-Domain General-Purpose LLM</title>
          <p>In a secondary analysis, we compared our domain-specific, transformer-based language models with a popular open-domain LLM, OpenAI’s GPT-4o (2024-08-06 snapshot), in detecting misspelled drug names. LLMs encode broad lexical and semantic knowledge acquired during pretraining, which may support recognition of drug names and detection of implausible misspellings. Although this task is primarily noncontextual, GPT-4o may still benefit from subword-level patterns and broader knowledge of biomedical terminology.</p>
          <p>We evaluated and compared model performance on a random sample of the RxNorm-derived test set and the LTCDC dataset, which contained 1100 (RxNorm: 1000; LTCDC: 100) correctly spelled and 1100 (RxNorm: 1000; LTCDC: 100) misspelled drug names. Input was structured in batches of one drug name per prompt.</p>
          <p>Before formal evaluation, we conducted a lightweight model selection procedure to assess the stability and performance of GPT-4o across different configurations using 250 RxNorm terms (125 negative and 125 positive instances). We evaluated four prompt styles (Table S1 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>): (1) baseline: user prompt only; (2) baseline + system role: the baseline prompt with an added system role; (3) enhanced instructions, which provided a more detailed task description; and (4) few-shot examples, which used example-based prompting. We also examined temperature settings of 0, 0.3, and 1.0. The model with both strong stability and good overall performance (as assessed by <italic>F</italic><sub>1</sub>-score, ROC-AUC, and PR-AUC; detailed in Table S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>) was included in the formal evaluation.</p>
          <p>For GPT-4o, we converted the token-level log-probabilities using NumPy’s exponential function. The positive-class probability was calculated as exp(logprob) when the generated token was “1” and as 1 − exp(logprob) when it was “0.” We then computed the ROC-AUC, PR-AUC, and Brier scores using these probability values. Because the positive-class probability was derived from the generated token’s log-probability rather than normalized over the “0” and “1” tokens, the complement of the probability for a generated “0” included residual probability assigned to other vocabulary tokens. The resulting probability estimates and metrics derived from them should therefore be interpreted cautiously.</p>
        </sec>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Descriptive Statistics of Datasets</title>
        <p>A total of 69,824 RxNorm drug names were used for model development and internal validation, which contained 41,941 (60.1%) generic names and 27,883 (39.9%) branded names (<xref ref-type="table" rid="table1">Table 1</xref>). On average, an RxNorm drug name contains 2.4 (SD 1.5) words and 19.0 (SD 12.1) characters, with generic drug names longer than branded names (<xref ref-type="table" rid="table1">Table 1</xref>). The length of drug names is similar across the training, development, and test sets.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Characteristics of RxNorm drug names (N=69,824).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="210"/>
            <col width="0"/>
            <col width="190"/>
            <col width="0"/>
            <col width="190"/>
            <col width="0"/>
            <col width="190"/>
            <col width="0"/>
            <col width="190"/>
            <thead>
              <tr valign="top">
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">Training</td>
                <td colspan="2">Development</td>
                <td colspan="2">Test</td>
                <td>Total</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="10">
                  <bold>Total number, n (%)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>All drug names</td>
                <td colspan="2">41,894 (60.0)</td>
                <td colspan="2">13,965 (20.0)</td>
                <td colspan="2">13,965 (20.0)</td>
                <td colspan="2">69,824 (100.0)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Generic names</td>
                <td colspan="2">25,218 (36.1)</td>
                <td colspan="2">8457 (12.1)</td>
                <td colspan="2">8266 (11.8)</td>
                <td colspan="2">41,941 (60.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Branded names</td>
                <td colspan="2">16,676 (23.9)</td>
                <td colspan="2">5508 (7.9)</td>
                <td colspan="2">5699 (8.2)</td>
                <td colspan="2">27,883 (39.9)</td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Length in words, mean (SD)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>All drug names</td>
                <td colspan="2">2.4 (1.5)</td>
                <td colspan="2">2.4 (1.5)</td>
                <td colspan="2">2.3 (1.5)</td>
                <td colspan="2">2.4 (1.5)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Generic names</td>
                <td colspan="2">2.5 (1.6)</td>
                <td colspan="2">2.5 (1.6)</td>
                <td colspan="2">2.5 (1.6)</td>
                <td colspan="2">2.5 (1.6)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Branded names</td>
                <td colspan="2">2.1 (1.4)</td>
                <td colspan="2">2.2 (1.5)</td>
                <td colspan="2">2.1 (1.4)</td>
                <td colspan="2">2.1 (1.4)</td>
              </tr>
              <tr valign="top">
                <td colspan="10">
                  <bold>Length in characters, mean (SD)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>All drug names</td>
                <td colspan="2">19.0 (12.1)</td>
                <td colspan="2">19.0 (12.0)</td>
                <td colspan="2">18.9 (12.1)</td>
                <td colspan="2">19.0 (12.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Generic names</td>
                <td colspan="2">21.6 (12.8)</td>
                <td colspan="2">21.4 (12.6)</td>
                <td colspan="2">21.6 (12.7)</td>
                <td colspan="2">21.6 (12.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Branded names</td>
                <td colspan="2">15.0 (9.6)</td>
                <td colspan="2">15.4 (10.1)</td>
                <td colspan="2">14.8 (9.7)</td>
                <td colspan="2">15.0 (9.7)</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>A total of 3803 LTCDC drug names (see Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> for examples) were used for external validation of the misspelling detection models, which contained 1665 (43.8%) low-frequency names and 2138 (56.2%) high-frequency names (<xref ref-type="table" rid="table2">Table 2</xref>). Analysis of 690 misspelled LTCDC terms (see Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) showed that most misspellings could be generated by our typo-creation method. Misspellings involving the exchange of 2 nonadjacent characters and phonetic-like substitutions, which could in principle be generated by our method but likely with low frequency, together accounted for approximately 10% of errors.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Characteristics of LTCDC<sup>a</sup> drug names (N=3803).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="310"/>
            <col width="0"/>
            <col width="120"/>
            <col width="0"/>
            <col width="260"/>
            <col width="0"/>
            <col width="280"/>
            <thead>
              <tr valign="top">
                <td colspan="3">Category</td>
                <td colspan="2">n (%)</td>
                <td colspan="2">Number of words, mean (SD)</td>
                <td>Number of characters, mean (SD)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="8">
                  <bold>Low-frequency terms<sup>b</sup> (n=1665, 43.8%)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct drug name</td>
                <td colspan="2">620 (37.2)</td>
                <td colspan="2">1.3 (0.9)</td>
                <td colspan="2">14.0 (9.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct nondrug name</td>
                <td colspan="2">84 (5.0)</td>
                <td colspan="2">1.1 (0.3)</td>
                <td colspan="2">8.2 (4.0)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct short drug name</td>
                <td colspan="2">64 (3.8)</td>
                <td colspan="2">1.3 (1.0)</td>
                <td colspan="2">16.0 (9.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Misspelled drug name</td>
                <td colspan="2">799 (48.0)</td>
                <td colspan="2">1.0 (0.1)</td>
                <td colspan="2">12.7 (6.7)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Misspelled nondrug name</td>
                <td colspan="2">5 (0.3)</td>
                <td colspan="2">1.0 (0.0)</td>
                <td colspan="2">7.4 (0.8)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not sure or nonstandard</td>
                <td colspan="2">93 (5.6)</td>
                <td colspan="2">2.1 (2.3)</td>
                <td colspan="2">16.1 (13.8)</td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>High-frequency terms<sup>c</sup> (n=2138, 56.2%)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct drug name</td>
                <td colspan="2">2084 (97.5)</td>
                <td colspan="2">1.8 (1.3)</td>
                <td colspan="2">14.6 (9.4)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct nondrug name</td>
                <td colspan="2">26 (1.2)</td>
                <td colspan="2">2.3 (0.9)</td>
                <td colspan="2">14.5 (6.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Correct short drug name</td>
                <td colspan="2">19 (0.9)</td>
                <td colspan="2">2.5 (1.4)</td>
                <td colspan="2">25.2 (6.1)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Not sure or nonstandard</td>
                <td colspan="2">9 (0.4)</td>
                <td colspan="2">2.9 (1.3)</td>
                <td colspan="2">25.6 (11.3)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>LTCDC: Long-Term Care Data Cooperative.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>Low frequency was defined as prescribed for ≤100 patients.</p>
            </fn>
            <fn id="table2fn3">
              <p><sup>c</sup>High frequency was defined as prescribed &#62;1000 times in LTCDC subset 1 and for &#62;100 patients in LTCDC subset 2, as described in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Internal Validation Using RxNorm-Derived Dataset</title>
        <p>With oversampling applied to the training set and performance evaluated on the development set (see <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>: Table S1 for BERTDrug and Table S2 for CharBERTDrug), BERTDrug’s performance plateaued at an oversampling rate of 6 (<italic>F</italic><sub>1</sub>-score=0.863; ROC-AUC=0.950; PR-AUC=0.952), while CharBERTDrug plateaued at a rate of 4 (<italic>F</italic><sub>1</sub>-score=0.838; ROC-AUC=0.910; PR-AUC=0.907). Overall, BERTDrug outperformed CharBERTDrug across all evaluation metrics and oversampling rates.</p>
        <p>When evaluated on the test set (<xref ref-type="table" rid="table3">Table 3</xref>), BERTDrug<sub>6</sub> (model trained at an oversampling rate of 6) performed best (<italic>F</italic><sub>1</sub>-score=0.859; ROC-AUC=0.947; PR-AUC=0.948), followed by CharBERTDrug<sub>4</sub> (<italic>F</italic><sub>1</sub>-score=0.833; ROC-AUC=0.906; PR-AUC=0.900) and fastText<sub>ML</sub> (<italic>F</italic><sub>1</sub>-score=0.795; ROC-AUC=0.846; PR-AUC=0.837). SpellChecker achieved a high recall (0.941) but the lowest precision (0.689). fastText<sub>ML</sub> achieved the highest precision (0.814). Resampling-based evaluation showed that model performance remained stable and showed similar patterns of performance differences between models. The 2-way bootstrap analysis showed statistically significant differences (ie, with 95% CIs for model differences excluding 0) between each transformer model and each baseline model across most metrics. The only nonsignificant comparison was between BERTDrug and SpellChecker for recall (95% CI for difference –0.005 to 0.005). Evaluation stratified by term type showed that branded drug names were more challenging for all models (<xref ref-type="table" rid="table3">Table 3</xref>). Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> provides the full results for all evaluation metrics.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Model performance on RxNorm-derived test set (n=27,930), with positive instances generated using text perturbation techniques<sup>a</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="130"/>
            <col width="0"/>
            <col width="190"/>
            <col width="0"/>
            <col width="160"/>
            <col width="0"/>
            <col width="170"/>
            <col width="0"/>
            <col width="150"/>
            <col width="0"/>
            <col width="170"/>
            <thead>
              <tr valign="top">
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">CharBERTDrug</td>
                <td colspan="2">BERTDrug</td>
                <td colspan="2">SpellChecker</td>
                <td colspan="2">fastText<sub>ML</sub></td>
                <td>BioWordVec<sub>ML</sub></td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="12">
                  <bold>All drug names (positive: 13,965; negative: 13,965)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.758</td>
                <td colspan="2">0.787</td>
                <td colspan="2">0.689</td>
                <td colspan="2">
                  <italic>0.814<sup>g</sup></italic>
                </td>
                <td colspan="2">0.711</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.925</td>
                <td colspan="2">
                  <italic>0.947</italic>
                </td>
                <td colspan="2">0.941</td>
                <td colspan="2">0.776</td>
                <td colspan="2">0.768</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.833</td>
                <td colspan="2">
                  <italic>0.859</italic>
                </td>
                <td colspan="2">0.796</td>
                <td colspan="2">0.795</td>
                <td colspan="2">0.738</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC<sup>b</sup></td>
                <td colspan="2">0.906</td>
                <td colspan="2">
                  <italic>0.947</italic>
                </td>
                <td colspan="2">0.738</td>
                <td colspan="2">0.846</td>
                <td colspan="2">0.790</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC<sup>c</sup></td>
                <td colspan="2">0.900</td>
                <td colspan="2">
                  <italic>0.948</italic>
                </td>
                <td colspan="2">0.652</td>
                <td colspan="2">0.837</td>
                <td colspan="2">0.757</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Generic drug names (positive: 8266; negative: 8266)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.782</td>
                <td colspan="2">0.817</td>
                <td colspan="2">0.700</td>
                <td colspan="2">
                  <italic>0.831</italic>
                </td>
                <td colspan="2">0.781</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.924</td>
                <td colspan="2">0.947</td>
                <td colspan="2">
                  <italic>0.949</italic>
                </td>
                <td colspan="2">0.775</td>
                <td colspan="2">0.768</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.847</td>
                <td colspan="2">
                  <italic>0.877</italic>
                </td>
                <td colspan="2">0.806</td>
                <td colspan="2">0.802</td>
                <td colspan="2">0.775</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.921</td>
                <td colspan="2">
                  <italic>0.958</italic>
                </td>
                <td colspan="2">0.738</td>
                <td colspan="2">0.854</td>
                <td colspan="2">0.845</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.918</td>
                <td colspan="2">
                  <italic>0.961</italic>
                </td>
                <td colspan="2">0.647</td>
                <td colspan="2">0.850</td>
                <td colspan="2">0.823</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Branded drug names (positive: 5699; negative: 5699)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.730</td>
                <td colspan="2">0.751</td>
                <td colspan="2">0.676</td>
                <td colspan="2">
                  <italic>0.792</italic>
                </td>
                <td colspan="2">0.636</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.927</td>
                <td colspan="2">
                  <italic>0.947</italic>
                </td>
                <td colspan="2">0.931</td>
                <td colspan="2">0.777</td>
                <td colspan="2">0.768</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.817</td>
                <td colspan="2">
                  <italic>0.837</italic>
                </td>
                <td colspan="2">0.783</td>
                <td colspan="2">0.785</td>
                <td colspan="2">0.696</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.885</td>
                <td colspan="2">
                  <italic>0.929</italic>
                </td>
                <td colspan="2">0.732</td>
                <td colspan="2">0.837</td>
                <td colspan="2">0.706</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.874</td>
                <td colspan="2">
                  <italic>0.929</italic>
                </td>
                <td colspan="2">0.659</td>
                <td colspan="2">0.822</td>
                <td colspan="2">0.659</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Stability evaluation, mean (95% CI)<sup>d,e,f</sup></bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.748 (0.742-0.754)</td>
                <td colspan="2">
                  <italic>0.778 (0.772-0.784)</italic>
                </td>
                <td colspan="2">0.679 (0.672-0.685)</td>
                <td colspan="2">0.770 (0.762-0.777)</td>
                <td colspan="2">0.664 (0.657-0.671)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.922 (0.918-0.926)</td>
                <td colspan="2">
                  <italic>0.942 (0.939-0.946)</italic>
                </td>
                <td colspan="2">
                  <italic>0.942 (0.939-0.946)</italic>
                </td>
                <td colspan="2">0.791 (0.783-0.798)</td>
                <td colspan="2">0.862 (0.855-0.869)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.826 (0.822-0.830)</td>
                <td colspan="2">
                  <italic>0.852 (0.848-0.856)</italic>
                </td>
                <td colspan="2">0.789 (0.785-0.794)</td>
                <td colspan="2">0.780 (0.775-0.784)</td>
                <td colspan="2">0.750 (0.745-0.755)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.898 (0.894-0.901)</td>
                <td colspan="2">
                  <italic>0.941 (0.938-0.943)</italic>
                </td>
                <td colspan="2">0.728 (0.722-0.734)</td>
                <td colspan="2">0.834 (0.830-0.839)</td>
                <td colspan="2">0.787 (0.782-0.792)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.892 (0.887-0.896)</td>
                <td colspan="2">
                  <italic>0.942 (0.939-0.945)</italic>
                </td>
                <td colspan="2">0.643 (0.635-0.651)</td>
                <td colspan="2">0.825 (0.818-0.831)</td>
                <td colspan="2">0.754 (0.748-0.761)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>Positive instances were misspelled RxNorm terms generated using text perturbation techniques (eg, character insertion and deletion, swapping of adjacent characters, and substitution with keyboard-adjacent characters). The best performance score for each metric across the models was indicated in italics. For the stability evaluation, the best-performing model was determined based on the point estimates.</p>
            </fn>
            <fn id="table3fn2">
              <p><sup>b</sup>ROC-AUC: area under the receiver operating characteristic curve.</p>
            </fn>
            <fn id="table3fn3">
              <p><sup>c</sup>PR-AUC: area under the precision-recall curve.</p>
            </fn>
            <fn id="table3fn4">
              <p><sup>d</sup>Model performance was evaluated using repeated resampling, with 80% of the training and test sets sampled to generate 20 random subsamples from each set.</p>
            </fn>
            <fn id="table3fn5">
              <p><sup>e</sup>Two-way bootstrap analyses of model differences were performed for all metrics, comparing each transformer-based model with each baseline model. Most comparisons showed statistically significant differences, with 95% CIs for model differences excluding 0. Nonsignificant comparisons included BERTDrug vs SpellChecker for recall (95% CI for difference –0.005 to 0.005). <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref> provides the 95% CIs for pairwise model differences across all comparisons.</p>
            </fn>
            <fn id="table3fn6">
              <p><sup>f</sup>Two-way bootstrap analyses compared the two transformer-based models across all metrics. All comparisons showed statistically significant differences, with 95% CIs for model differences excluding 0. <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref> provides the 95% CIs for pairwise model differences across all comparisons.</p>
            </fn>
            <fn id="table3fn7">
              <p><sup>g</sup>The best performance score for each metric across the models is indicated in italics. For the stability evaluation, the best-performing model was determined based on the point estimates.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>External Validation Using the LTCDC Dataset</title>
        <p>On the cleaned out-of-vocabulary LTCDC dataset (first block in <xref ref-type="table" rid="table4">Table 4</xref> and <xref rid="figure2" ref-type="fig">Figure 2</xref>), CharBERTDrug achieved the best <italic>F</italic><sub>1</sub>-score (0.696) and ROC-AUC (0.788), while BERTDrug achieved the best PR-AUC (0.665). SpellChecker achieved the highest recall (0.996). Full results for all evaluation metrics are provided in Table S1 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. Bootstrap analyses of model differences showed statistically significant differences between the best-performing transformer-based model and each baseline model across most metrics, with 95% CIs of model differences excluding 0 (<xref ref-type="table" rid="table4">Table 4</xref>, note c). In contrast, differences between the two transformer-based models were not statistically significant for most metrics (<xref ref-type="table" rid="table4">Table 4</xref>, note d). Detailed model comparison results are provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. The ROC curves also demonstrated clear performance differences among the models (<xref rid="figure3" ref-type="fig">Figure 3</xref>). Evaluation on the cleaned full LTCDC dataset (fourth block in <xref ref-type="table" rid="table4">Table 4</xref>) and the uncleaned LTCDC dataset (Table S2 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>) showed similar patterns in model performance differences.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Model performance on LTCDC<sup>a</sup> datasets, with positive instances defined as misspelled drug names<sup>b,c,d</sup>.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="120"/>
            <col width="0"/>
            <col width="190"/>
            <col width="0"/>
            <col width="160"/>
            <col width="0"/>
            <col width="170"/>
            <col width="0"/>
            <col width="160"/>
            <col width="0"/>
            <col width="170"/>
            <thead>
              <tr valign="top">
                <td colspan="3">
                  <break/>
                </td>
                <td colspan="2">CharBERTDrug</td>
                <td colspan="2">BERTDrug</td>
                <td colspan="2">SpellChecker</td>
                <td colspan="2">fastText<sub>ML</sub></td>
                <td>BioWordVec<sub>ML</sub></td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned out-of-vocabulary dataset</bold>
                  <bold>(positive: 765; negative: 1157)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.630 (0.609-0.652)</td>
                <td colspan="2">
                  <italic>0.631 (0.606-0.655)<sup>g</sup></italic>
                </td>
                <td colspan="2">0.486 (0.477-0.496)</td>
                <td colspan="2">0.381 (0.360-0.403)</td>
                <td colspan="2">0.533 (0.513-0.555)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.778 (0.749-0.809)</td>
                <td colspan="2">0.712 (0.680-0.744)</td>
                <td colspan="2">
                  <italic>0.996 (0.991-1.000)</italic>
                </td>
                <td colspan="2">0.475 (0.439-0.508)</td>
                <td colspan="2">0.708 (0.676-0.740)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">
                  <italic>0.696 (0.676-0.718)</italic>
                </td>
                <td colspan="2">0.669 (0.645-0.692)</td>
                <td colspan="2">0.654 (0.645-0.663)</td>
                <td colspan="2">0.423 (0.397-0.448)</td>
                <td colspan="2">0.609 (0.587-0.630)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC<sup>e</sup></td>
                <td colspan="2">
                  <italic>0.788 (0.767-0.808)</italic>
                </td>
                <td colspan="2">0.786 (0.765-0.806)</td>
                <td colspan="2">0.646 (0.622-0.670)</td>
                <td colspan="2">0.465 (0.440-0.492)</td>
                <td colspan="2">0.655 (0.631-0.681)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC<sup>f</sup></td>
                <td colspan="2">0.639 (0.608-0.672)</td>
                <td colspan="2">
                  <italic>0.665 (0.634-0.699)</italic>
                </td>
                <td colspan="2">0.457 (0.439-0.480)</td>
                <td colspan="2">0.377 (0.360-0.399)</td>
                <td colspan="2">0.544 (0.523-0.567)</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned out-of-vocabulary dataset, generic drug names</bold>
                  <bold>(positive: 461; negative: 489)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">
                  <italic>0.824 (0.794-0.855)</italic>
                </td>
                <td colspan="2">0.790 (0.756-0.824)</td>
                <td colspan="2">0.603 (0.588-0.620)</td>
                <td colspan="2">0.479 (0.444-0.512)</td>
                <td colspan="2">0.649 (0.619-0.680)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.783 (0.744-0.820)</td>
                <td colspan="2">0.709 (0.668-0.751)</td>
                <td colspan="2">
                  <italic>1.000 (1.000-1.000)</italic>
                </td>
                <td colspan="2">0.464 (0.416-0.510)</td>
                <td colspan="2">0.683 (0.642-0.727)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">
                  <italic>0.803 (0.775-0.830)</italic>
                </td>
                <td colspan="2">0.747 (0.715-0.779)</td>
                <td colspan="2">0.753 (0.741-0.766)</td>
                <td colspan="2">0.471 (0.432-0.506)</td>
                <td colspan="2">0.666 (0.635-0.697)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">
                  <italic>0.875 (0.852-0.899)</italic>
                </td>
                <td colspan="2">0.850 (0.826-0.874)</td>
                <td colspan="2">0.621 (0.583-0.659)</td>
                <td colspan="2">0.475 (0.437-0.511)</td>
                <td colspan="2">0.668 (0.634-0.703)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.826 (0.788-0.864)</td>
                <td colspan="2">
                  <italic>0.828 (0.795-0.860)</italic>
                </td>
                <td colspan="2">0.533 (0.507-0.567)</td>
                <td colspan="2">0.480 (0.453-0.514)</td>
                <td colspan="2">0.651 (0.621-0.682)</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned out-of-vocabulary dataset, branded drug names</bold>
                  <bold>(positive: 304; negative: 668)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.462 (0.435-0.490)</td>
                <td colspan="2">
                  <italic>0.484 (0.454-0.516)</italic>
                </td>
                <td colspan="2">0.375 (0.365-0.386)</td>
                <td colspan="2">0.294 (0.267-0.323)</td>
                <td colspan="2">0.427 (0.403-0.454)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.770 (0.724-0.816)</td>
                <td colspan="2">0.717 (0.668-0.763)</td>
                <td colspan="2">
                  <italic>0.990 (0.977-1.000)</italic>
                </td>
                <td colspan="2">0.490 (0.434-0.546)</td>
                <td colspan="2">0.747 (0.701-0.796)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.577 (0.548-0.607)</td>
                <td colspan="2">
                  <italic>0.578 (0.544-0.612)</italic>
                </td>
                <td colspan="2">0.544 (0.533-0.556)</td>
                <td colspan="2">0.368 (0.333-0.403)</td>
                <td colspan="2">0.544 (0.514-0.574)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.715 (0.683-0.745)</td>
                <td colspan="2">
                  <italic>0.733 (0.702-0.764)</italic>
                </td>
                <td colspan="2">0.585 (0.550-0.618)</td>
                <td colspan="2">0.456 (0.418-0.496)</td>
                <td colspan="2">0.664 (0.631-0.698)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.459 (0.422-0.505)</td>
                <td colspan="2">
                  <italic>0.487 (0.448-0.533)</italic>
                </td>
                <td colspan="2">0.334 (0.318-0.356)</td>
                <td colspan="2">0.285 (0.267-0.309)</td>
                <td colspan="2">0.447 (0.420-0.476)</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned full dataset</bold>
                  <bold>(positive: 799; negative: 2787)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">
                  <italic>0.586 (0.564-0.610)</italic>
                </td>
                <td colspan="2">
                  <italic>0.586 (0.561-0.612)</italic>
                </td>
                <td colspan="2">0.495 (0.482-0.511)</td>
                <td colspan="2">0.256 (0.240-0.272)</td>
                <td colspan="2">0.438 (0.419-0.457)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.780 (0.751-0.809)</td>
                <td colspan="2">0.718 (0.686-0.748)</td>
                <td colspan="2">
                  <italic>0.995 (0.990-0.999)</italic>
                </td>
                <td colspan="2">0.469 (0.434-0.503)</td>
                <td colspan="2">0.700 (0.668-0.731)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">
                  <italic>0.669 (0.648-0.691)</italic>
                </td>
                <td colspan="2">0.645 (0.621-0.669)</td>
                <td colspan="2">0.661 (0.649-0.675)</td>
                <td colspan="2">0.331 (0.309-0.352)</td>
                <td colspan="2">0.539 (0.518-0.559)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.868 (0.853-0.881)</td>
                <td colspan="2">
                  <italic>0.869 (0.856-0.883)</italic>
                </td>
                <td colspan="2">0.850 (0.838-0.862)</td>
                <td colspan="2">0.548 (0.526-0.571)</td>
                <td colspan="2">0.748 (0.729-0.767)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.596 (0.564-0.632)</td>
                <td colspan="2">
                  <italic>0.630 (0.598-0.664)</italic>
                </td>
                <td colspan="2">0.464 (0.444-0.489)</td>
                <td colspan="2">0.250 (0.236-0.266)</td>
                <td colspan="2">0.451 (0.430-0.474)</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned full dataset, generic drug names</bold>
                  <bold>(positive: 485; negative: 1197)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">
                  <italic>0.817 (0.786-0.848)</italic>
                </td>
                <td colspan="2">0.783 (0.748-0.816)</td>
                <td colspan="2">0.613 (0.590-0.636)</td>
                <td colspan="2">0.341 (0.314-0.368)</td>
                <td colspan="2">0.590 (0.556-0.622)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.784 (0.746-0.819)</td>
                <td colspan="2">0.713 (0.674-0.755)</td>
                <td colspan="2">
                  <italic>0.998 (0.994-1.000)</italic>
                </td>
                <td colspan="2">0.456 (0.410-0.499)</td>
                <td colspan="2">0.672 (0.631-0.713)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">
                  <italic>0.800 (0.772-0.825)</italic>
                </td>
                <td colspan="2">0.746 (0.716-0.776)</td>
                <td colspan="2">0.759 (0.742-0.777)</td>
                <td colspan="2">0.390 (0.357-0.422)</td>
                <td colspan="2">0.628 (0.596-0.660)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">
                  <italic>0.918 (0.901-0.932)</italic>
                </td>
                <td colspan="2">0.909 (0.893-0.925)</td>
                <td colspan="2">0.842 (0.823-0.861)</td>
                <td colspan="2">0.565 (0.535-0.594)</td>
                <td colspan="2">0.777 (0.752-0.802)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.806 (0.768-0.843)</td>
                <td colspan="2">
                  <italic>0.809 (0.776-0.843)</italic>
                </td>
                <td colspan="2">0.539 (0.512-0.577)</td>
                <td colspan="2">0.338 (0.314-0.367)</td>
                <td colspan="2">0.600 (0.567-0.635)</td>
              </tr>
              <tr valign="top">
                <td colspan="12">
                  <bold>Cleaned full dataset, branded drug names</bold>
                  <bold>(positive: 314; negative: 1590)</bold>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Precision</td>
                <td colspan="2">0.406 (0.380-0.433)</td>
                <td colspan="2">
                  <italic>0.424 (0.395-0.455)</italic>
                </td>
                <td colspan="2">0.382 (0.365-0.400)</td>
                <td colspan="2">0.188 (0.169-0.207)</td>
                <td colspan="2">0.322 (0.300-0.344)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Recall</td>
                <td colspan="2">0.774 (0.726-0.822)</td>
                <td colspan="2">0.726 (0.675-0.774)</td>
                <td colspan="2">
                  <italic>0.990 (0.978-1.000)</italic>
                </td>
                <td colspan="2">0.490 (0.433-0.545)</td>
                <td colspan="2">0.742 (0.691-0.790)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">0.532 (0.503-0.562)</td>
                <td colspan="2">0.535 (0.501-0.568)</td>
                <td colspan="2">
                  <italic>0.551 (0.534-0.570)</italic>
                </td>
                <td colspan="2">0.272 (0.243-0.300)</td>
                <td colspan="2">0.449 (0.420-0.477)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ROC-AUC</td>
                <td colspan="2">0.825 (0.802-0.847)</td>
                <td colspan="2">
                  <italic>0.839 (0.816-0.859)</italic>
                </td>
                <td colspan="2">0.822 (0.805-0.840)</td>
                <td colspan="2">0.534 (0.497-0.569)</td>
                <td colspan="2">0.749 (0.718-0.778)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>PR-AUC</td>
                <td colspan="2">0.407 (0.370-0.454)</td>
                <td colspan="2">
                  <italic>0.445 (0.404-0.495)</italic>
                </td>
                <td colspan="2">0.339 (0.320-0.366)</td>
                <td colspan="2">0.178 (0.164-0.199)</td>
                <td colspan="2">0.340 (0.313-0.369)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table4fn1">
              <p><sup>a</sup>LTCDC: Long-Term Care Data Cooperative.</p>
            </fn>
            <fn id="table4fn2">
              <p><sup>b</sup>Positive instances: terms labeled as “misspelled drug name”; negative instances: terms labeled as “correct drug name” and “correct short drug name.”</p>
            </fn>
            <fn id="table4fn3">
              <p><sup>c</sup>Bootstrap analyses of model differences were performed for all metrics comparing each transformer-based model with each baseline model. Across most pairwise comparisons, the best-performing transformer-based model for each metric differed significantly from the baseline models, with 95% CIs for model differences excluding 0. Nonsignificant differences between the best-performing transformer-based model and the best-performing baseline models included BERTDrug vs BioWordVec<sub>ML</sub> for PR-AUC on cleaned, branded out-of-vocabulary (OOV) terms; BERTDrug vs SpellChecker for <italic>F</italic><sub>1</sub>-score and ROC-AUC on cleaned branded terms; and CharBERTDrug vs SpellChecker for <italic>F</italic><sub>1</sub>-score on cleaned terms. The best performance score for each metric across the models was indicated in italics. The best-performing model was determined based on the point estimates. <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref> provides the 95% CIs for pairwise model differences across all comparisons.</p>
            </fn>
            <fn id="table4fn4">
              <p><sup>d</sup>Bootstrap analyses compared the two transformer-based models across all metrics. Most comparisons did not show statistically significant differences, with 95% CIs for model differences including 0. Significant differences between CharBERTDrug vs BERTDrug included recall and <italic>F</italic><sub>1</sub>-score on cleaned OOV terms; precision, recall, <italic>F</italic><sub>1</sub>-score, and ROC-AUC on cleaned, generic OOV terms; recall on cleaned terms; and precision, recall, and <italic>F</italic><sub>1</sub>-score on cleaned, generic terms. <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref> provides the 95% CIs for pairwise model differences across all comparisons.</p>
            </fn>
            <fn id="table4fn5">
              <p><sup>e</sup>ROC-AUC: area under the receiver operating characteristic curve.</p>
            </fn>
            <fn id="table4fn6">
              <p><sup>f</sup>PR-AUC: area under the precision-recall curve.</p>
            </fn>
            <fn id="table4fn7">
              <p><sup>g</sup>The best performance score for each metric across the models was indicated in italics. The best-performing model was determined based on the point estimates.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Model performance on the cleaned out-of-vocabulary LTCDC test set. LTCDC: Long-Term Care Data Cooperative; PR-AUC: area under the precision-recall curve; ROC-AUC: area under the receiver operating characteristic curve.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e91151_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Models’ ROC curves on the cleaned out-of-vocabulary LTCDC test set. For each model, ROC curves were estimated using bootstrap resampling with 2000 replicates. In each replicate, true positive rate (TPR) values were linearly interpolated on a fixed false positive rate (FPR) grid (0-1, step=0.01). At each grid point, TPRs were aggregated across bootstrap replicates and summarized as mean and 95% CI, with the solid curve representing the mean and the shaded area indicating the 95% CI. AUC: area under the curve; LTCDC: Long-Term Care Data Cooperative; ROC: receiver operating characteristic curve.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e91151_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Evaluation stratified by term type showed that branded drug names were more challenging for all models (<xref ref-type="table" rid="table4">Table 4</xref> and Table S1 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). CharBERTDrug performed best on most metrics for generic drug names in both the cleaned out-of-vocabulary set and the cleaned full set (second and fifth blocks in <xref ref-type="table" rid="table4">Table 4</xref>), whereas BERTDrug performed best on most metrics for branded drug names in both the cleaned out-of-vocabulary set and the cleaned full set (third and sixth blocks in <xref ref-type="table" rid="table4">Table 4</xref>). Bootstrap analyses of model differences showed that, for branded drug names, the two transformer-based models did not differ significantly on most metrics (<xref ref-type="table" rid="table4">Table 4</xref>, note d).</p>
        <p>Frequency-stratified evaluation (Tables S3 and S4 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>) showed that CharBERTDrug performed best on high-frequency terms, while BERTDrug performed best on low-frequency terms across most metrics. The high-frequency subsets were extremely imbalanced, with no positive instances in the cleaned LTCDC dataset and only 9 (0.4% of 2138) positive instances in the uncleaned LTCDC dataset. In the uncleaned dataset, metrics sensitive to minority-class performance, including precision, <italic>F</italic><sub>1</sub>-score, and PR-AUC, were poor across all models (eg, <italic>F</italic><sub>1</sub>-score≤0.05), whereas accuracy and ROC-AUC were comparable to those observed for low-frequency terms (Table S4 in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>).</p>
      </sec>
      <sec>
        <title>Error Analysis</title>
        <p>A total of 3586 cleaned LTCDC terms were analyzed. Among them, 1088 (30.3%) terms were classified as easy (ie, consistently predicted correctly by all models), while 100 (2.8%) terms were difficult (ie, consistently predicted incorrectly by all models). Among the 100 difficult terms, only one term (1%), <italic>Cyanocobalamine</italic> instead of <italic>Cyanocobalamin</italic>, was a misspelled drug name that was incorrectly flagged as correct by the models. The remaining 99 (99%) difficult terms were correctly spelled but incorrectly flagged as misspellings by the models; among them, 73 (73.7%) were branded drug names, including <italic>Hempvana</italic> (topical product for pain relief), <italic>Geri-kot</italic> (senna), <italic>Veozah</italic> (Fezolinetant), <italic>Zenoptiq</italic> (ocular nutritional supplement), and more than 20 branded vaccine names (eg, <italic>Flublok Quad 2019-2020 (PF) (flu vac qv 2019(18yr up)rc(pf))</italic> and <italic>Moderna COVID-19 Bival Booster)</italic>.</p>
        <p>Among the 128 terms that were predicted correctly by SpellChecker but incorrectly by both BERTDrug and CharBERTDrug, 72 (56.3%) were misspelled. In contrast, all 382 terms that were predicted incorrectly by SpellChecker but correctly by both BERTDrug and CharBERTDrug were correctly spelled medication names.</p>
      </sec>
      <sec>
        <title>Effects of Term Type and Term Length on Model Performance</title>
        <p>As shown in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>, in the RxNorm test set, after adjusting for term length, BERTDrug was less likely to make correct predictions for branded drug names than for generic names (odds ratio [OR] 0.883, 95% CI 0.825-0.945); while CharBERTDrug’s prediction accuracy did not differ significantly by term type (OR 0.989, 95% CI 0.928-1.054). In the LTCDC out-of-vocabulary test set, after adjusting for term length, both BERTDrug and CharBERTDrug were less likely to make correct predictions for branded drug names than for generic names (BERTDrug: OR 0.528, 95% CI 0.423-0.660; CharBERTDrug: OR 0.353, 95% CI 0.280-0.444).</p>
        <p>Term length was positively associated with prediction accuracy in the RxNorm test set but negatively associated with prediction accuracy in the LTCDC test set (Table S1 in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>).</p>
      </sec>
      <sec>
        <title>Comparison With an Open-Domain General-Purpose LLM</title>
        <p>The GPT-4o model (baseline prompt + system role, temperature=0.0) showed strong stability and good overall performance (Table S2 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>) and was therefore included in the formal evaluation. Both specialized transformer-based models (ie, BERTDrug and CharBERTDrug) outperformed GPT-4o on RxNorm terms across most metrics, including <italic>F</italic><sub>1</sub>-score (BERTDrug: 0.855, CharBERTDrug: 0.831, GPT-4o: 0.721), ROC-AUC (BERTDrug: 0.951, CharBERTDrug: 0.911, GPT-4o: 0.856), and PR-AUC (BERTDrug: 0.953, CharBERTDrug: 0.908, GPT-4o: 0.838). However, this performance advantage diminished substantially for the LTCDC terms: the transformer-based models outperformed GPT-4o only in precision and specificity, while GPT-4o performed better on all other metrics. Table S3 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> provides the full results.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>This study presents a deep-learning approach that leverages specialized, transformer-based language models to detect typographical errors in out-of-vocabulary medication names. Internal validation on the RxNorm-derived test set showed strong performance of both BERTDrug and CharBERTDrug models, with ROC-AUC scores of 0.947 and 0.906, respectively. External validation on the LTCDC terms demonstrated adequate generalizability of both models, with ROC-AUC scores of 0.786 for BERTDrug and 0.788 for CharBERTDrug on the cleaned out-of-vocabulary set and 0.869 and 0.868, respectively, on the cleaned full set.</p>
      </sec>
      <sec>
        <title>Related Work</title>
        <p>Detecting misspellings from medical text has been an active area of research. Most prior studies used dictionary-based methods to detect misspelled medical terms [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Two recent studies have applied deep learning techniques to correct misspellings in non-English clinical texts; however, neither addressed out-of-vocabulary terms nor independently evaluated their error detection components [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. To our knowledge, this study is the first to develop and evaluate deep learning models—particularly transformer-based language models—that detect misspellings from out-of-vocabulary medication names. Our approach addresses a key limitation of dictionary-based methods, which classify all out-of-vocabulary terms as misspellings and therefore suffer from excessive false positives (ie, low precision) and low specificity. For example, compared to BERTDrug, the dictionary-based SpellChecker had a significantly lower precision (0.486 vs 0.631, 95% CI for difference<sub>BERTDrug_vs_SpellChecker</sub>: 0.121 to 0.168) and specificity (0.304 vs 0.724, 95% CI for difference<sub>BERTDrug_vs_SpellChecker</sub>: 0.389 to 0.451) in detecting errors in out-of-vocabulary LTCDC medication names (refer to the highlighted rows in worksheet “Table4_Appendix4-TableS1” in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>). This means that for every 100 correctly spelled new drug names (ie, out-of-vocabulary terms), SpellChecker generates 42 more false alarms for misspellings than BERTDrug (100 * (0.724 – 0.304)). Expanding the dictionary used by SpellChecker can improve precision and specificity, but will decrease dictionary search efficiency. More importantly, it requires ongoing monitoring of newly developed drugs and continual updates to the dictionary. In contrast, the transformer-based models (BERTDrug and CharBERTDrug) provided a superior balance across precision, specificity, and recall, along with improved overall performance as measured by <italic>F</italic><sub>1</sub>-score and ROC-AUC. This finding highlights the advantage of using transformer-based language models for detecting misspellings in out-of-vocabulary terms. It is worth noting that even in clinical care settings—where high sensitivity in medication error detection is prioritized—a balance across precision, specificity, and recall is still crucial, particularly for hospitals that intend to deploy fully automated pipelines for error detection and correction. Excessive false alarms during error detection not only increase alert burden but may also prompt unnecessary downstream corrections, inadvertently creating new medication name errors.</p>
        <p>Compared to the medical domain, progress in misspelling detection and correction has advanced more rapidly in the general domain. Recent research explored end-to-end approaches that leverage deep learning-based sequential models to jointly detect and correct misspellings [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref46">46</xref>]. Commonly used techniques included recurrent neural networks (RNNs) such as long short-term memory (LSTM) and gated recurrent unit (GRU) models [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], as well as transformer-based models like BERT [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. Our approach is closely related to studies that used BERT-based models for error detection [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>], but it differs in model implementation. Jayanthi and colleagues [<xref ref-type="bibr" rid="ref44">44</xref>] applied a BERT-based model to detect and correct misspelled words. For each word in the input sequence, their model outputs a probability distribution over a predefined vocabulary. As such, their model could not handle misspellings in out-of-vocabulary words. Klemen and colleagues [<xref ref-type="bibr" rid="ref47">47</xref>] fine-tuned the Slovene BERT model—SloBERTa [<xref ref-type="bibr" rid="ref48">48</xref>]—for detecting misspelled words in an input sentence. They introduced a special token after each word (or after the last subword of a word) to facilitate word-level error detection. In contrast, our BERT-based models took a medication name (which was a single-word or multiple-word expression) as the input and classified it as misspelled or correctly spelled. A few prior studies have used character-based CNNs to learn local character patterns and have integrated them with RNNs to detect misspellings [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. Similarly, our CharBERTDrug model combined a character-level CNN with BERT.</p>
      </sec>
      <sec>
        <title>Model Performance and Trade-Offs</title>
        <p>On the cleaned full LTCDC set, CharBERTDrug achieved higher recall than BERTDrug (0.780 vs 0.718), while showing slightly lower specificity (0.842 vs 0.854). The difference likely arose from their tokenization focus: CharBERTDrug leverages character-level information, which enhances its sensitivity to subtle misspellings and boosts recall. There is a trade-off between maximizing error detection and minimizing false positives; the choice of model should be guided by the specific needs of the applications. For example, high recall in detecting misspelled medication names is preferred in applications that support pharmacovigilance and safety monitoring, whereas high precision and specificity are more desirable in real-time alerts embedded within clinical workflows, where avoiding alert fatigue is critical—particularly when late-stage quality assurance processes are available to catch additional misspellings.</p>
        <p>Both BERTDrug and CharBERTDrug demonstrated significantly higher precision and specificity than the 3 baseline models. As shown in Figure S1 in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>, both models had a substantially lower projected review burden per 1000 medication entries than the baseline models across plausible misspelling prevalences. As expected, SpellChecker had the lowest projected number of missed misspellings (Figure S2 in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>), but it also incurred a substantially higher review burden than BERTDrug and CharBERTDrug (Figure S1 in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>).</p>
        <p>Error analysis revealed that the majority of difficult cases for all models were branded drug names that were correctly spelled but misclassified as misspellings. Association analyses also showed that the transformer-based models were less likely to make correct predictions on branded names (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>). A possible reason is that brand names exhibit less regular spelling patterns and are therefore more difficult for the models to learn and recognize. Specifically, unlike generic drug names, brand names lack consistent morphological structure including shared prefixes or suffixes (eg, -statin, -caine, and -olol), which limits the model’s ability to generalize via subword patterns or character-level <italic>n</italic>-grams and forces the model to rely more heavily on memorization of individual lexical items. In addition, brand names often violate common orthographic and phonotactic patterns of natural language and contain atypical character sequences (eg, Vraylar, Xeljanz, and Qulipta) that are more likely to be interpreted by the model as noise or misspellings.</p>
        <p>In addition to term type, term length was also associated with prediction accuracy for the two transformer-based models. However, the direction of this association differed across datasets: longer RxNorm terms were more likely to be classified correctly, whereas longer LTCDC terms were more difficult for the models to classify correctly (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>). One possible explanation is that, although longer terms may provide additional contextual information that can benefit transformer-based models, those in the external evaluation set often contained previously unseen spelling patterns arising from combinations of formulation, strength, dosage form, and abbreviations not represented in the training data.</p>
        <p>We also noted that, because only 9 positive instances were identified among the 2138 uncleaned high-frequency terms and all 9 positive instances were nonstandard spellings, all models performed poorly on metrics sensitive to minority-class performance (eg, precision, <italic>F</italic><sub>1</sub>-score, and PR-AUC) on this subset. The observed pattern of low precision and moderate recall among the transformer-based models suggests that fine-tuning on nonstandard misspellings is needed to improve detection of this type of misspelling. For fully automated classification, future work could also explore hybrid strategies, such as fine-tuning separate models for high-frequency terms, when a sufficient number of positive cases are available for training, and for low-frequency terms.</p>
        <p>Generative LLMs have demonstrated strong general-purpose capabilities, including interpreting free-text inputs and generating clinically relevant insights from vast medical knowledge [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>]. Unlike specialized language models fine-tuned for classification tasks (eg, the models developed in this study), generative LLMs are not inherently designed as probabilistic classifiers. However, they have demonstrated strong performance on complex clinical classification tasks such as disease diagnosis [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. In our secondary analysis, we found that our models outperformed the generative LLM GPT-4o in precision and speed when detecting misspelled medication names, but with lower overall classification performance than GPT-4o on LTCDC terms. These findings suggest that domain shift can substantially reduce the performance advantages of domain-specific training. Future work should improve model generalizability by incorporating more diverse training data representing the terminology, formatting conventions, and spelling patterns encountered across real-world clinical settings. Note that our analysis is preliminary, and the performance of generative LLMs could potentially be improved with more advanced models and refined prompting strategies. Future studies that systematically compare general-purpose generative LLMs and specialized language models for misspelling detection are warranted to gain deeper insights into their relative strengths and limitations.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>This study has several limitations. First, we trained the misspelling detection models using an augmented dataset derived from RxNorm drug names, in which misspellings were introduced through commonly used text perturbation techniques. Although this approach provided a large volume of training data and improved the generalizability of the model, the generated misspellings may not fully reflect the error patterns observed in real-world databases such as LTCDC. This discrepancy likely contributed to the reduced model performance on the external validation set. Future work that incorporates misspelled medication names from real-world datasets into model training may further enhance performance. Second, the generalizability of our findings beyond the LTCDC remains uncertain. Future studies should assess performance in other contexts, such as hospital and retail pharmacy settings, where error patterns may differ. Third, our error analysis revealed that drug brand names were the most challenging cases across all models. Moreover, all models performed poorly in detecting rare misspellings (0.4%) among high-frequency drug names. Because all misspelled high-frequency terms were nonstandard spellings, this low performance is likely attributable to two factors: (1) severe class imbalance and (2) the difficulty of detecting nonstandard spellings. Specialized handling strategies or additional training data may be needed to mitigate these limitations. Fourth, the definition of high-frequency medication terms in the LTCDC external validation set was not fully consistent. LTCDC terms were sampled from two different database releases using different approaches, resulting in two separate criteria for high-frequency terms (&#62;1000 occurrences vs &#62;100 patients). Because earlier database releases were no longer available after LTCDC’s platform transition in late 2024, we could not repeat the original sampling process to harmonize these definitions. This inconsistency may have introduced some heterogeneity into the external evaluation. Fifth, our data augmentation strategy relied primarily on structural text perturbations (eg, insertion, deletion, and substitution). While these transformations capture many typographical errors, they do not explicitly model phonetic misspellings or improper word splitting, which are also common in real-world medication name errors. This may have limited the range of misspelling patterns represented during training and contributed to the reduced performance on the external validation set. Future work should incorporate phonetic and segmentation-based augmentation strategies, such as dropping silent-like letters, substituting similar-sounding letter groups (eg, ph versus f, c versus k), and inserting or removing spaces or hyphens in long medication names, to generate more realistic misspellings in the training data. Sixth, LTCDC terms frequently combine drug formulation, strength, dosage form, and abbreviations, making them structurally different from RxNorm drug names. This compositional shift may confound the observed association between term length and model performance in this dataset. Seventh, many medication names in the RxNorm and LTCDC datasets, or their component words, are also available through publicly accessible online resources and may have been included in the pretraining corpus of GPT-4o. Therefore, benchmarking GPT-4o on terms sampled from these datasets may introduce potential data contamination or data leakage, which could lead to an overestimation of model performance. Finally, we did not examine misspelling detection in longer clinical contexts because appropriate annotated datasets for model training and evaluation were not available. Extending this work to detect misspelled drug names within richer clinical context would be an important and practically relevant direction for future research.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>Our findings demonstrate that specialized language models can improve the accuracy of automated drug-name misspelling detection, particularly for out-of-vocabulary terms. When paired with downstream error correction methods, these models may further enhance medication data quality and, in future applications, support patient safety.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Data augmentation using RxNorm.</p>
        <media xlink:href="medinform_v14i1e91151_app1.pdf" xlink:title="PDF File  (Adobe PDF File), 63 KB"/>
      </supplementary-material>
      <supplementary-material id="app2">
        <label>Multimedia Appendix 2</label>
        <p>Creation of the LTCDC evaluation set. LTCDC: Long-Term Care Data Cooperative.</p>
        <media xlink:href="medinform_v14i1e91151_app2.pdf" xlink:title="PDF File  (Adobe PDF File), 159 KB"/>
      </supplementary-material>
      <supplementary-material id="app3">
        <label>Multimedia Appendix 3</label>
        <p>Additional information on the training and internal validation of BERTDrug and CharBERTDrug with the RxNorm-derived dataset.</p>
        <media xlink:href="medinform_v14i1e91151_app3.pdf" xlink:title="PDF File  (Adobe PDF File), 277 KB"/>
      </supplementary-material>
      <supplementary-material id="app4">
        <label>Multimedia Appendix 4</label>
        <p>Model performance on the LTCDC dataset and subsets. LTCDC: Long-Term Care Data Cooperative.</p>
        <media xlink:href="medinform_v14i1e91151_app4.pdf" xlink:title="PDF File  (Adobe PDF File), 198 KB"/>
      </supplementary-material>
      <supplementary-material id="app5">
        <label>Multimedia Appendix 5</label>
        <p>Performance comparison between CharBERTDrug, BERTDrug, and GPT-4o.</p>
        <media xlink:href="medinform_v14i1e91151_app5.pdf" xlink:title="PDF File  (Adobe PDF File), 142 KB"/>
      </supplementary-material>
      <supplementary-material id="app6">
        <label>Multimedia Appendix 6</label>
        <p>Bootstrap-based model comparisons in the RxNorm and LTCDC test sets and subsets. LTCDC: Long-Term Care Data Cooperative.</p>
        <media xlink:href="medinform_v14i1e91151_app6.xlsx" xlink:title="XLSX File  (Microsoft Excel File), 96 KB"/>
      </supplementary-material>
      <supplementary-material id="app7">
        <label>Multimedia Appendix 7</label>
        <p>Effects of term type and term length on model performance.</p>
        <media xlink:href="medinform_v14i1e91151_app7.pdf" xlink:title="PDF File  (Adobe PDF File), 98 KB"/>
      </supplementary-material>
      <supplementary-material id="app8">
        <label>Multimedia Appendix 8</label>
        <p>Projected model outputs in real-world application scenarios.</p>
        <media xlink:href="medinform_v14i1e91151_app8.pdf" xlink:title="PDF File  (Adobe PDF File), 191 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">CNN</term>
          <def>
            <p>convolutional neural network</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">CPOE</term>
          <def>
            <p>computerized provider order entry</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">DDI</term>
          <def>
            <p>drug-drug interaction</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">EHR</term>
          <def>
            <p>electronic health record</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">GRU</term>
          <def>
            <p>gated recurrent unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">HIPAA</term>
          <def>
            <p>Health Insurance Portability and Accountability Act</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">IRB</term>
          <def>
            <p>Institutional Review Board</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">LSTM</term>
          <def>
            <p>long short-term memory</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">LTCDC</term>
          <def>
            <p>Long-Term Care Data Cooperative</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">MIMIC-III</term>
          <def>
            <p>Medical Information Mart for Intensive Care III</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">MLM</term>
          <def>
            <p>masked language modeling</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">NSP</term>
          <def>
            <p>next sentence prediction</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb15">OR</term>
          <def>
            <p>odds ratio</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb16">PR-AUC</term>
          <def>
            <p>area under the precision-recall curve</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb17">RNN</term>
          <def>
            <p>recurrent neural network</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb18">ROC-AUC</term>
          <def>
            <p>area under the receiver operating characteristic curve</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors would like to thank Haochun Huang and Bargav Jagatha for their contributions to the initial annotation of the Long-Term Care Data Cooperative medication names and Dr Melissa Riester for adjudicating difficult cases during the process. OpenAI’s ChatGPT was used to support selected parts of the model-comparison experiments (as described in the Methods section) and to assist with language editing and text refinement. All outputs were reviewed, verified as appropriate, and approved by the authors.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The RxNorm database is publicly available from the U.S. National Library of Medicine. The code used to generate the synthetic dataset from RxNorm, the external validation dataset containing 3803 unique drug names from the Long-Term Care Data Cooperative database, the misspelling detection models, as well as other computer code developed for this study, are available through GitHub [<xref ref-type="bibr" rid="ref53">53</xref>].</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>Dr Chen was a Real World Data Scholar, funded by the National Institute on Aging of the National Institutes of Health (U54AG063546), which funds the National Institute on Aging’s Imbedded Pragmatic Alzheimer’s and AD-Related Dementias Clinical Trials (IMPACT) Collaboratory. Additional support was provided by the Boston University Chobanian &#38; Avedisian School of Medicine Data Science Core. The study used data from the Long-Term Care Data Cooperative (LTCDC), supported through a supplemental grant (U54AG063546-S6). The content of this paper is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health, the investigators of the NIA IMPACT Collaboratory, or the LTCDC. Dr Zullo was supported, in part, by grant R01AG077620 from the National Institute on Aging. The views expressed in this article are those of the authors and do not necessarily reflect the position or policy of the US Department of Veterans Affairs or the US government.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>JC conceived of and designed the study. JL developed the transformer-based misspelling detection models and associated code for data analysis and figure generation and performed the formal analysis. JC supervised the project, providing critical feedback on model development as well as guidance on experimental design and model evaluation. All authors (JL, KWM, ARZ, and JC) were involved in the acquisition and curation of data and contributed to the interpretation of findings. KWM and ARZ provided expertise in pharmacy and adjudicated difficult cases during the annotation of the Long-Term Care Data Cooperative medication names. JL and JC drafted the manuscript. All authors reviewed and revised the manuscript for important intellectual content.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Senger</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kaltschmidt</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Schmitt</surname>
              <given-names>SPW</given-names>
            </name>
            <name name-style="western">
              <surname>Pruszydlo</surname>
              <given-names>MG</given-names>
            </name>
            <name name-style="western">
              <surname>Haefeli</surname>
              <given-names>WE</given-names>
            </name>
          </person-group>
          <article-title>Misspellings in drug information system queries: characteristics of drug name spelling errors and strategies for their prevention</article-title>
          <source>Int J Med Inform</source>
          <year>2010</year>
          <volume>79</volume>
          <issue>12</issue>
          <fpage>832</fpage>
          <lpage>839</lpage>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2010.09.005</pub-id>
          <pub-id pub-id-type="medline">20951634</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(10)00166-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Darby</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Karas</surname>
              <given-names>BL</given-names>
            </name>
            <name name-style="western">
              <surname>Wagner</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>An analysis of the safety of medication ordering using typo correction within an academic medical system</article-title>
          <source>Appl Clin Inform</source>
          <year>2021</year>
          <volume>12</volume>
          <issue>3</issue>
          <fpage>655</fpage>
          <lpage>663</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/s-0041-1731745"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/s-0041-1731745</pub-id>
          <pub-id pub-id-type="medline">34341981</pub-id>
          <pub-id pub-id-type="pmcid">PMC8328746</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="web">
          <article-title>Medication safety for look-alike, sound-alike medicines</article-title>
          <source>World Health Organization</source>
          <year>2023</year>
          <month>10</month>
          <day>20</day>
          <access-date>2026-08-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.who.int/publications/i/item/9789240058897">https://www.who.int/publications/i/item/9789240058897</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dasaro</surname>
              <given-names>CR</given-names>
            </name>
            <name name-style="western">
              <surname>Sabra</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jeon</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>TA</given-names>
            </name>
            <name name-style="western">
              <surname>Sloan</surname>
              <given-names>NL</given-names>
            </name>
            <name name-style="western">
              <surname>Todd</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Teitelbaum</surname>
              <given-names>SL</given-names>
            </name>
          </person-group>
          <article-title>A comparison of two user-friendly methods to identify and support correction of misspelled medications</article-title>
          <source>Prev Med Rep</source>
          <year>2024</year>
          <volume>43</volume>
          <fpage>102765</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2211-3355(24)00180-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.pmedr.2024.102765</pub-id>
          <pub-id pub-id-type="medline">38798907</pub-id>
          <pub-id pub-id-type="pii">S2211-3355(24)00180-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC11127154</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Mahoney</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Shakurova</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Goss</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>FY</given-names>
            </name>
            <name name-style="western">
              <surname>Bates</surname>
              <given-names>DW</given-names>
            </name>
            <name name-style="western">
              <surname>Rocha</surname>
              <given-names>RA</given-names>
            </name>
          </person-group>
          <article-title>How many medication orders are entered through free-text in EHRs? A study on hypoglycemic agents</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2012</year>
          <volume>2012</volume>
          <fpage>1079</fpage>
          <lpage>1088</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23304384"/>
          </comment>
          <pub-id pub-id-type="medline">23304384</pub-id>
          <pub-id pub-id-type="pmcid">PMC3540584</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Turchin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Chu</surname>
              <given-names>JT</given-names>
            </name>
            <name name-style="western">
              <surname>Shubina</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Einbinder</surname>
              <given-names>JS</given-names>
            </name>
          </person-group>
          <article-title>Identification of misspelled words without a comprehensive dictionary using prevalence analysis</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2007</year>
          <volume>2007</volume>
          <fpage>751</fpage>
          <lpage>755</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/18693937"/>
          </comment>
          <pub-id pub-id-type="medline">18693937</pub-id>
          <pub-id pub-id-type="pmcid">PMC2813663</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lai</surname>
              <given-names>KH</given-names>
            </name>
            <name name-style="western">
              <surname>Topaz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Goss</surname>
              <given-names>FR</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Automated misspelling detection and correction in clinical free-text records</article-title>
          <source>J Biomed Inform</source>
          <year>2015</year>
          <volume>55</volume>
          <fpage>188</fpage>
          <lpage>195</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(15)00075-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2015.04.008</pub-id>
          <pub-id pub-id-type="medline">25917057</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(15)00075-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hussain</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Qamar</surname>
              <given-names>U</given-names>
            </name>
          </person-group>
          <article-title>Identification and correction of misspelled drugs' names in electronic medical records (EMR)</article-title>
          <source>Proceedings of the 18th International Conference on Enterprise Information Systems</source>
          <year>2016</year>
          <publisher-loc>Setubal, Portugal</publisher-loc>
          <publisher-name>SciTePress, Science and Technology Publications</publisher-name>
          <fpage>333</fpage>
          <lpage>338</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="web">
          <article-title>Medication errors linked to drug name confusion</article-title>
          <source>Patient Safety Authority</source>
          <year>2004</year>
          <access-date>2026-08-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://patientsafety.pa.gov/ADVISORIES/Pages/200412_07.aspx">https://patientsafety.pa.gov/ADVISORIES/Pages/200412_07.aspx</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Radley</surname>
              <given-names>DC</given-names>
            </name>
            <name name-style="western">
              <surname>Wasserman</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Olsho</surname>
              <given-names>LE</given-names>
            </name>
            <name name-style="western">
              <surname>Shoemaker</surname>
              <given-names>SJ</given-names>
            </name>
            <name name-style="western">
              <surname>Spranca</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Bradshaw</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Reduction in medication errors in hospitals due to adoption of computerized provider order entry systems</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2013</year>
          <volume>20</volume>
          <issue>3</issue>
          <fpage>470</fpage>
          <lpage>476</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23425440"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/amiajnl-2012-001241</pub-id>
          <pub-id pub-id-type="medline">23425440</pub-id>
          <pub-id pub-id-type="pii">amiajnl-2012-001241</pub-id>
          <pub-id pub-id-type="pmcid">PMC3628057</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hsu</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>CL</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ho</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>CY</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>YC</given-names>
            </name>
          </person-group>
          <article-title>Physicians failed to write flawless prescriptions when computerized physician order entry system crashed</article-title>
          <source>Clin Ther</source>
          <year>2015</year>
          <month>05</month>
          <day>01</day>
          <volume>37</volume>
          <issue>5</issue>
          <fpage>1076</fpage>
          <lpage>1080.e1</lpage>
          <pub-id pub-id-type="doi">10.1016/j.clinthera.2015.03.003</pub-id>
          <pub-id pub-id-type="medline">25841544</pub-id>
          <pub-id pub-id-type="pii">S0149-2918(15)00140-X</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Elshayib</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pawola</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Computerized provider order entry-related medication errors among hospitalized patients: an integrative review</article-title>
          <source>Health Informatics J</source>
          <year>2020</year>
          <volume>26</volume>
          <issue>4</issue>
          <fpage>2834</fpage>
          <lpage>2859</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/1460458220941750?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/1460458220941750</pub-id>
          <pub-id pub-id-type="medline">32744148</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Westbrook</surname>
              <given-names>JI</given-names>
            </name>
            <name name-style="western">
              <surname>Baysari</surname>
              <given-names>MT</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Burke</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Richardson</surname>
              <given-names>KL</given-names>
            </name>
            <name name-style="western">
              <surname>Day</surname>
              <given-names>RO</given-names>
            </name>
          </person-group>
          <article-title>The safety of electronic prescribing: manifestations, mechanisms, and rates of system-related errors associated with two commercial systems in hospitals</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2013</year>
          <volume>20</volume>
          <issue>6</issue>
          <fpage>1159</fpage>
          <lpage>1167</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23721982"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/amiajnl-2013-001745</pub-id>
          <pub-id pub-id-type="medline">23721982</pub-id>
          <pub-id pub-id-type="pii">amiajnl-2013-001745</pub-id>
          <pub-id pub-id-type="pmcid">PMC3822121</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ferner</surname>
              <given-names>RE</given-names>
            </name>
            <name name-style="western">
              <surname>Aronson</surname>
              <given-names>JK</given-names>
            </name>
          </person-group>
          <article-title>Nominal ISOMERs (Incorrect Spellings Of Medicines Eluding Researchers)—variants in the spellings of drug names in PubMed: a database review</article-title>
          <source>BMJ</source>
          <year>2016</year>
          <volume>355</volume>
          <fpage>i4854</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.bmj.com/lookup/pmidlookup?view=long&#38;pmid=27974346"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/bmj.i4854</pub-id>
          <pub-id pub-id-type="medline">27974346</pub-id>
          <pub-id pub-id-type="pmcid">PMC5156610</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tolentino</surname>
              <given-names>HD</given-names>
            </name>
            <name name-style="western">
              <surname>Matters</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Walop</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Law</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Tong</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Fontelo</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kohl</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>DC</given-names>
            </name>
          </person-group>
          <article-title>A UMLS-based spell checker for natural language processing in vaccine safety</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2007</year>
          <volume>7</volume>
          <fpage>3</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/1472-6947-7-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/1472-6947-7-3</pub-id>
          <pub-id pub-id-type="medline">17295907</pub-id>
          <pub-id pub-id-type="pii">1472-6947-7-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC1805499</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pimpalkhute</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Patki</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Nikfarjam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Phonetic spelling filter for keyword selection in drug mention mining from social media</article-title>
          <source>AMIA Jt Summits Transl Sci Proc</source>
          <year>2014</year>
          <volume>2014</volume>
          <fpage>90</fpage>
          <lpage>95</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/25717407"/>
          </comment>
          <pub-id pub-id-type="medline">25717407</pub-id>
          <pub-id pub-id-type="pmcid">PMC4333687</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Calix</surname>
              <given-names>RA</given-names>
            </name>
            <name name-style="western">
              <surname>Bernard</surname>
              <given-names>GR</given-names>
            </name>
          </person-group>
          <article-title>A data-driven method of discovering misspellings of medication names on Twitter</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2018</year>
          <volume>247</volume>
          <fpage>136</fpage>
          <lpage>140</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/29677938"/>
          </comment>
          <pub-id pub-id-type="medline">29677938</pub-id>
          <pub-id pub-id-type="pmcid">PMC6009827</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Forrest</surname>
              <given-names>CB</given-names>
            </name>
            <name name-style="western">
              <surname>McTigue</surname>
              <given-names>KM</given-names>
            </name>
            <name name-style="western">
              <surname>Hernandez</surname>
              <given-names>AF</given-names>
            </name>
            <name name-style="western">
              <surname>Cohen</surname>
              <given-names>LW</given-names>
            </name>
            <name name-style="western">
              <surname>Cruz</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Haynes</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Kaushal</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Kho</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Marsolo</surname>
              <given-names>KA</given-names>
            </name>
            <name name-style="western">
              <surname>Nair</surname>
              <given-names>VP</given-names>
            </name>
            <name name-style="western">
              <surname>Platt</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Puro</surname>
              <given-names>JE</given-names>
            </name>
            <name name-style="western">
              <surname>Rothman</surname>
              <given-names>RL</given-names>
            </name>
            <name name-style="western">
              <surname>Shenkman</surname>
              <given-names>EA</given-names>
            </name>
            <name name-style="western">
              <surname>Waitman</surname>
              <given-names>LR</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>NA</given-names>
            </name>
            <name name-style="western">
              <surname>Carton</surname>
              <given-names>TW</given-names>
            </name>
          </person-group>
          <article-title>PCORnet® 2020: current state, accomplishments, and future directions</article-title>
          <source>J Clin Epidemiol</source>
          <year>2021</year>
          <volume>129</volume>
          <fpage>60</fpage>
          <lpage>67</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/33002635"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.09.036</pub-id>
          <pub-id pub-id-type="medline">33002635</pub-id>
          <pub-id pub-id-type="pii">S0895-4356(20)31122-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC7521354</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>The All of Us Research Program Investigators</collab>
            <name name-style="western">
              <surname>Denny</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Rutter</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Goldstein</surname>
              <given-names>DB</given-names>
            </name>
            <name name-style="western">
              <surname>Philippakis</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Smoller</surname>
              <given-names>JW</given-names>
            </name>
            <name name-style="western">
              <surname>Jenkins</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Dishman</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>The "All of Us" Research Program</article-title>
          <source>N Engl J Med</source>
          <year>2019</year>
          <volume>381</volume>
          <issue>7</issue>
          <fpage>668</fpage>
          <lpage>676</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31412182"/>
          </comment>
          <pub-id pub-id-type="doi">10.1056/NEJMsr1809937</pub-id>
          <pub-id pub-id-type="medline">31412182</pub-id>
          <pub-id pub-id-type="pmcid">PMC8291101</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dore</surname>
              <given-names>DD</given-names>
            </name>
            <name name-style="western">
              <surname>Myles</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Recker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Burns</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Rogers Murray</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gifford</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Mor</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>The long-term care data cooperative: the next generation of data integration</article-title>
          <source>J Am Med Dir Assoc</source>
          <year>2022</year>
          <volume>23</volume>
          <issue>12</issue>
          <fpage>2031</fpage>
          <lpage>2033</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1525-8610(22)00729-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jamda.2022.09.006</pub-id>
          <pub-id pub-id-type="medline">36209889</pub-id>
          <pub-id pub-id-type="pii">S1525-8610(22)00729-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC9742312</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Patrick</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sabbagh</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Jain</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Spelling correction in clinical notes with emphasis on first suggestion accuracy</article-title>
          <year>2010</year>
          <conf-name>Proceedings of 2nd Workshop on Building and Evaluating Resources for Biomedical Text Mining (BioTxtM)</conf-name>
          <conf-date>2010</conf-date>
          <conf-loc>Valletta, Malta</conf-loc>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Pogrebnoi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Funkner</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kovalchuk</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>RuMedSpellchecker: a new approach for advanced spelling error correction in Russian electronic health records</article-title>
          <source>J Comput Sci</source>
          <year>2024</year>
          <volume>82</volume>
          <fpage>102393</fpage>
          <pub-id pub-id-type="doi">10.1016/j.jocs.2024.102393</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bravo-Candel</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>López-Hernández</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>García-Díaz</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Molina-Molina</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>García-Sánchez</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Automatic correction of real-word errors in Spanish clinical texts</article-title>
          <source>Sensors</source>
          <year>2021</year>
          <volume>21</volume>
          <issue>9</issue>
          <fpage>2893</fpage>
          <pub-id pub-id-type="doi">10.3390/s21092893</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kukich</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Techniques for automatically correcting words in text</article-title>
          <source>ACM Comput Surv</source>
          <year>1992</year>
          <volume>24</volume>
          <issue>4</issue>
          <fpage>377</fpage>
          <lpage>439</lpage>
          <pub-id pub-id-type="doi">10.1145/146370.146380</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <collab>Wei Ma</collab>
            <name name-style="western">
              <surname>Moore</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ganesan</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Nelson</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>RxNorm: prescription for electronic drug information exchange</article-title>
          <source>IT Prof</source>
          <year>2005</year>
          <volume>7</volume>
          <issue>5</issue>
          <fpage>17</fpage>
          <lpage>23</lpage>
          <pub-id pub-id-type="doi">10.1109/MITP.2005.122</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bodenreider</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Cornet</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Vreeman</surname>
              <given-names>DJ</given-names>
            </name>
          </person-group>
          <article-title>Recent developments in clinical terminologies – SNOMED CT, LOINC, and RxNorm</article-title>
          <source>Yearb Med Inform</source>
          <year>2018</year>
          <volume>27</volume>
          <issue>1</issue>
          <fpage>129</fpage>
          <lpage>139</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/s-0038-1667077"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/s-0038-1667077</pub-id>
          <pub-id pub-id-type="medline">30157516</pub-id>
          <pub-id pub-id-type="pmcid">PMC6115234</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhuang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zuccon</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Dealing with typos for BERT-based passage retrieval and ranking</article-title>
          <source>Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing</source>
          <year>2021</year>
          <publisher-loc>Kerrville, TX</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>2836</fpage>
          <lpage>2842</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>El Boukkouri</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Ferret</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Lavergne</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Noji</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zweigenbaum</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Tsujii</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>CharacterBERT: reconciling ELMo and BERT for word-level open-vocabulary representations from characters</article-title>
          <source>Proceedings of the 28th International Conference on Computational Linguistics</source>
          <year>2020</year>
          <publisher-loc>Barcelona, Spain</publisher-loc>
          <publisher-name>International Committee on Computational Linguistics</publisher-name>
          <fpage>6903</fpage>
          <lpage>6915</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Devlin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Toutanova</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title>
          <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)</source>
          <year>2019</year>
          <publisher-loc>Kerrville, TX</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>4171</fpage>
          <lpage>4186</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Aftan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Shah</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>A survey on BERT and its applications</article-title>
          <source>2023 20th Learning and Technology Conference (L&#38;T)</source>
          <year>2023</year>
          <publisher-loc>New York</publisher-loc>
          <publisher-name>IEEE</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaswani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shazeer</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Uszkoreit</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gomez</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Attention is all you need</article-title>
          <source>Proceedings of the 31st International Conference on Neural Information Processing Systems</source>
          <year>2017</year>
          <publisher-loc>Red Hook, NY</publisher-loc>
          <publisher-name>Curran Associates</publisher-name>
          <fpage>6000</fpage>
          <lpage>6010</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>AEW</given-names>
            </name>
            <name name-style="western">
              <surname>Pollard</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lehman</surname>
              <given-names>LH</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ghassemi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Moody</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Szolovits</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Celi</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Mark</surname>
              <given-names>RG</given-names>
            </name>
          </person-group>
          <article-title>MIMIC-III, a freely accessible critical care database</article-title>
          <source>Sci Data</source>
          <year>2016</year>
          <volume>3</volume>
          <fpage>160035</fpage>
          <pub-id pub-id-type="doi">10.1038/sdata.2016.35</pub-id>
          <pub-id pub-id-type="medline">27219127</pub-id>
          <pub-id pub-id-type="pii">sdata201635</pub-id>
          <pub-id pub-id-type="pmcid">PMC4878278</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="web">
          <article-title>PMC Open Access Subset</article-title>
          <source>National Library of Medicine</source>
          <year>2023</year>
          <access-date>2026-08-19</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://pmc.ncbi.nlm.nih.gov/tools/openftlist/">https://pmc.ncbi.nlm.nih.gov/tools/openftlist/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Srivastava</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Greff</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Schmidhuber</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Training very deep networks</article-title>
          <source>Proceedings of the 29th International Conference on Neural Information Processing Systems</source>
          <year>2015</year>
          <publisher-loc>Cambridge, MA</publisher-loc>
          <publisher-name>MIT Press</publisher-name>
          <fpage>2377</fpage>
          <lpage>2385</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Krallinger</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rabal</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Akhondi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pérez</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Santamaría</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Rodríguez</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Overview of the BioCreative VI chemical-protein interaction track</article-title>
          <year>2017</year>
          <conf-name>Proceedings of the Sixth BioCreative Challenge Evaluation Workshop</conf-name>
          <conf-date>October 18, 2017</conf-date>
          <conf-loc>Bethesda, MA</conf-loc>
          <fpage>141</fpage>
          <lpage>146</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Herrero-Zazo</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Segura-Bedmar</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Martínez</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Declerck</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>The DDI corpus: an annotated corpus with pharmacological substances and drug-drug interactions</article-title>
          <source>J Biomed Inform</source>
          <year>2013</year>
          <volume>46</volume>
          <issue>5</issue>
          <fpage>914</fpage>
          <lpage>920</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(13)00112-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2013.07.011</pub-id>
          <pub-id pub-id-type="medline">23906817</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(13)00112-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>López-Hernández</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Almela</surname>
              <given-names>Á</given-names>
            </name>
            <name name-style="western">
              <surname>Valencia-García</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Automatic spelling detection and correction in the medical domain: a systematic literature review</article-title>
          <source>Technologies and Innovation</source>
          <year>2019</year>
          <publisher-loc>Cham</publisher-loc>
          <publisher-name>Springer International Publishing</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hládek</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Staš</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pleva</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Survey of automatic spelling correction</article-title>
          <source>Electronics</source>
          <year>2020</year>
          <volume>9</volume>
          <issue>10</issue>
          <fpage>1670</fpage>
          <pub-id pub-id-type="doi">10.3390/electronics9101670</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bojanowski</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Grave</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Joulin</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mikolov</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Enriching word vectors with subword information</article-title>
          <source>Trans Assoc Comput Linguist</source>
          <year>2017</year>
          <volume>5</volume>
          <fpage>135</fpage>
          <lpage>146</lpage>
          <pub-id pub-id-type="doi">10.1162/tacl_a_00051</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref40">
        <label>40</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>BioWordVec, improving biomedical word embeddings with subword information and MeSH</article-title>
          <source>Sci Data</source>
          <year>2019</year>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>52</fpage>
          <pub-id pub-id-type="doi">10.1038/s41597-019-0055-0</pub-id>
          <pub-id pub-id-type="medline">31076572</pub-id>
          <pub-id pub-id-type="pmcid">PMC6510737</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref41">
        <label>41</label>
        <nlm-citation citation-type="web">
          <source>GitHub</source>
          <access-date>2026-08-19</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/ncbi-nlp/BioSentVec#text-corpora">https://github.com/ncbi-nlp/BioSentVec#text-corpora</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref42">
        <label>42</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Spelling error correction with soft-masked BERT</article-title>
          <source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source>
          <year>2020</year>
          <publisher-loc>Kerrville, TX</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>882</fpage>
          <lpage>890</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref43">
        <label>43</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Etoori</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Chinnakotla</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mamidi</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Automatic spelling correction for resource-scarce languages using deep learning</article-title>
          <source>Proceedings of ACL 2018, Student Research Workshop</source>
          <year>2018</year>
          <publisher-loc>Kerrville, TX</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>146</fpage>
          <lpage>152</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref44">
        <label>44</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jayanthi</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Pruthi</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Neubig</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>NeuSpell: a neural spelling correction toolkit</article-title>
          <source>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations</source>
          <year>2020</year>
          <publisher-loc>Kerrville, TX</publisher-loc>
          <publisher-name>Association for Computational Linguistics</publisher-name>
          <fpage>158</fpage>
          <lpage>164</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref45">
        <label>45</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ghosh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kristensson</surname>
              <given-names>PO</given-names>
            </name>
          </person-group>
          <article-title>Neural networks for text correction and completion in keyboard decoding</article-title>
          <source>arXiv. Preprint posted online on Sep 19, 2017</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.1709.06429</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref46">
        <label>46</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kasmaiee</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kasmaiee</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Homayounpour</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Correcting spelling mistakes in Persian texts with rules and deep learning methods</article-title>
          <source>Sci Rep</source>
          <year>2023</year>
          <volume>13</volume>
          <issue>1</issue>
          <fpage>19945</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-023-47295-2</pub-id>
          <pub-id pub-id-type="medline">37968293</pub-id>
          <pub-id pub-id-type="pmcid">PMC10652024</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref47">
        <label>47</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Klemen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Božič</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Holdt</surname>
              <given-names>SA</given-names>
            </name>
            <name name-style="western">
              <surname>Robnik-Šikonja</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Neural spell-checker: beyond words with synthetic data generation</article-title>
          <source>International Conference on Text, Speech, and Dialogue. TSD 2024. Lecture Notes in Computer Science</source>
          <year>2024</year>
          <publisher-loc>Cham</publisher-loc>
          <publisher-name>Springer</publisher-name>
          <fpage>85</fpage>
          <lpage>96</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref48">
        <label>48</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ulčar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Robnik Šikonja</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>SloBERTa: Slovene monolingual large pretrained masked language model</article-title>
          <year>2021</year>
          <conf-name>Proceedings of Data Mining and Data Warehousing</conf-name>
          <conf-date>Oct 4, 2021</conf-date>
          <conf-loc>Ljubljana</conf-loc>
          <fpage>17</fpage>
          <lpage>20</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref49">
        <label>49</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Health</surname>
              <given-names>TLD</given-names>
            </name>
          </person-group>
          <article-title>Large language models: a new chapter in digital health</article-title>
          <source>Lancet Digit Health</source>
          <year>2024</year>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>e1</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2589-7500(23)00254-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/S2589-7500(23)00254-6</pub-id>
          <pub-id pub-id-type="medline">38123249</pub-id>
          <pub-id pub-id-type="pii">S2589-7500(23)00254-6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref50">
        <label>50</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sandmann</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Hegselmann</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fujarski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bickmann</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Wild</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Eils</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Varghese</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title>
          <source>Nat Med</source>
          <year>2025</year>
          <volume>31</volume>
          <issue>8</issue>
          <fpage>2546</fpage>
          <lpage>2549</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id>
          <pub-id pub-id-type="medline">40267970</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref51">
        <label>51</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Myers</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Dligach</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Miller</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Bitterman</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Mayampurath</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Churpek</surname>
              <given-names>MM</given-names>
            </name>
            <name name-style="western">
              <surname>Afshar</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Uncertainty estimation in diagnosis generation from large language models: next-word probability is not pre-test probability</article-title>
          <source>JAMIA Open</source>
          <year>2025</year>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>ooae154</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/jamiaopen/article-lookup/doi/10.1093/jamiaopen/ooae154"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamiaopen/ooae154</pub-id>
          <pub-id pub-id-type="medline">39802674</pub-id>
          <pub-id pub-id-type="pii">ooae154</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref52">
        <label>52</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gu</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Desai</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Probabilistic medical predictions of large language models</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>367</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01366-4</pub-id>
          <pub-id pub-id-type="medline">39702641</pub-id>
          <pub-id pub-id-type="pmcid">PMC11659327</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref53">
        <label>53</label>
        <nlm-citation citation-type="web">
          <article-title>Dataset</article-title>
          <source>GitHub</source>
          <access-date>2026-09-08</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/jchen-BUlab/NLP_misspelling_detect">https://github.com/jchen-BUlab/NLP_misspelling_detect</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
