<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMI</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id>
      <journal-title>JMIR Medical Informatics</journal-title>
      <issn pub-type="epub">2291-9694</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v14i1e87831</article-id>
      <article-id pub-id-type="pmid">42497410</article-id>
      <article-id pub-id-type="doi">10.2196/87831</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Persona-Driven Data Augmentation for Disease Name Recognition Across Rare and General Disease Corpora: Comparative Evaluation Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Pathak</surname>
            <given-names>Dhrubajyoti</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Xie</surname>
            <given-names>Jie</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>García-Barragán</surname>
            <given-names>Álvaro</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Pierre</surname>
            <given-names>Jude Crener Junior</given-names>
          </name>
          <degrees>ME</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-5021-9344</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Nishiyama</surname>
            <given-names>Tomohiro</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1538-8266</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Peng</surname>
            <given-names>Shaowen</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-4020-9100</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Wakamiya</surname>
            <given-names>Shoko</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-9371-1340</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Aramaki</surname>
            <given-names>Eiji</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution/>
            <institution>Nara Institute of Science and Technology</institution>
            <addr-line>8916-5 Takayamacho</addr-line>
            <addr-line>Ikoma, Nara, 630-0192</addr-line>
            <country>Japan</country>
            <phone>81 743725250</phone>
            <email>aramaki@is.naist.jp</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-0201-3609</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Nara Institute of Science and Technology</institution>
        <addr-line>Ikoma, Nara</addr-line>
        <country>Japan</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Eiji Aramaki <email>aramaki@is.naist.jp</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>24</day>
        <month>7</month>
        <year>2026</year>
      </pub-date>
      <volume>14</volume>
      <elocation-id>e87831</elocation-id>
      <history>
        <date date-type="received">
          <day>15</day>
          <month>11</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>18</day>
          <month>3</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>29</day>
          <month>6</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Jude Crener Junior Pierre, Tomohiro Nishiyama, Shaowen Peng, Shoko Wakamiya, Eiji Aramaki. Originally published in JMIR Medical Informatics (https://medinform.jmir.org), 24.07.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on https://medinform.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://medinform.jmir.org/2026/1/e87831" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Medical information extraction requires automatically identifying disease names and related terms in text. This task, known as named entity recognition (NER), relies on expert-annotated data that are costly to produce and often available only in limited quantities. Data augmentation (DA) aims to expand available training data; however, standard techniques such as synonym replacement and back-translation may introduce inappropriate substitutions or fail to preserve entity-label alignment, which is critical for sequence-labeling tasks. Although large language models can generate fluent text, their outputs may also contain factual inconsistencies or unintended changes if not carefully controlled.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study investigated whether persona-driven, document-level DA using a large language model could improve biomedical disease NER performance by generating diverse rephrasings of medical documents while preserving annotated entities.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We designed a DA framework using multiple personas that varied in medical expertise, personality, tone, and narrative style. Using prompting constrained by XML tags, each persona rephrased training documents while aiming to preserve annotated entity spans. We evaluated the framework on 2 biomedical disease NER datasets with complementary roles: RareDis, a low-resource rare disease corpus, and National Center for Biotechnology Information (NCBI) disease, a more general disease benchmark. Semantic fidelity and lexical diversity were measured using BERTScore and Bilingual Evaluation Understudy (BLEU-4), respectively, and personas were grouped into high-, balanced-, and low-fidelity subsets. Biomedical pretrained BioBERT models were fine-tuned and evaluated under multiple settings, including gold-standard (GS) data only, synonym replacement, single-persona augmentation, curated persona subsets, and all-persona augmentation. Performance was assessed using microaveraged entity-level precision, recall, and <italic>F</italic><sub>1</sub>-score, and results were examined at both the overall and individual entity-type levels. Performance values are reported as mean (SD).</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Persona-driven DA improved NER performance over GS-only training in both datasets, with the strongest gains obtained by combining multiple persona-generated variants with GS data. In RareDis, the best result was achieved by the low-fidelity subset (mean <italic>F</italic><sub>1</sub>-score 73.35, SD 0.19 vs baseline 71.22, SD 0.45), while in NCBI disease, the all-personas setting performed best (mean <italic>F</italic><sub>1</sub>-score 89.32, SD 0.26 vs baseline 87.82, SD 0.18). In low-resource experiments, the all-personas and high-fidelity persona settings in NCBI disease exceeded the performance of the model trained on 100% GS data using only 60% of the training data, whereas gains in RareDis were more modest. Entity-level analysis showed improvements across RareDis categories, particularly for symptom, and confusion analysis indicated reduced symptom-sign confusion under augmentation.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>Persona-driven DA improved biomedical disease NER by introducing controlled linguistic variation while largely preserving annotated entities. The strongest gains were obtained when multiple persona-generated variants were combined with GS data, although the benefit varied across datasets. These findings suggest that this approach is a promising strategy for low-resource biomedical NER.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>data augmentation</kwd>
        <kwd>disease</kwd>
        <kwd>large language model</kwd>
        <kwd>LLM</kwd>
        <kwd>low-resource</kwd>
        <kwd>named entity recognition</kwd>
        <kwd>natural language processing</kwd>
        <kwd>NER</kwd>
        <kwd>NLP</kwd>
        <kwd>persona</kwd>
        <kwd>rare disease</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The identification of medical conditions in clinical texts and the literature relies on natural language processing (NLP) techniques, particularly named entity recognition (NER), which detects words or token spans and categorizes them into established medical categories [<xref ref-type="bibr" rid="ref1">1</xref>]. In the biomedical domain, this task is also known as biomedical NER (BioNER) and typically focuses on entities such as diseases, drugs, and proteins. Accurate NER supports many medical applications. It enables the creation of structured information from electronic health records by extracting entities from unstructured notes, supports pharmacovigilance by detecting adverse drug events in clinical text, and supports precision medicine through patient phenotyping [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref7">7</xref>].</p>
      <p>NER has long been recognized as a cornerstone of information extraction, with its importance first highlighted during the Message Understanding Conference-6 (MUC-6) [<xref ref-type="bibr" rid="ref8">8</xref>]. Early NER systems relied on rule-based and dictionary-driven methods, which provided modest accuracy but were limited in scalability [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. These were gradually replaced by statistical models [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref14">14</xref>] and later by deep learning approaches that incorporated transformer-based architectures such as BERT (Bidirectional Encoder Representations from Transformers) [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. More recently, the introduction of large language models (LLMs) has further advanced the scalability of NER tasks [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>].</p>
      <p>Despite these benefits, progress in BioNER remains limited by the scarcity of annotated data. Existing datasets are often small, unbalanced, and restricted to narrow domains, which can lead to model overfitting and weaken generalization. This low-resource setting represents a major barrier across medical NLP tasks, since annotation requires costly domain expertise [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p>
      <p>Data augmentation (DA) has emerged as a strategy to address this challenge by generating synthetic examples from existing data. Traditional methods such as back-translation, synonym replacement, and rule-based perturbations have shown promising results for tasks involving single sentences (eg, sentiment analysis and text classification) and sentence-pair tasks (eg, natural language inference and machine translation), as long as sentence coherence is preserved [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. However, for NER, these approaches are often less effective: they may fail to maintain entity–label consistency, struggle to produce diverse yet coherent outputs, and perform poorly with specialized biomedical terminology [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. In particular, BioNER DA presents additional challenges, as many existing methods rely on external linguistic resources, such as dependency parsers and WordNet, which are designed for the general domain rather than biomedical text. As a result, they often produce incorrect parses and inappropriate substitutions. Other strategies depend on training task-specific language models, which increases computational costs and limits applicability in low-resource settings [<xref ref-type="bibr" rid="ref24">24</xref>]. Together, these limitations motivate the evaluation of more controllable augmentation strategies for BioNER.</p>
      <p>The recent rise of generative AI systems with LLMs has created new opportunities. These models can generate fluent, natural-sounding text for biomedical DA. However, they also introduce risks, such as factual inaccuracies, limited lexical diversity, and scientific inconsistencies that can degrade model performance if used directly without supervision as training data. To address these issues, careful prompt design and postprocessing are necessary to ensure quality and accuracy [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref26">26</xref>].</p>
      <p>In this study, we introduce a controlled, document-level DA framework for BioNER based on persona-conditioned LLM prompting. Instead of traditional sentence-level paraphrasing, our method generates document-level rephrasings using persona-conditioned prompts that introduce controlled linguistic variation through differences in medical expertise, communication style, and narrative perspective. By constraining entity mentions within XML tags, the framework is intended to preserve annotated spans while allowing controlled variation in the surrounding nonentity text.</p>
      <p>To evaluate the proposed method across biomedical disease NER settings, we assess it on 2 datasets with complementary roles: a low-resource rare disease corpus and a more general disease-focused benchmark. Rare disease NER serves as an important motivating case for low-resource biomedical information extraction because such conditions are associated with limited clinical exposure, specialized terminology, and costly expert annotation [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]. In contrast, the more general disease benchmark allows us to examine whether the effects of persona-driven DA are specific to the rare disease setting or also hold in a broader disease NER context. Together, these datasets enable a more comprehensive evaluation across specialized and more general biomedical disease NER conditions.</p>
      <p>Specifically, we address the following 3 research questions (RQs):</p>
      <list list-type="bullet">
        <list-item>
          <p>RQ1: can persona-driven DA improve disease NER performance over nonaugmented training?</p>
        </list-item>
        <list-item>
          <p>RQ2: which persona types are most effective for generating useful rephrasings?</p>
        </list-item>
        <list-item>
          <p>RQ3: how do persona selection strategies (all vs curated subsets) influence NER performance?</p>
        </list-item>
      </list>
      <p>Through these questions, we examine the utility of persona-driven DA for biomedical disease NER, including its effects on performance, linguistic variation, and the relative value of curated subsets vs all-persona combinations. By studying these factors across complementary disease NER settings, this study provides a broader evaluation of performance and linguistic variation.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Datasets</title>
        <p>We evaluated the proposed augmentation framework on 2 biomedical disease NER datasets: RareDis and National Center for Biotechnology Information (NCBI) disease. RareDis served as the primary low-resource rare disease benchmark, whereas NCBI disease was included as an additional disease-focused corpus for broader validation beyond the rare disease setting.</p>
        <p>RareDis is a corpus specifically developed for rare disease NER and was built from 1041 English texts collected from the National Organization for Rare Disorders (NORD) website [<xref ref-type="bibr" rid="ref33">33</xref>]. RareDis contains 4 main entity types: rare disease, disease, sign, and symptom. Diseases represent general medical conditions commonly encountered in clinical settings. Rare diseases, by contrast, are low-prevalence conditions formally recognized as distinct clinical entities. Signs refer to objective findings observed during physical examination or laboratory testing, whereas symptoms denote subjective experiences reported by patients that cannot be directly observed through medical tests. We used the original split of 729 training documents, 104 development documents, and 208 test documents.</p>
        <p>NCBI disease is a biomedical disease corpus of 793 documents derived from PubMed abstracts [<xref ref-type="bibr" rid="ref34">34</xref>]. Compared with RareDis, which consists of rare disease descriptions, NCBI disease reflects scientific biomedical literature that is typically more concise and terminologically dense. We used the standard split of 593 training documents, 100 development documents, and 100 test documents.</p>
        <p>The 2 datasets also differ in annotation scope. RareDis includes 4 clinically related entity categories, making it suitable for analyzing confusion among semantically close labels, whereas NCBI disease focuses on disease mentions only. Together, these datasets allowed us to assess the proposed method in both a multiclass rare disease NER setting and a broader disease-focused BioNER setting.</p>
      </sec>
      <sec>
        <title>Persona Design</title>
        <p>To introduce controlled stylistic variation during the document rephrasing process in the proposed DA method, we introduced a persona-driven data generation framework, as shown in <xref rid="figure1" ref-type="fig">Figure 1</xref>. We defined 10 personas with distinct characteristics, each representing a different level of medical expertise, from expert clinicians to nonspecialist narrators, while differing in age, educational background, and professional focus. In addition, personas were distinguished by different characteristics such as tone, writing style, and personality traits, which influenced the formality and narrative perspective of the generated outputs. Each persona also included a role-specific description that provided additional instructions for the generative model, guiding the LLM to produce rephrasings that consistently reflected the intended persona. The same persona set was applied across both RareDis and NCBI disease to support comparable evaluation across datasets. Complete persona definitions are reported in <xref ref-type="table" rid="table1">Table 1</xref>.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Persona-driven document-level data augmentation (DA) pipeline. Original documents with XML-tagged biomedical entities are rephrased by the LLM (GPT-4o) following persona-conditioned prompts. Persona definitions guided linguistic variation and helped ensure that gold-standard entity boundaries were preserved. Examples of synthetic outputs (P1, P2, and P10) illustrate controlled variation across medical knowledge levels.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e87831_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Characteristics of the 10 personas used to guide synthetic text generation. For each persona, we report demographic background, communication style, personality traits, and the specific role intended to shape the generated medical narratives.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="180"/>
            <col width="140"/>
            <col width="150"/>
            <col width="150"/>
            <col width="380"/>
            <thead>
              <tr valign="top">
                <td>Persona ID, age (years), and education</td>
                <td>Profession</td>
                <td>Tone and style</td>
                <td>Personality traits</td>
                <td>Role in data generation</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>P1; 48; PhD in molecular medicine</td>
                <td>Principal investigator: rare disease research laboratory</td>
                <td>Technical and methodical</td>
                <td>Rigorous and detail-oriented</td>
                <td>Synthesizes scientific knowledge about rare diseases into precise medical rephrasings. Emphasizes accuracy in terminology and structural clarity, ensuring high fidelity of biomedical content.</td>
              </tr>
              <tr valign="top">
                <td>P2; 50; MD (cardiology)</td>
                <td>Urban cardiologist</td>
                <td>Analytical and detailed</td>
                <td>Precise and innovative</td>
                <td>Conveys complex and technical insights into precise, structured explanations, ensuring that intricate medical information is communicated in an accessible and clinically relevant manner.</td>
              </tr>
              <tr valign="top">
                <td>P3; 38; PharmD</td>
                <td>Clinical Pharmacist</td>
                <td>Concise and pragmatic</td>
                <td>Detail-oriented and cautious</td>
                <td>Translates clinical-medication interactions and disease mechanisms into practical, structured explanations. Focuses on treatment implications and physiological clarity, avoiding overgeneralizations while maintaining clinical accuracy.</td>
              </tr>
              <tr valign="top">
                <td>P4; 45; nursing</td>
                <td>Community family physician</td>
                <td>Warm and accessible</td>
                <td>Compassionate and pragmatic</td>
                <td>Synthesizes advanced clinical insights into accessible explanations, effectively bridging expert-level knowledge and everyday language for a broader audience.</td>
              </tr>
              <tr valign="top">
                <td>P5; 41; master’s in science communication</td>
                <td>Health journalist</td>
                <td>Neutral and explanatory</td>
                <td>Curious and balanced</td>
                <td>Explains complex health topics for a general audience using journalistic clarity. Bridges scientific accuracy with accessible narrative flow, offering well-informed yet nontechnical language aligned with public health communication norms.</td>
              </tr>
              <tr valign="top">
                <td>P6; 36; BEd in biology education</td>
                <td>High school biology teacher</td>
                <td>Clear and analogy-driven</td>
                <td>Patient and enthusiastic</td>
                <td>Uses pedagogical framing and familiar metaphors to explain difficult medical ideas. Focuses on simplicity, using teaching logic to convey meaning.</td>
              </tr>
              <tr valign="top">
                <td>P7; 35; bachelor’s in science communications</td>
                <td>Patient advocate</td>
                <td>Empathetic and engaging</td>
                <td>Approachable and articulate</td>
                <td>Communicates health information in an accessible and motivating way, translating clinical data into actionable advice for patients.</td>
              </tr>
              <tr valign="top">
                <td>P8; 58; high school certificate</td>
                <td>Family caregiver (elderly mother)</td>
                <td>Gentle and anecdotal</td>
                <td>Empathetic and observant</td>
                <td>Rephrases medical narratives through lived caregiving experience, often focusing on symptoms and outcomes. Prioritizes clarity, emotional tone, and lay understanding, especially around patient experience and interpretation.</td>
              </tr>
              <tr valign="top">
                <td>P9; 40; high school diploma</td>
                <td>Informed layperson</td>
                <td>Conversational and straightforward</td>
                <td>Pragmatic and resourceful</td>
                <td>Represents the everyday perspective, highlighting common health challenges, concerns, and lay interpretations of medical information.</td>
              </tr>
              <tr valign="top">
                <td>P10; 32; associate degree in creative writing</td>
                <td>Chronic illness blogger</td>
                <td>Conversational and expressive</td>
                <td>Candid and emotional</td>
                <td>Reframes medical descriptions into storytelling formats. Captures the human and affective side of disease while preserving the scientific integrity. Adds emotional depth and informal diction to the text.</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
      </sec>
      <sec>
        <title>Synthetic Data Generation Pipeline</title>
        <p>For synthetic data generation, only documents in the training split were augmented. Each document in the RareDis training set and the NCBI disease training set was rephrased once per persona, resulting in 10 synthetic variants per source document. Throughout the text and figures, we refer to the original training documents from both datasets as gold-standard (GS) and to persona-generated variants as Pn (n = 1 – 10).</p>
        <p>The process was implemented using the GPT-4o (OpenAI) generative model with persona-customized prompts in a one-shot format. Each prompt specified the persona’s characteristics and instructions and included a single example of a GS document paired with its rephrased version to illustrate the desired transformation style. The illustrative example was first generated in a zero-shot setting and manually reviewed to confirm that XML-tagged entities were preserved and that the output matched the intended persona profile. The same single RareDis training document was used as the illustrative example across all personas and for both datasets to keep the prompting setup consistent; this example also included disease mentions, making it compatible with the NCBI disease setting. For GPT-4o text generation, the temperature parameter was set to 0 to reduce output variability and encourage more consistent, controlled rephrasings across personas.</p>
        <p>To preserve annotation quality, all prompts explicitly instructed the model to keep entity mentions wrapped in XML tags unchanged. This constrained rephrasing to nonentity spans and helped ensure that GS entity boundaries were preserved. A descriptive statistical overview of the persona-based variants and GS documents across both datasets is provided in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Baselines and Comparison Methods</title>
        <p>To evaluate the contribution of persona-driven DA, we compared the proposed method with 2 baseline settings: a GS baseline and a synonym replacement (SR) baseline. In the GS setting, models were trained using only the original GS training data without augmentation. This served as the primary reference condition for both RareDis and NCBI disease.</p>
        <p>As a traditional augmentation baseline, we used SR following Dai and Adel [<xref ref-type="bibr" rid="ref25">25</xref>]. We used the authors’ released implementation from the associated GitHub repository [<xref ref-type="bibr" rid="ref35">35</xref>] to generate the SR-augmented training data. In this NER-oriented formulation, selected tokens are replaced with WordNet synonyms, and labels are updated when multitoken synonyms are introduced, meaning that the original label sequence is not always preserved. Similar to persona-driven DA, SR was applied only to the training splits. However, unlike the proposed XML-constrained persona-driven framework, SR does not explicitly preserve original entity mentions. SR therefore served as a conventional lexical-perturbation baseline for comparison with the proposed method.</p>
        <p>In addition to these baselines, we evaluated several persona-driven DA settings in which the same original GS training data were augmented using either a single persona, selected persona subsets, or all personas. Subsets were defined either through development-set selection or by grouping personas according to their fidelity/diversity characteristics. These settings were used to examine whether the effectiveness of the proposed framework depended primarily on individual personas, selected persona combinations, or broader stylistic diversity.</p>
      </sec>
      <sec>
        <title>Evaluation Metrics</title>
        <p>To assess the quality of the generated texts, we used 2 complementary metrics. Semantic fidelity was measured using BERTScore [<xref ref-type="bibr" rid="ref36">36</xref>], which compares reference and candidate texts using contextual token embeddings rather than exact word matches. Given a reference text <italic>x</italic> and a candidate text <italic>x̂</italic>, BERTScore precision, recall, and <italic>F</italic><sub>1</sub>-score are defined as:</p>
        <graphic xlink:href="medinform_v14i1e87831_fig6.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <p>Where <italic>x</italic>ᵢᵀ <italic>x̂</italic>ⱼ denotes the cosine similarity between contextual token embeddings. Higher BERTScore indicates stronger semantic preservation of the GS text.</p>
        <p>Lexical variation was characterized using Bilingual Evaluation Understudy (BLEU) [<xref ref-type="bibr" rid="ref37">37</xref>], specifically BLEU-4, which measures modified n-gram overlap up to 4-grams. BLEU-4 is computed as:</p>
        <graphic xlink:href="medinform_v14i1e87831_fig7.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        <p>where <italic>p<sub>n</sub></italic> is the modified n-gram precision, <italic>BP</italic> is the brevity penalty, and <italic>w<sub>n</sub></italic>=1/4. In our setting, lower BLEU-4 indicates greater surface variation from the GS text, whereas higher BLEU-4 indicates closer lexical overlap. For each GS training document in each dataset, we compared it with its persona-driven rephrased version and averaged the resulting BERTScore and BLEU-4 values across outputs for each persona.</p>
        <p>For downstream NER evaluation, we used microaveraged entity-level precision, recall, and <italic>F</italic><sub>1</sub>-score with strict span matching. Using the <italic>seqeval</italic> library, a predicted entity was considered correct, or a true positive (TP), only if its boundary and entity type exactly matched those of a GS entity. Predictions with incorrect boundaries or labels, as well as additional predicted entities not present in the GS, were counted as false positives (FPs), whereas GS entities missed by the model were counted as false negatives (FNs). Formally,</p>
        <graphic xlink:href="medinform_v14i1e87831_fig8.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </sec>
      <sec>
        <title>Experimental Setting</title>
        <p>All models were trained and evaluated separately on RareDis and NCBI disease using their respective standard training, development, and test splits. We adopted the standard inside-outside-beginning tagging scheme for NER, in which B- marks the beginning of an entity span, I- marks its continuation, and O marks all nonentity tokens. For the model backbone, we used BioBERT (dmis-lab/biobert-v1.1) [<xref ref-type="bibr" rid="ref17">17</xref>], a BERT [<xref ref-type="bibr" rid="ref16">16</xref>] model pretrained on biomedical texts.</p>
        <p>Training and evaluation were conducted at the sentence level. Each document was first split into sentences. XML tags were removed, and the sentences were tokenized so that input tokens could be paired with their corresponding entity labels. During training, these token-label pairs were fed into the model to optimize entity predictions. During evaluation, sentences from the test set were tokenized in the same way but without labels, allowing the model to output predicted entity tags for comparison with the GS test set.</p>
        <p>Models were trained with a maximum sequence length of 256 tokens, a batch size of 32, a learning rate of 1e-5, and the AdamW optimizer. Training was run for up to 10 epochs, with early stopping applied if no improvement was observed on the development set for 5 consecutive epochs. Each experiment was repeated 3 times using different seeds. For each run, the checkpoint with the best development-set <italic>F</italic><sub>1</sub>-score was retained for final evaluation. Final precision, recall, and <italic>F</italic><sub>1</sub>-score were then summarized across the 3 runs and reported as mean values with their corresponding SDs. All experiments were conducted on a single NVIDIA RTX PRO 6000 graphics processing unit.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>This study used previously collected BioNER datasets. The original dataset publications describe the data sources and collection procedures. NCBI disease is publicly available, and RareDis was used in accordance with the access conditions specified by its original authors. According to the Ethical Guidelines for Medical and Health Research Involving Human Subjects in Japan, this study does not fall under research involving human subjects and is not subject to ethical review. This determination is consistent with the institutional regulations of Nara Institute of Science and Technology [<xref ref-type="bibr" rid="ref38">38</xref>]. Because this study involved only secondary analysis of existing datasets and did not involve new data collection from human participants, identifiable personal information, identifiable participant images, or participant compensation, formal ethics review and additional informed consent were not required.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p>Using our persona-driven DA framework, we generated multiple rephrasings of the RareDis and NCBI disease training documents and evaluated their impact on NER performance. The use of distinct personas introduced measurable and meaningful linguistic variation. Our evaluation covered several stages: analysis of the fidelity and diversity of generated texts, evaluation of individual persona settings, comparison of augmentation settings that combined persona-generated texts with GS data, low-resource experiments, and error analysis.</p>
      </sec>
      <sec>
        <title>Diversity vs Fidelity Landscape</title>
        <p>We first assessed the quality of the persona-generated texts by quantifying both semantic fidelity and lexical diversity, as shown in <xref rid="figure2" ref-type="fig">Figure 2</xref>A and Tables S3 and S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Fidelity refers to how well the rephrased text preserves the semantic content of the GS texts, which is important for avoiding unintended changes in the training data. Diversity, on the other hand, measures the degree of surface-level variation in wording and phrasing, which may help reduce overfitting, expose the model to alternative lexical forms, and support generalization to unseen texts.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Diversity–fidelity landscape and its relation to named entity recognition (NER) performance across personas in the RareDis and National Center for Biotechnology Information (NCBI) disease datasets. (A) Plots persona-generated texts by BERTScore F1-score and Bilingual Evaluation Understudy-4 (BLEU-4) for RareDis and NCBI disease, respectively, where BERTScore F1-score reflects semantic fidelity to the gold-standard (GS) texts and lower BLEU-4 indicates greater lexical diversity. Shaded regions denote high-, balanced-, and low-fidelity persona groups. (B) NER F1-score for each persona setting in relation to BERTScore F1-score and BLEU-4. Dashed and dotted horizontal lines indicate the GS-only and synonym replacement (SR) baselines.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e87831_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>BERTScore evaluated semantic fidelity by comparing persona-generated texts against the GS texts, while BLEU-4 was used to assess surface-level similarity and relative variation across persona outputs. Overall, both datasets showed 3 broad persona groupings. In RareDis, high-fidelity personas (P2 and P4) achieved the highest BERTScore (≥97.0) and BLEU-4 (≥55), generating outputs that remained very close to the GS texts with limited stylistic variation. Low-fidelity personas (P6, P8, P9, and P10) showed the lowest values (BERTScore ≤96.0; BLEU-4 ≤45), indicating the most diverse but least faithful outputs. Balanced-fidelity personas (P1, P3, P5, and P7) occupied the middle range (BERTScore: 96.1-96.8; BLEU-4: 46-53). In NCBI disease, high-fidelity personas (P1, P2, P3, and P4) also formed the top cluster (BERTScore ≥97.4; BLEU-4 ≥62.0), whereas low-fidelity personas (P6, P9, and P10) showed the lowest values (BERTScore ≤96.5; BLEU-4 ≤54.5). Balanced-fidelity personas (P5, P7, and P8) remained in the intermediate range (BERTScore 96.7–97.3; BLEU-4 56–61).</p>
        <p>Overall, NCBI disease showed generally higher BERTScore and BLEU-4 values than RareDis, indicating that persona-generated texts remained closer to the GS texts. Although several personas remained relatively strong across datasets, the overall clustering pattern was not identical, indicating that fidelity and diversity associated with each persona depended in part on the linguistic properties of the target corpus.</p>
      </sec>
      <sec>
        <title>Individual Persona NER Performance</title>
        <p>We next evaluated the effect of training solely on persona-generated texts, without combining them with the GS data. In this setting, each model trained on a single persona’s outputs was evaluated on the RareDis and NCBI disease test sets (<xref rid="figure2" ref-type="fig">Figure 2</xref>B). The results suggested that individual-persona NER performance tended to align more closely with higher-fidelity outputs than with more lexically diverse, lower-fidelity outputs. In RareDis, P4 and P7 achieved the highest average <italic>F</italic><sub>1</sub>-scores, with P4 performing best. These results included both high-fidelity and balanced-fidelity personas, with other personas such as P2 and P5 also remaining competitive. In NCBI disease, high-fidelity personas again tended to perform best, especially P2 and P3, although the ranking pattern was not identical across datasets. By contrast, low-fidelity personas, which introduced greater lexical variation, generally produced lower <italic>F</italic><sub>1</sub>-scores.</p>
        <p>These results reveal an important property of the framework: no single persona captures sufficient linguistic diversity to match the GS training data on its own, while all individual persona settings consistently remained above the SR baseline, confirming that persona-generated texts preserve task-relevant information. This pattern suggests that the strength of persona-driven DA lies not in any individual persona but in the complementary diversity that emerges when multiple perspectives are combined, with each persona contributing a distinct stylistic dimension that no single rephrasing style can replicate alone. Consequently, we next examined whether combining persona-generated texts with GS texts offered a more effective strategy for improving NER performance.</p>
      </sec>
      <sec>
        <title>Comparison of Persona-Mix Augmentation Strategies</title>
        <p>We then examined whether combining persona-generated texts could lead to stronger improvements in NER performance. In the single-persona DA setting (GS + P), each persona’s outputs were added to the GS texts. In the curated-subset DA setting, we tested combinations based on the fidelity clusters, as well as a top-3 subset formed from the strongest individual personas based on development-set performance in the single-persona (GS + P) setting, as shown in Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The top-3 personas were P2, P4, and P7 for RareDis, and P2, P3, and P8 for NCBI disease. Detailed test-set precision, recall, and <italic>F</italic><sub>1</sub>-scores, together with their SDs, are provided in Tables S6 and S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>As shown in <xref rid="figure3" ref-type="fig">Figure 3</xref>, combining personas was generally more effective than relying on a single persona. In RareDis, the low-fidelity subset achieved the best overall performance, followed closely by the all-personas setting, while the balanced-fidelity and top-3 subsets also produced strong gains. In NCBI disease, the all-personas setting achieved the highest mean <italic>F</italic><sub>1</sub>-score, followed closely by the high-fidelity and top-3 subsets.</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Overall F1-scores for single-persona and curated-subset data augmentation (DA) settings on the (A) RareDis and (B) National Center for Biotechnology Information (NCBI) disease datasets. The dashed horizontal lines indicate the gold-standard (GS) and synonym replacement (SR) baseline performances.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e87831_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>This trend was also reflected in the per-entity analysis presented in Tables S8 and S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. In RareDis, the best scores were obtained by broader persona mixtures rather than by any single-persona DA setting, with the top-3 subset performing best for disease, the all-personas setting for rare disease, and the low-fidelity subset for sign and symptom. In NCBI disease, the highest disease <italic>F</italic><sub>1</sub>-score was achieved by the all-personas setting. Although differences in entity counts across augmentation settings (Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) may have contributed in part to some of these patterns, they do not fully account for them.</p>
      </sec>
      <sec>
        <title>Low-Resource NER</title>
        <p>Because manually annotated biomedical corpora are often limited, we further examined whether persona-based augmentation could improve data efficiency under low-resource conditions. To do so, we randomly subsampled 20% to 100% of the GS training data, in 20% increments, together with the corresponding augmented data. For each seed, document subsampling was performed independently at each training percentage, and results were averaged across the 3 runs. We then compared 4 settings: GS-only, GS + SR, GS combined with the best-performing curated subset for each dataset (low-fidelity for RareDis and high-fidelity for NCBI disease), and GS combined with all personas (<xref rid="figure4" ref-type="fig">Figures 4</xref>A and 4B). The results showed that the gains from augmentation were larger in NCBI disease than in RareDis. Detailed low-resource average <italic>F</italic><sub>1</sub>-scores and SDs per training data percentage are provided in Tables S13 and S14 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Average F1-scores across training data percentages (20%-100%) for gold-standard (GS) and selected augmentation settings on (A) RareDis and (B) National Center for Biotechnology Information (NCBI) disease. The dashed horizontal line indicates the baseline performance obtained with 100% GS data.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e87831_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>In NCBI disease, GS combined with all personas achieved the highest average <italic>F</italic><sub>1</sub>-scores across training-data percentages, while the high-fidelity subset also performed strongly. Both settings surpassed the 100% GS baseline using only 60% of the GS training data. In RareDis, the all-personas setting performed best from 20% to 80% of the GS training data, whereas the low-fidelity subset achieved the highest score under the 100% training condition. However, the gains were modest, and no augmented setting surpassed the 100% GS baseline before the full training condition. This likely reflects the greater difficulty of RareDis as a multiclass corpus with specialized terminology and fine-grained entity distinctions. Overall, these findings indicate that persona-driven DA produced clearer gains for NCBI disease under low-resource settings, while its effects on RareDis were more moderate.</p>
      </sec>
      <sec>
        <title>Error Analysis of Generated Entities and NER Confusions</title>
        <p>Although XML-constrained prompting was designed to preserve GS entity spans, analysis of the generated texts revealed occasional modifications by the LLM. To identify candidate cases, we compared XML-annotated entities of the same label between the GS and persona-generated texts using Python’s built-in library <italic>difflib.SequenceMatcher</italic>. Cases with a similarity score of 1.0 were treated as unchanged and excluded, whereas cases with scores between 0.70 and 0.99 were treated as modified entities. Representative examples are shown in <xref ref-type="table" rid="table2">Table 2</xref>.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Examples of entity-level modifications in persona outputs. The table compares gold-standard (GS) and persona-generated sentences, highlighting cases in which entities were modified.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="100"/>
            <col width="450"/>
            <col width="450"/>
            <thead>
              <tr valign="top">
                <td>Source</td>
                <td>Gold-standard text</td>
                <td>Persona-generated text</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>RareDis</td>
                <td>...is caused by a &#60;sign&#62;<italic>deficiency of an enzyme galactose-1-phosphate uridylyl transferase</italic>&#60;/sign&#62; (GALT)...</td>
                <td>...results from a &#60;sign&#62;<italic>deficiency of the enzyme galactose-1-phosphate uridylyl transferase</italic>&#60;/sign&#62; (GALT)...</td>
              </tr>
              <tr valign="top">
                <td>RareDis</td>
                <td>...This causes the &#60;sign&#62;<italic>eyes to become red</italic>&#60;/sign&#62; and may cause &#60;sign&#62;blurred vision&#60;/sign&#62;...</td>
                <td>...leading to &#60;sign&#62;<italic>redness of the eyes</italic>&#60;/sign&#62; and potentially &#60;sign&#62;blurred vision&#60;/sign&#62;...</td>
              </tr>
              <tr valign="top">
                <td>RareDis</td>
                <td>...is a rare &#60;disease&#62;<italic>cancer of the nasal cavity and/or paranasal sinuses</italic>&#60;/disease&#62;...</td>
                <td>...is a rare form of &#60;disease&#62;<italic>cancer affecting the nasal cavity and/or paranasal sinuses</italic>&#60;/disease&#62;...</td>
              </tr>
              <tr valign="top">
                <td>NCBI<sup>a</sup></td>
                <td>...confirming the &#60;disease&#62;<italic>Pelizaeus Merzbacher disease</italic>&#60;/disease&#62;, but neither the aunt nor the fetus carried a duplication.</td>
                <td>...confirming the &#60;disease&#62;<italic>Pelizaeus-Merzbacher disease</italic>&#60;/disease&#62; diagnosis, but neither the aunt nor the fetus had the duplication.</td>
              </tr>
              <tr valign="top">
                <td>NCBI</td>
                <td>...indicated no close linkage between the &#60;disease&#62;<italic>CM deficienty</italic>&#60;/disease&#62; and the HLA system...</td>
                <td>...showed no close linkage between the &#60;disease&#62;<italic>CM deficiency</italic>&#60;/disease&#62; and the HLA system...</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>NCBI: National Center for Biotechnology Information.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>The selected examples in <xref ref-type="table" rid="table2">Table 2</xref> indicate that these changes were generally minor lexical or orthographic reformulations, such as rewording sign expressions, adding function words, or slightly reformulating disease mentions, while usually preserving the underlying medical meaning. Overall, entity integrity was strongly preserved across both datasets: entity modification rates ranged from 0.79% (72/9134) to 1.60% (146/9134) in RareDis (Table S11 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) and from 0.21% (11/5246) to 0.42% (22/5246) in NCBI disease (Table S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), confirming that XML-constrained prompting was highly effective at preventing unintended entity changes.</p>
        <p>As also reflected in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, persona-generated texts sometimes contained slightly fewer entity occurrences than the GS texts, likely because some personas did not always preserve the same frequency of repeated entity mentions when an entity appeared multiple times in the original document. For training, token-label pairs were generated from the final XML annotations in each synthetic document. Therefore, omitted repeated mentions simply did not appear as labeled spans, while tagged, reformulated mentions remained aligned with their labels.</p>
        <p>Beyond entity preservation, we analyzed model errors using normalized confusion matrices averaged across 3 seeds for the GS baseline and the best-performing augmented method in each dataset (<xref rid="figure5" ref-type="fig">Figures 5</xref>A and 5B). In RareDis, both models showed recurrent confusions between semantically close classes, especially disease vs rare disease and symptom vs sign. A visible improvement from augmentation was a reduction in symptom-sign confusion, whereas disease-rare disease confusions remained largely similar overall. In NCBI disease, confusion patterns were simpler, and augmentation produced only minor changes. Qualitative examples in Table S10 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> show that augmentation corrected some baseline errors, although some confusions remained.</p>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Average normalized confusion matrices across 3 seeds for the gold-standard (GS) baseline and the best-performing augmented model on (A) RareDis and (B) National Center for Biotechnology Information (NCBI) disease.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e87831_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Overview</title>
        <p>In this study, we introduced a persona-driven, document-level DA framework for biomedical disease NER and evaluated it on both a rare disease corpus and a more general disease benchmark. Our findings indicate that persona-driven DA can be applied in a way that largely preserves annotated entity spans and can improve baseline performance, particularly when multiple persona-generated variants are combined with the original training data. More broadly, the results provide insight into which persona types are most useful, how persona selection strategies influences performance, and how this framework may support low-resource training.</p>
      </sec>
      <sec>
        <title>RQ1: Can Persona-Driven DA Improve Disease NER Performance Over Nonaugmented Training?</title>
        <p>Our experiments show that persona-driven DA can improve disease NER performance over nonaugmented training, although the benefit depended on the augmentation strategy and the target corpus. In both datasets, the strongest gains came from combining multiple persona-generated variants with the GS data rather than using any single persona alone. In RareDis, the low-fidelity subset performed best, followed closely by the all-personas setting, whereas in NCBI disease, the all-personas setting achieved the highest mean <italic>F</italic><sub>1</sub>-score, with the high-fidelity and top-3 subsets also performing strongly. The low-resource analysis further supported this pattern. In NCBI disease, the all-personas and high-fidelity persona settings surpassed the baseline trained on 100% of the GS data using only 60% of the training data, whereas in RareDis, the gains were more modest and no augmented setting exceeded the full-data GS baseline until the 100% training condition. Taken together, these findings suggest that persona-driven DA can improve performance and, in some cases, data efficiency, although its impact depends on the characteristics of the target corpus.</p>
      </sec>
      <sec>
        <title>RQ2: Which Persona Types Are Most Effective for Generating Useful Rephrasings?</title>
        <p>Our results suggest that high-fidelity personas were the most effective when used individually. This pattern may reflect a more favorable balance between semantic fidelity and useful linguistic variation during rephrasing. In RareDis, P4 achieved the strongest individual performance, while in NCBI disease, the best results came from P2 and P3. By contrast, low-fidelity personas, although more diverse, generally performed less well when used alone. However, when combined with the GS data, personas from different fidelity groups contributed complementary strengths, suggesting that effective rephrasings depend on balancing semantic preservation with controlled linguistic variation.</p>
      </sec>
      <sec>
        <title>RQ3: How Do Persona Selection Strategies Influence NER Performance?</title>
        <p>Our comparison of augmentation strategies showed that the most effective persona selection strategy differed across datasets, indicating that no single strategy was uniformly optimal. This difference was also evident in the dataset-specific top-3 curated subsets selected from development-set performance in the GS + P setting: P2, P4, and P7 for RareDis, and P2, P3, and P8 for NCBI disease. Notably, these curated subsets included personas from different fidelity groups, suggesting that combining different rephrasing styles may be beneficial. Broader persona mixtures also introduced more synthetic training examples, so the observed gains may reflect both increased stylistic diversity and increased data volume. Overall, these findings indicate that persona selection matters, but the strongest strategy depends on the target corpus.</p>
      </sec>
      <sec>
        <title>Practical Implications of Persona-Driven Augmentation</title>
        <p>The present findings suggest that persona-driven DA may be particularly useful in BioNER settings in which annotated data are limited. In such contexts, generating controlled rephrasings of existing documents may help expand linguistic coverage without requiring additional manual annotations. The observed gains were not uniform across datasets, indicating that the practical benefit of this approach is likely to depend on corpus characteristics and task conditions.</p>
        <p>More broadly, the proposed framework shows that it is possible to generate multiple rephrased versions of biomedical texts while largely preserving annotated entity spans. This may be useful for other BioNER settings that face similar challenges related to limited annotated data and specialized terminology.</p>
      </sec>
      <sec>
        <title>Limitations and Future Work</title>
        <p>While promising, this work has several limitations. Although we evaluated the proposed framework on 2 biomedical disease NER datasets, RareDis and NCBI disease, both are disease-focused benchmarks. Further validation on other biomedical entity types, clinical text settings, and additional domains is needed to establish the broader generalizability of persona-driven DA.</p>
        <p>Another limitation concerns the design of the personas. Their characteristics were selected heuristically based on intuition about what might produce useful variation in biomedical rephrasings. Although the results suggest that this strategy was effective, it was not derived from a systematic optimization procedure and may not transfer equally well to other tasks or corpora.</p>
        <p>A further limitation is that we did not evaluate all possible persona combinations. With 10 personas, the combinatorial space is large, and exhaustive evaluation would be computationally costly. Instead, we focused on representative settings, including individual personas, fidelity-based subsets, top-3 curated subsets, and the all-personas configuration, in order to capture the main contrasts relevant to our research questions.</p>
        <p>Although the prompting framework was designed to preserve XML-tagged entities, quantitative analysis confirmed that modifications were rare, with rates remaining below 2% in RareDis and below 0.5% in NCBI disease (Tables S11 and S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Where modifications did occur, they were minor lexical or orthographic reformulations that generally did not alter the underlying medical meaning. Some persona-generated documents also contained slightly fewer entity occurrences than the original GS texts, likely because repeated mentions were not always reproduced during rephrasing. This indicates that the current framework was designed primarily to preserve entity forms and boundaries rather than to explicitly control entity frequency. We also note that the generated texts were not evaluated by clinical experts. This study focused on assessing whether persona-driven rephrasings are useful as augmentation data for NER under entity-preservation constraints rather than evaluating the clinical correctness or validity of the generated texts. Therefore, BERTScore and BLEU-4 were used only as descriptive proxy measures of semantic fidelity and lexical variation. Future work should include expert review to assess the factual accuracy and biomedical appropriateness of the generated outputs.</p>
        <p>An additional limitation concerns the dependence on GPT-4o, a closed-source proprietary model, for text generation. This raises concerns about accessibility, cost, and reproducibility. Moreover, while the present experiments were conducted in a research setting, the use of external proprietary LLMs may raise privacy, information governance, and potential information leakage concerns in real clinical deployment. Future work should therefore investigate whether open-source or on-premises LLMs can achieve a similar level of fidelity in following XML constraints and preserving entities with minimal alteration. Establishing such alternatives would improve the transparency, accessibility, and practical deployability of persona-driven DA in biomedical settings.</p>
      </sec>
      <sec>
        <title>Conclusion</title>
        <p>This study evaluated a persona-driven DA framework for improving biomedical disease NER using document-level rephrasings generated by GPT-4o. By combining XML-constrained prompts with persona-guided rephrasings, the framework produced synthetic data that introduced controlled linguistic variation while largely preserving annotated entities and improved performance over nonaugmented training in several settings.</p>
        <p>The analysis further showed that the effectiveness of persona-driven DA depended on both the persona selection strategy and the target corpus. The strongest gains were generally observed when multiple persona-generated variants were combined with the GS data. Most notably, in low-resource experiments on NCBI disease, the all-personas and high-fidelity persona settings outperformed the model trained on 100% of the GS data using only 60% of the training data, suggesting that persona-driven DA can meaningfully reduce annotation requirements without sacrificing performance. Entity-level analysis suggested some additional benefit in RareDis, where augmentation reduced certain confusions between closely related categories.</p>
        <p>Overall, these findings support the potential of persona-driven DA as a useful strategy for BioNER in settings with limited annotated data. Further evaluation across other biomedical tasks, datasets, and generation models will help clarify its broader applicability.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Supplementary dataset statistics, generation-quality metrics, named entity recognition (NER) performance results, entity-preservation analyses, error examples, and low-resource evaluation results for RareDis and National Center for Biotechnology Information (NCBI) disease.</p>
        <media xlink:href="medinform_v14i1e87831_app1.docx" xlink:title="DOCX File , 788 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">BERT</term>
          <def>
            <p>Bidirectional Encoder Representations from Transformers</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">BioNER</term>
          <def>
            <p>biomedical NER</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">BLEU</term>
          <def>
            <p>Bilingual Evaluation Understudy</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">DA</term>
          <def>
            <p>data augmentation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">FN</term>
          <def>
            <p>false negative</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">FP</term>
          <def>
            <p>false positive</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">GS</term>
          <def>
            <p>gold-standard</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">MUC-6</term>
          <def>
            <p>Message Understanding Conference-6</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">NCBI</term>
          <def>
            <p>National Center for Biotechnology Information</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">NER</term>
          <def>
            <p>named entity recognition</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">NLP</term>
          <def>
            <p>natural language processing</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">NORD</term>
          <def>
            <p>National Organization for Rare Disorders</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">RQ</term>
          <def>
            <p>research question</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb15">SR</term>
          <def>
            <p>synonym replacement</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb16">TP</term>
          <def>
            <p>true positive</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare that generative AI tools were used in a fully supervised manner during manuscript preparation and code development. Specifically, these tools were used only for language polishing of text originally written by the authors, improvements to grammar, clarity, and concision, as well as coding assistance, such as implementation suggestions and debugging support. All AI-assisted code was carefully tested and verified by the authors. The literature review, study design, methodology, analysis, interpretation of results, and all core scientific contributions were carried out by the authors. The authors take full responsibility for the final manuscript, its scientific content, and all reported results. Generative AI tools are not credited as authors.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The National Center for Biotechnology Information (NCBI) disease corpus used in this study is publicly available from the National Library of Medicine and National Institutes of Health. The code developed for augmentation and model training, together with the synthetic texts generated from the NCBI disease corpus, is available in our GitHub repository [<xref ref-type="bibr" rid="ref39">39</xref>]. For the RareDis dataset, access to the original texts is handled by the dataset authors on request. Because our RareDis-based synthetic texts were derived from source materials that are not openly redistributed and may remain close to the original texts, they are not publicly released.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>This work was supported by the Cross-ministerial Strategic Innovation Promotion Program (SIP) on “Integrated Health Care System” (grant number JPJ012425).</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>JCJP contributed to the conceptualization, methodology, data curation, investigation, formal analysis, visualization, and writing of the original draft, as well as the review and editing of the manuscript. TN contributed to review and editing of the manuscript and provided supervision. SP contributed to supervision. SW contributed to visualization and review and editing of the manuscript. EA contributed to visualization and supervision.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Alamro</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Gojobori</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Essack</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>BioBBC: a multi-feature model that enhances the detection of biomedical entities</article-title>
          <source>Sci Rep</source>
          <year>2024</year>
          <volume>14</volume>
          <issue>1</issue>
          <fpage>7697</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-024-58334-x</pub-id>
          <pub-id pub-id-type="medline">38565624</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Qi</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Bolton</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Manning</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Stanza: A python natural language processing toolkit for many human languages</article-title>
          <year>2020</year>
          <conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: System Demonstrations</conf-name>
          <conf-date>July 5-10, 2020</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>101</fpage>
          <lpage>108</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2020.acl-demos.14</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leaman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Islamaj Dogan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>DNorm: disease name normalization with pairwise learning to rank</article-title>
          <source>Bioinformatics</source>
          <year>2013</year>
          <volume>29</volume>
          <issue>22</issue>
          <fpage>2909</fpage>
          <lpage>2917</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23969135"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btt474</pub-id>
          <pub-id pub-id-type="medline">23969135</pub-id>
          <pub-id pub-id-type="pii">btt474</pub-id>
          <pub-id pub-id-type="pmcid">PMC3810844</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Wen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Large language models struggle in token-level clinical named entity recognition</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2024</year>
          <volume>2024</volume>
          <fpage>748</fpage>
          <lpage>757</lpage>
          <pub-id pub-id-type="medline">40417588</pub-id>
          <pub-id pub-id-type="pii">4784</pub-id>
          <pub-id pub-id-type="pmcid">PMC12099373</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shyr</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Bastarache</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hamid</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Harris</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Identifying and extracting rare diseases and their phenotypes with large language models</article-title>
          <source>J Healthc Inform Res</source>
          <year>2024</year>
          <volume>8</volume>
          <issue>2</issue>
          <fpage>438</fpage>
          <lpage>461</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38681753"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s41666-023-00155-0</pub-id>
          <pub-id pub-id-type="medline">38681753</pub-id>
          <pub-id pub-id-type="pii">155</pub-id>
          <pub-id pub-id-type="pmcid">PMC11052982</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Segura-Bedmar</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Camino-Perdones</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Guerrero-Aspizua</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Exploring deep learning methods for recognizing rare diseases and their clinical manifestations from texts</article-title>
          <source>BMC Bioinformatics</source>
          <year>2022</year>
          <volume>23</volume>
          <issue>1</issue>
          <fpage>263</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-022-04810-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12859-022-04810-y</pub-id>
          <pub-id pub-id-type="medline">35794528</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12859-022-04810-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC9258216</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cao</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Cross</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>An automatic and end-to-end system for rare disease knowledge graph construction based on ontology-enhanced large language models: development study</article-title>
          <source>JMIR Med Inform</source>
          <year>2024</year>
          <volume>12</volume>
          <fpage>e60665</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2024//e60665/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/60665</pub-id>
          <pub-id pub-id-type="medline">39693482</pub-id>
          <pub-id pub-id-type="pii">v12i1e60665</pub-id>
          <pub-id pub-id-type="pmcid">PMC11683654</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grishman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Sundheim</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Message Understanding Conference-6: a brief history</article-title>
          <year>1996</year>
          <conf-name>Proceedings of the 16th International Conference on Computational Linguistics (COLING 1996), Volume 1</conf-name>
          <conf-date>August 5-9, 1996</conf-date>
          <conf-loc>Copenhagen, Denmark</conf-loc>
          <pub-id pub-id-type="doi">10.3115/992628.992709</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Levow</surname>
              <given-names>GA</given-names>
            </name>
          </person-group>
          <article-title>The third international Chinese language processing bakeoff: word segmentation and named entity recognition</article-title>
          <year>2006</year>
          <conf-name>Proceedings of the Fifth SIGHAN Workshop on Chinese Language Processing</conf-name>
          <conf-date>July 22-23, 2006</conf-date>
          <conf-loc>Sydney, Australia</conf-loc>
          <pub-id pub-id-type="doi">10.3115/1119250.1119269</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hettne</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Stierum</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Schuemie</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hendriksen</surname>
              <given-names>PJM</given-names>
            </name>
            <name name-style="western">
              <surname>Schijvenaars</surname>
              <given-names>BJA</given-names>
            </name>
            <name name-style="western">
              <surname>Mulligen</surname>
              <given-names>EMvan</given-names>
            </name>
            <name name-style="western">
              <surname>Kleinjans</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kors</surname>
              <given-names>JA</given-names>
            </name>
          </person-group>
          <article-title>A dictionary to identify small molecules and drugs in free text</article-title>
          <source>Bioinformatics</source>
          <year>2009</year>
          <volume>25</volume>
          <issue>22</issue>
          <fpage>2983</fpage>
          <lpage>2991</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/bioinformatics/article-lookup/doi/10.1093/bioinformatics/btp535"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btp535</pub-id>
          <pub-id pub-id-type="medline">19759196</pub-id>
          <pub-id pub-id-type="pii">btp535</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>YC</given-names>
            </name>
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>TK</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>YS</given-names>
            </name>
            <name name-style="western">
              <surname>Yen</surname>
              <given-names>SJ</given-names>
            </name>
          </person-group>
          <article-title>Extracting named entities using support vector machines</article-title>
          <source>Knowledge Discovery in Life Science Literature. Lecture Notes in Computer Science</source>
          <year>2006</year>
          <publisher-loc>Berlin, Germany</publisher-loc>
          <publisher-name>Springer</publisher-name>
          <fpage>91</fpage>
          <lpage>103</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Su</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Effective adaptation of a hidden Markov model-based named entity recognizer for biomedical domain</article-title>
          <year>2003</year>
          <conf-name>Proceedings of the ACL 2003 Workshop on Natural Language Processing in Biomedicine</conf-name>
          <conf-date>July 11, 2003</conf-date>
          <conf-loc>Sapporo, Japan</conf-loc>
          <fpage>49</fpage>
          <lpage>56</lpage>
          <pub-id pub-id-type="doi">10.3115/1118958.1118965</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leaman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>TaggerOne: joint named entity recognition and normalization with semi-Markov models</article-title>
          <source>Bioinformatics</source>
          <year>2016</year>
          <volume>32</volume>
          <issue>18</issue>
          <fpage>2839</fpage>
          <lpage>2846</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/27283952"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btw343</pub-id>
          <pub-id pub-id-type="medline">27283952</pub-id>
          <pub-id pub-id-type="pii">btw343</pub-id>
          <pub-id pub-id-type="pmcid">PMC5018376</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leaman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>BANNER: an executable survey of advances in biomedical named entity recognition</article-title>
          <source>Biocomputing 2008: Proceedings of the Pacific Symposium</source>
          <year>2008</year>
          <publisher-loc>Hackensack, NJ</publisher-loc>
          <publisher-name>World Scientific</publisher-name>
          <fpage>652</fpage>
          <lpage>663</lpage>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lample</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Ballesteros</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Subramanian</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kawakami</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Dyer</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Neural architectures for named entity recognition</article-title>
          <year>2016</year>
          <conf-name>Proceedings of the 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</conf-name>
          <conf-date>June 12-17, 2016</conf-date>
          <conf-loc>San Diego,</conf-loc>
          <fpage>260</fpage>
          <lpage>270</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/n16-1030</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Devlin</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Toutanova</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>BERT: Pre-training of deep bidirectional transformers for language understanding</article-title>
          <year>2019</year>
          <conf-name>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)</conf-name>
          <conf-date>June 2-7, 2019</conf-date>
          <conf-loc>Minneapolis, Minnesota</conf-loc>
          <fpage>4171</fpage>
          <lpage>4186</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/n18-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Yoon</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>So</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>
          <source>Bioinformatics</source>
          <year>2020</year>
          <volume>36</volume>
          <issue>4</issue>
          <fpage>1234</fpage>
          <lpage>1240</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31501885"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>
          <pub-id pub-id-type="medline">31501885</pub-id>
          <pub-id pub-id-type="pii">5566506</pub-id>
          <pub-id pub-id-type="pmcid">PMC7703786</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ghosh</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tyagi</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Kumar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Manocha</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>BioAug: Conditional generation based data augmentation for low-resource biomedical NER</article-title>
          <year>2023</year>
          <conf-name>Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval</conf-name>
          <conf-date>July 23-27, 2023</conf-date>
          <conf-loc>Taipei Taiwan</conf-loc>
          <fpage>1853</fpage>
          <lpage>1858</lpage>
          <pub-id pub-id-type="doi">10.1145/3539618.3591957</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Mu</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Jia</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Meng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Few-shot biomedical NER empowered by LLMs-assisted data augmentation and multi-scale feature extraction</article-title>
          <source>BioData Min</source>
          <year>2025</year>
          <volume>18</volume>
          <issue>1</issue>
          <fpage>28</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://biodatamining.biomedcentral.com/articles/10.1186/s13040-025-00443-y"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13040-025-00443-y</pub-id>
          <pub-id pub-id-type="medline">40181396</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13040-025-00443-y</pub-id>
          <pub-id pub-id-type="pmcid">PMC11969866</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liesting</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Frasincar</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Truşcă</surname>
              <given-names>MM</given-names>
            </name>
          </person-group>
          <article-title>Data augmentation in a hybrid approach for aspect-based sentiment analysis</article-title>
          <year>2021</year>
          <conf-name>Proceedings of the 36th Annual ACM/SIGAPP Symposium on Applied Computing</conf-name>
          <conf-date>March 22-26, 2021</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>828</fpage>
          <lpage>835</lpage>
          <pub-id pub-id-type="doi">10.1145/3412841.3441958</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>EDA: Easy data augmentation techniques for boosting performance on text classification tasks</article-title>
          <year>2019</year>
          <conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name>
          <conf-date>November 3-7, 2019</conf-date>
          <conf-loc>Hong Kong, China</conf-loc>
          <fpage>6382</fpage>
          <lpage>6388</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/d19-1670</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yoo</surname>
              <given-names>KM</given-names>
            </name>
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Data augmentation for spoken language understanding via joint variational generation</article-title>
          <source>Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence (AAAI-19)</source>
          <year>2019</year>
          <volume>33</volume>
          <issue>01</issue>
          <fpage>7402</fpage>
          <lpage>7409</lpage>
          <pub-id pub-id-type="doi">10.1609/aaai.v33i01.33017402</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Pham</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Neubig</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>SwitchOut: An efficient data augmentation algorithm for neural machine translation</article-title>
          <year>2018</year>
          <conf-name>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</conf-name>
          <conf-date>October 31-November 4, 2018</conf-date>
          <conf-loc>Brussels, Belgium</conf-loc>
          <fpage>856</fpage>
          <lpage>861</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/d18-1100</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yaseen</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Langer</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Data augmentation for low-resource named entity recognition using backtranslation</article-title>
          <year>2021</year>
          <conf-name>Proceedings of the 18th International Conference on Natural Language Processing (ICON)</conf-name>
          <conf-date>December 22-24, 2021</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>352</fpage>
          <lpage>358</lpage>
          <pub-id pub-id-type="doi">10.48550/arXiv.2108.11703</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Adel</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>An analysis of simple data augmentation for named entity recognition</article-title>
          <year>2020</year>
          <conf-name>Proceedings of the 28th International Conference on Computational Linguistics</conf-name>
          <conf-date>December 8-13, 2020</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>3861</fpage>
          <lpage>3867</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2020.coling-main.343</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Santoso</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Sutanto</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Cahyadi</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Setiawan</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Pushing the limits of low-resource NER using LLM artificial data generation</article-title>
          <year>2024</year>
          <conf-name>Findings of the Association for Computational Linguistics: ACL 2024</conf-name>
          <conf-date>August 11-16, 2024</conf-date>
          <conf-loc>Bangkok, Thailand</conf-loc>
          <fpage>9652</fpage>
          <lpage>9667</lpage>
          <pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.575</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>BHY</given-names>
            </name>
            <name name-style="western">
              <surname>Chau</surname>
              <given-names>JFT</given-names>
            </name>
            <name name-style="western">
              <surname>Wong</surname>
              <given-names>GK</given-names>
            </name>
          </person-group>
          <article-title>Rare versus common diseases: a false dichotomy in precision medicine</article-title>
          <source>NPJ Genom Med</source>
          <year>2021</year>
          <volume>6</volume>
          <issue>1</issue>
          <fpage>19</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41525-021-00176-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41525-021-00176-x</pub-id>
          <pub-id pub-id-type="medline">33627657</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41525-021-00176-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC7904920</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Whiting</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Rath</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Anido</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ardigò</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Baynam</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Dawkins</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Hamosh</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Le Cam</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Malherbe</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Molster</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Monaco</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Padilla</surname>
              <given-names>CD</given-names>
            </name>
            <name name-style="western">
              <surname>Pariser</surname>
              <given-names>AR</given-names>
            </name>
            <name name-style="western">
              <surname>Robinson</surname>
              <given-names>PN</given-names>
            </name>
            <name name-style="western">
              <surname>Rodwell</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Schaefer</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Weber</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Macchia</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Operational description of rare diseases: a reference to improve the recognition and visibility of rare diseases</article-title>
          <source>Orphanet J Rare Dis</source>
          <year>2024</year>
          <volume>19</volume>
          <issue>1</issue>
          <fpage>334</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://ojrd.biomedcentral.com/articles/10.1186/s13023-024-03322-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13023-024-03322-7</pub-id>
          <pub-id pub-id-type="medline">39261914</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13023-024-03322-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11389069</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nguengang Wakap</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lambert</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Olry</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rodwell</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gueydan</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lanneau</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Murphy</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Le Cam</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rath</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Estimating cumulative point prevalence of rare diseases: analysis of the Orphanet database</article-title>
          <source>Eur J Hum Genet</source>
          <year>2020</year>
          <volume>28</volume>
          <issue>2</issue>
          <fpage>165</fpage>
          <lpage>173</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31527858"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41431-019-0508-0</pub-id>
          <pub-id pub-id-type="medline">31527858</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41431-019-0508-0</pub-id>
          <pub-id pub-id-type="pmcid">PMC6974615</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Suárez-Paniagua</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Casey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Davidson</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Alex</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Whiteley</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Ontology-driven and weakly supervised rare disease identification from clinical notes</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2023</year>
          <volume>23</volume>
          <issue>1</issue>
          <fpage>86</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-023-02181-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-023-02181-9</pub-id>
          <pub-id pub-id-type="medline">37147628</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-023-02181-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC10162001</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Patra</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Ali</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Scordis</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>A hybrid framework with large language models for rare disease phenotyping</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2024</year>
          <volume>24</volume>
          <issue>1</issue>
          <fpage>289</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-024-02698-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-024-02698-7</pub-id>
          <pub-id pub-id-type="medline">39375687</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-024-02698-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11460004</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Banerjee</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Taroni</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Allaway</surname>
              <given-names>RJ</given-names>
            </name>
            <name name-style="western">
              <surname>Prasad</surname>
              <given-names>DV</given-names>
            </name>
            <name name-style="western">
              <surname>Guinney</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Greene</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Machine learning in rare disease</article-title>
          <source>Nat Methods</source>
          <year>2023</year>
          <volume>20</volume>
          <issue>6</issue>
          <fpage>803</fpage>
          <lpage>814</lpage>
          <pub-id pub-id-type="doi">10.1038/s41592-023-01886-z</pub-id>
          <pub-id pub-id-type="medline">37248386</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41592-023-01886-z</pub-id>
          <pub-id pub-id-type="pmcid">PMC13293574</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Martínez-deMiguel</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Segura-Bedmar</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Chacón-Solano</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Guerrero-Aspizua</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>The RareDis corpus: a corpus annotated with rare diseases, their signs and symptoms</article-title>
          <source>J Biomed Inform</source>
          <year>2022</year>
          <volume>125</volume>
          <fpage>103961</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(21)00290-2"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2021.103961</pub-id>
          <pub-id pub-id-type="medline">34879250</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(21)00290-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Doğan</surname>
              <given-names>RI</given-names>
            </name>
            <name name-style="western">
              <surname>Leaman</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>NCBI disease corpus: a resource for disease name recognition and concept normalization</article-title>
          <source>J Biomed Inform</source>
          <year>2014</year>
          <volume>47</volume>
          <fpage>1</fpage>
          <lpage>10</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(13)00197-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2013.12.006</pub-id>
          <pub-id pub-id-type="medline">24393765</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(13)00197-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC3951655</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dai</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Adel</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>data-augmentation-coling2020 software</article-title>
          <source>GitHub</source>
          <year>2020</year>
          <access-date>2026-03-20</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/boschresearch/data-augmentation-coling2020">https://github.com/boschresearch/data-augmentation-coling2020</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Kishore</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Weinberger</surname>
              <given-names>KQ</given-names>
            </name>
            <name name-style="western">
              <surname>Artzi</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>BERTScore: Evaluating text generation with BERT</article-title>
          <year>2020</year>
          <conf-name>International Conference on Learning Representations (ICLR)</conf-name>
          <conf-date>April 26-30, 2020</conf-date>
          <conf-loc>Online</conf-loc>
          <fpage>8</fpage>
          <lpage>10</lpage>
          <pub-id pub-id-type="doi">10.48550/arXiv.1904.09675</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Papineni</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Roukos</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ward</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>BLEU: a method for automatic evaluation of machine translation</article-title>
          <year>2002</year>
          <conf-name>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics</conf-name>
          <conf-date>July 7-12, 2002</conf-date>
          <conf-loc>Philadelphia</conf-loc>
          <fpage>311</fpage>
          <lpage>318</lpage>
          <pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="web">
          <article-title>Regulations on life science and medical research involving human subjects</article-title>
          <source>Nara Institute of Science and Technology</source>
          <access-date>2026-07-17</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://reiki.naist.jp/kiyaku/pdf/03135.pdf">https://reiki.naist.jp/kiyaku/pdf/03135.pdf</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref39">
        <label>39</label>
        <nlm-citation citation-type="web">
          <article-title>Persona_Driven_NER_DA</article-title>
          <source>GitHub</source>
          <access-date>2026-07-09</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/Jude-P24/Persona_Driven_NER_DA">https://github.com/Jude-P24/Persona_Driven_NER_DA</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
