<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e85088</article-id><article-id pub-id-type="doi">10.2196/85088</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Semantic Similarity Search Approach to Extract Exemplars of Stigmatizing and Positive Language in Obstetric Clinical Notes: Exploratory Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Scroggins</surname><given-names>Jihye Kim</given-names></name><degrees>PhD, RN</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hulchafo</surname><given-names>Ismael Ibrahim</given-names></name><degrees>MD, MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Barcelona</surname><given-names>Veronica</given-names></name><degrees>PhD, RN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Davoudi</surname><given-names>Anahita</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Moen</surname><given-names>Hans</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Harkins</surname><given-names>Sarah</given-names></name><degrees>PhD, RN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Scharp</surname><given-names>Danielle</given-names></name><degrees>PhD, APRN, FNP-BC</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cato</surname><given-names>Kenrick</given-names></name><degrees>PhD, RN, CPHIMS, FAAN</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tadiello</surname><given-names>Michele</given-names></name><degrees>MMCi, BSN</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Topaz</surname><given-names>Maxim</given-names></name><degrees>PhD, RN</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>School of Nursing, University of North Carolina at Chapel Hill</institution><addr-line>120 Medical Drive</addr-line><addr-line>Chapel Hill</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff2"><institution>School of Nursing, Columbia University</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff3"><institution>VNS Health</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Computer Science, Aalto University</institution><addr-line>Espoo</addr-line><country>Finland</country></aff><aff id="aff5"><institution>Icahn School of Medicine at Mount Sinai</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff6"><institution>School of Nursing, University of Pennsylvania</institution><addr-line>Philadelphia</addr-line><addr-line>PA</addr-line><country>United States</country></aff><aff id="aff7"><institution>Center for Community-Engaged Health Informatics and Data Science, Columbia University Irving Medical Center</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Quansah</surname><given-names>Ama</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chrimes</surname><given-names>Dillon</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Al-Agil</surname><given-names>Mohammad</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jihye Kim Scroggins, PhD, RN, School of Nursing, University of North Carolina at Chapel Hill, 120 Medical Drive, Chapel Hill, NC, 27514, United States, 1 (919) 966-4260; <email>jihyeks@unc.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>22</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e85088</elocation-id><history><date date-type="received"><day>30</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>13</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>13</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jihye Kim Scroggins, Ismael Ibrahim Hulchafo, Veronica Barcelona, Anahita Davoudi, Hans Moen, Sarah Harkins, Danielle Scharp, Kenrick Cato, Michele Tadiello, Maxim Topaz. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 22.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e85088"/><abstract><sec><title>Background</title><p>Natural language processing can extract meaningful information from clinical notes. However, human annotation is time-consuming and costly, and scarce data poses a challenge.</p></sec><sec><title>Objective</title><p>This study aimed to explore a semantic similarity search approach to extract exemplars of stigmatizing and positive language in obstetric clinical notes.</p></sec><sec sec-type="methods"><title>Methods</title><p>We used electronic health record data from labor and birth admissions at 2 hospitals in the United States from 2017 to 2019. We used a semantic similarity search approach, which used 200 randomly selected true exemplars, stratified by language categories, as queries to search for similar exemplar candidates. We extracted the top 5 candidates with the highest cosine similarities, which were assessed for accuracy.</p></sec><sec sec-type="results"><title>Results</title><p>We retrieved 1000 candidates. An average precision of 0.69 was achieved when candidates with cosine similarity thresholds of 0.75 or higher were included, at which point 68.8% (64/93) of exemplar candidates accurately represented true cases. At the 0.75 threshold, the proportion of true cases was higher for preferred language (41/56, 73.2%) and unilateral/authoritarian decisions (5/7, 71.4%). The proportion of true cases was lower for difficult patients (2/7, 28.6%) and marginalized identities (3/9, 33.3%).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The semantic similarity search approach shows promise in efficiently extracting exemplars while reducing the annotation burden, laying the groundwork for future applications in other domains.</p></sec></abstract><kwd-group><kwd>natural language processing</kwd><kwd>electronic health records</kwd><kwd>stigmatizing language</kwd><kwd>health communication</kwd><kwd>bias</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Natural language processing (NLP) can efficiently and accurately extract meaningful information from large text data, such as clinical notes within electronic health records (EHRs) [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. One significant challenge is obtaining sufficient human-annotated data to train NLP models for specific tasks and datasets. Human annotation is labor-intensive and costly, requiring extensive effort to review and label data for exemplars representing specific concepts or categories within the text. For example, annotating 50 discharge summaries can take approximately 80 hours for a team of several clinical experts; this time grows exponentially for datasets with a low prevalence of the target information [<xref ref-type="bibr" rid="ref3">3</xref>]. Given the substantial time and manpower needed for human annotation, it is common to encounter limited human-labeled data for NLP model development [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Prior NLP research has focused on reducing annotation burden through approaches such as active learning and data augmentation. Traditional active learning, most commonly using uncertainty-based querying, supports annotation by iteratively selecting unlabeled cases for which the model has the highest uncertainty (lowest confidence) and prioritizing these cases for human review [<xref ref-type="bibr" rid="ref5">5</xref>]. However, uncertainty-based querying may be less effective for low-prevalence or rare concepts, which can be under-selected when models exhibit high confidence in assigning majority-class labels; for example, rare positive cases may be confidently misclassified as negative and therefore not selected for human review [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Data augmentation, including synonym replacement, paraphrasing, and more recently, synthetic data generation using large language models, has also been used to expand training datasets [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. However, these approaches may unintentionally introduce noise or alter subtle linguistic or contextual cues essential for capturing nuanced context-dependent concepts [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. For example, human evaluation of synthetic data quality in the context of stigmatizing language has shown that synthetic exemplars exhibited limited clinical realism [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Semantic similarity search offers a complementary strategy for expanding training data by efficiently extracting additional exemplars that are semantically similar to existing human-annotated exemplars. Semantic similarity refers to the degree to which 2 pieces of text share similar meanings based on their linguistic content and context [<xref ref-type="bibr" rid="ref13">13</xref>]. Semantic similarity has been widely studied. Early approaches focused on document-level similarity based on lexical and distributional overlap [<xref ref-type="bibr" rid="ref14">14</xref>]. More recent work has focused on sentence-level similarity using neural embeddings, which enable more nuanced comparisons across short text spans [<xref ref-type="bibr" rid="ref15">15</xref>]. In the clinical domain, prior work has largely focused on developing and evaluating models for estimating similarity between pairs of clinical sentences [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. These studies typically use trained models to predict similarity scores between sentence pairs and evaluate how closely these predictions align with human ratings [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Semantic similarity has also been used in information retrieval and retrieval-augmented generation systems to retrieve clinically relevant texts, passages, or documents [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. These approaches are primarily used for downstream tasks such as question answering and clinical decision support [<xref ref-type="bibr" rid="ref20">20</xref>]. In contrast, the use of semantic similarity search to support annotation workflows, specifically for retrieving additional exemplars, remains relatively limited.</p><p>In the current study, we applied semantic similarity search to support annotation by using human-annotated true exemplars as queries to retrieve additional exemplar candidates. This approach has the potential to reduce the human labor associated with reviewing and labeling new text data from scratch. Unlike traditional active learning, this approach directly retrieves semantically similar exemplar candidates to human-annotated true exemplars, without relying on iterative model training or uncertainty-based querying. This strategy may be particularly useful for identifying rare concepts, as it can retrieve semantically similar text even when such cases are infrequent, allowing expansion of sparse categories. In addition, by using true exemplars as queries, this approach can preserve real-world language and subtle contextual features that may not be well captured by synthetic data.</p><p>Specifically, we explored the application of this approach in the context of identifying stigmatizing and positive language categories in obstetric clinical notes. Stigmatizing language can convey implicit or explicit bias [<xref ref-type="bibr" rid="ref21">21</xref>], which can negatively influence patient-clinician relationships and satisfaction in health care [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Identifying and mitigating stigmatizing language is essential for providing respectful and unbiased care for perinatal populations affected by significant health disparities [<xref ref-type="bibr" rid="ref24">24</xref>]. The prevalence of stigmatizing language can be as low as 1% to 2% for certain language categories or patient populations [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. This presents challenges for manual annotation and obtaining sufficient data for optimal NLP model training. Additionally, identifying positive language that respects patients&#x2019; views and autonomy is important to promote the use of strength-based language in clinical documentation [<xref ref-type="bibr" rid="ref27">27</xref>]. Our efforts to identify stigmatizing and positive language revealed the need for an efficient approach to extract additional exemplars with limited resources [<xref ref-type="bibr" rid="ref28">28</xref>]. This study aimed to evaluate semantic similarity search as a more efficient approach for identifying additional exemplars of stigmatizing and positive language in obstetric clinical notes. We assessed retrieval accuracy through human evaluation and precision metrics.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Data</title><p>We used EHR data from patients at &#x003E;20 weeks&#x2019; gestation admitted for labor and birth at 2 urban hospitals in the Northeast United States from 2017 to 2019. All clinical notes in the EHR during the inpatient hospital stay were eligible. Note types with limited clinician narrative text describing patient assessments or impressions, such as medication orders, transfer notes, and template-based statements about procedures or operations, were excluded. Seven note types were used in the current study: obstetric postpartum note, obstetric admission note, obstetric triage note, anesthesia resident note, miscellaneous nursing note, social work initial assessment, and initial nutrition assessment.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>We received institutional review board (IRB) approval from Columbia University Medical Center (AAAT9870) for this study under expedited review, category 5: research involving materials (data, documents, records, or specimens) that have been collected or will be collected solely for nonresearch purposes, such as medical treatment or diagnosis. A waiver of informed consent was granted by the IRB because the research involved no more than minimal risk, did not adversely affect the rights and welfare of human subjects, could not be carried out without the waiver, and the study could not be completed without using identifiable private information from patients discharged from the study hospitals. To protect privacy and confidentiality, study data were stored on a secure server following institutional data security guidelines, with access restricted to authorized study personnel with a direct need for data management and analysis. All study personnel completed IRB-required training in the responsible conduct of research and protection of human subjects. No compensation was provided because this study involved secondary use of data without direct participant involvement, interaction, or recruitment.</p></sec><sec id="s2-3"><title>Semantic Similarity Search Approach</title><sec id="s2-3-1"><title>Approach Overview</title><p>We explored using semantic similarity search to expand the initial human-annotated dataset in a less manually intensive way. This approach uses initial human-annotated exemplars (ie, <italic>true exemplars</italic>) as queries to search for similar exemplars (ie, <italic>exemplar candidates</italic>). Then, human experts can perform a relatively faster review of the exemplar candidates to determine the accuracy.</p></sec><sec id="s2-3-2"><title>Language Categories</title><p>Stigmatizing and positive language categories are presented in <xref ref-type="table" rid="table1">Table 1</xref>. The categories and operational definitions were developed through iterative qualitative analysis [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref29">29</xref>] and informed by prior work on stigmatizing language [<xref ref-type="bibr" rid="ref30">30</xref>]. A multidisciplinary team of experts, including obstetrics and gynecology physicians, nurses, and nonclinical researchers, contributed diverse perspectives to the development of these categories. These efforts were intended to enhance the rigor of the operational definitions.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Stigmatizing and positive language categories.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Language category</td><td align="left" valign="bottom">Definition</td><td align="left" valign="bottom">Examples</td><td align="left" valign="bottom">True exemplars (N=200), n</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Stigmatizing language</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Difficult patient</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Nonadherence or noncompliance with plan of care, refusal of care, referrals, or services.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;missed nutrition appt again says she doesn&#x2019;t want to go.&#x201D;</p></list-item><list-item><p>&#x201C;exam unchanged however now c/o [complains of] ctx [contraction] pain.&#x201D;</p></list-item><list-item><p>&#x201C;poor compliance with antenatal visits.&#x201D;</p></list-item></list></td><td align="left" valign="top">28</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unilateral/ authoritarian decisions</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Language that supports clinician&#x2019;s authority over patients. Upholds hierarchy centering clinician, not patient.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;cervix unfavorable therefore will start induction with Cytotec.&#x201D;</p></list-item><list-item><p>&#x201C;SW [social worker] has advised pt [patient] that if there continues to be yelling in room, ACS [adult and child services] will need to be contacted.&#x201D;</p></list-item><list-item><p>&#x201C;also very upset about persistent shaking in her UE [upper extremities]. Explained to patient multiple times that shaking is normal and a result of hormones and the anesthesia.&#x201D;</p></list-item></list></td><td align="left" valign="top">14</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Power/privilege</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Describes power and privilege identities related to psychological or social-ecological status.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;pt [patient] reports having a nurturing marriage with FOB [father of the baby] who works as a Lawyer.&#x201D;</p></list-item><list-item><p>&#x201C;private pt [patient] of mine at 39+6 wks [weeks] with multiple episodes of emesis this AM.&#x201D;</p></list-item><list-item><p>&#x201C;Caucasian female. pleasantly engaged in conversation.&#x201D;</p></list-item></list></td><td align="left" valign="top">17</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Marginalized identities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Documentation of social and behavioral risk factors in narrative that could contribute to marginalization.</p></list-item><list-item><p>Restating for emphasis that is already in the checklist/form data or unnecessary patient descriptor (eg, &#x201C;toxic habits,&#x201D; &#x201C;financially supports self,&#x201D; &#x201C;teen mother,&#x201D; &#x201C;late registrant,&#x201D; and &#x201C;obesity&#x201D;).</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;social disarray, estranged from family with FOB [father of the baby] not involved.&#x201D;</p></list-item><list-item><p>&#x201C;Dominican female. Initial psychosocial assessment due to teen pregnancy late registrant.&#x201D;</p></list-item><list-item><p>&#x201C;Patient is a 32yo [year old] Dominican unmarried unemployed female.&#x201D;</p></list-item></list></td><td align="left" valign="top">74</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Questioning patient credibility</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Disbelief in patient&#x2019;s report of health and social history or status.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;unsure if patient telling the truth.&#x201D;</p></list-item><list-item><p>&#x201C;pt [patient] denied any other DV [domestic violence] incidents and was adamant that relationship with spouse was healthy.&#x201D;</p></list-item><list-item><p>&#x201C;Pt [patient] does not think she has gestational diabetes she states she ate very sweet rice (a dish from her country) the day before the glucose challenge which is what she believes this is the reason for the high value states it has never been high before.&#x201D;</p></list-item></list></td><td align="left" valign="top">11</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Disapproval</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Behaviors of patient not in alignment with health care provider expectations.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;FOB [father of baby] is 23 yo [year old] unemployed and not as involved as he should be.&#x201D;</p></list-item><list-item><p>&#x201C;Pt [patient] states she tried to obtain f/u [follow-up] appt [appointment]. but did not get a call back. Advised she should have gone to the clinic to straighten things out- in the future to take initiative-importance stressed as well as appts [appointments] for her newborn.&#x201D;</p></list-item><list-item><p>&#x201C;PP [postpartum] BCM [bridge contraceptive method]- pt states she prefers to use condoms will continue to readdress.&#x201D;</p></list-item></list></td><td align="left" valign="top">6</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Structural/interprofessional hierarchy</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Notes stating issues in care are attributed to another clinician, often nurses. Notes stating that care was not delivered in a timely manner due to staffing or other structural issues.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;Nurse states IV [intravenous] &#x2018;infiltrated,&#x2019; but it was fine.&#x201D;</p></list-item><list-item><p>&#x201C;Unable to staff nursing for placement of epidural in triage. Requested that charge nurse please inform anesthesiology team about when there is enough nursing staff.&#x201D;</p></list-item><list-item><p>&#x201C;Patient continues to wait in triage until L D [Labor &#x0026; Delivery] bed available.&#x201D;</p></list-item></list></td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top" colspan="4">Positive language</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Preferred language</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Preferred words that can convey patient&#x2019;s point of view respectively and objectively (eg, endorses, reports, or states).</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;patient desires to ambulate and encourage natural labor.&#x201D;</p></list-item><list-item><p>&#x201C;patient states she feels some pain on right side.&#x201D;</p></list-item><list-item><p>&#x201C;states she has had irregular ctx [contractions] starting 24 hours ago. endorses FM [fetal movement]. Endorses N/V [nausea/vomiting]&#x201D;</p></list-item></list></td><td align="left" valign="top">44</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Autonomy for birth</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Notes that indicate patient exercising autonomy around birth and plan of care.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;Pt [patient] declines epidural at this time- trying to deliver without analgesia.&#x201D;</p></list-item><list-item><p>&#x201C;will give patient the option to have continuous monitoring overnight and give her the opportunity to go into spontaneous labor with plans for c-section tomorrow if she does not progress.&#x201D;</p></list-item><list-item><p>&#x201C;low risk screening declined diagnostic. She was given the option of admission with possible augmentation of labor vs discharge home to labor. Plan was made for admission.&#x201D;</p></list-item></list></td><td align="left" valign="top">4</td></tr></tbody></table></table-wrap></sec><sec id="s2-3-3"><title>Initial Human Annotation</title><p>To generate the initial human-annotated dataset, 4 researchers with expertise in qualitative research and/or nursing independently annotated clinical notes following an established codebook (see Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The codebook was developed through iterative inductive-deductive content analysis [<xref ref-type="bibr" rid="ref26">26</xref>]. All annotators were trained prior to participating in annotations, which included review of the codebook, discussion of example phrases and sentences, and coding demonstrations. Annotators manually labeled true exemplars from a total of 1771 clinical notes, which typically spanned 1 to 3 sentences. To ensure reliability, 2 annotators reviewed and annotated the same clinical notes. We resolved disagreements through iterative discussions among the annotators. The initial agreement among the annotators across the 1771 clinical notes was fair (Cohen &#x03BA;=0.4, agreement rate=72%), which reflects the inherently nuanced nature of language use and subjectivity involved with interpreting complex and nuanced language [<xref ref-type="bibr" rid="ref31">31</xref>]. We spent extensive effort and time discussing any discrepancies among annotators to reach a consensus and ensure the quality and accuracy of the final annotated dataset [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec><sec id="s2-3-4"><title>Sentence-Transformer Model</title><p>We used a sentence-transformer model, &#x201C;multi-qa-distilbert-cos-v1,&#x201D; from the Hugging Face Model Hub [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref32">32</xref>] to search for additional exemplar candidates that were semantically similar to the set of true exemplars. This model is specifically designed for semantic similarity search and retrieval tasks using a contrastive learning objective to generate sentence-level embeddings that enable comparison of a given sentence with other sentences that are most closely related in meaning [<xref ref-type="bibr" rid="ref32">32</xref>]. This design enables efficient, out-of-the-box application of the model for similarity-based retrieval tasks, making it well suited for the current task. Although this model is not trained on clinical text, it is trained on large-scale, question-answer-style data from diverse sources [<xref ref-type="bibr" rid="ref32">32</xref>], which contributes to capturing contextual meaning across varied linguistic expressions. This general-domain training may be helpful for identifying broader sociolinguistic patterns and contextual features in clinical text that can extend beyond clinical language. Prior work suggests that general-domain embedding models can perform well for semantic similarity tasks in clinical text, although performance may vary across specific models and tasks [<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>Exemplar candidates were selected from clinical notes not used in the initial human annotation. We preprocessed the clinical notes by converting the free-text portions into a format similar to the true exemplars (single, double, and triple consecutive sentences). Sentence tokenization was performed first using natural language toolkit&#x2019;s &#x201C;sent_tokenize,&#x201D; which identifies sentence boundaries based on learned punctuation and linguistic patterns rather than simple rule-based splitting. After tokenization, sliding windows of 1-, 2-, and 3-sentence spans were generated by advancing one sentence at a time. Duplicate candidates were then detected using a 6-word prefix and removed in 2 stages: first keeping only the earliest instance of each prefix, and then alternately dropping any remaining repeats to ensure only 1 representative remained. No normalization of casing, numbers, or abbreviations was applied. All text was embedded in its original form to preserve real-world clinical text. Examples of synthetic, free-text note snippets can be found in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>We used the sentence-transformer model to encode true exemplars into fixed-size vectorized embeddings [<xref ref-type="bibr" rid="ref15">15</xref>]. Then, true exemplars were used as queries to search and recommend new exemplars that were likely candidates from the unused clinical note datasets. We calculated semantic similarity between the true exemplars and exemplar candidates using the cosine similarity applied to their associated embedding representations. The Facebook AI similarity search library was used to perform the similarity searches [<xref ref-type="bibr" rid="ref34">34</xref>]. We stratified and randomly selected 200 true exemplars across language categories to be used as queries. Stratification was conducted by language category to preserve each category&#x2019;s natural prevalence in the initial human-annotated dataset. As a result, less frequent categories contributed fewer exemplars (<xref ref-type="table" rid="table1">Table 1</xref>). For each of the 200 true exemplars, the top 5 similar exemplar candidates with the highest cosine similarities were retrieved (N=1000 exemplar candidates) for human review. Pseudocode blocks for the semantic similarity search pipeline employed in this study are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec></sec><sec id="s2-4"><title>Examining Accuracy of Exemplar Candidates</title><p>Four researchers who performed the initial annotation independently reviewed and assessed the exemplar candidates. Reviewers labeled the exemplar candidate as a &#x201C;match&#x201D; if it accurately represented the corresponding language category and as &#x201C;not a match&#x201D; if it did not. We resolved disagreements through iterative discussions among reviewers to reach consensus. We calculated the frequency and percentage of matched and nonmatched cases as true positives and false positives. We examined how precision changes when different cosine similarity boundaries were used to inform optimizing this approach for future research and applications. We calculated and reported average precision and corresponding 95% Wilson CIs across different cosine similarity thresholds, ranging from 0.5 to 0.95 in increments of 0.05. To our knowledge, there is no universally accepted precision cutoff for information retrieval tasks to support annotation workflows. Information retrieval literature supports that performance should be evaluated in relation to the intended use and context of each task, as performance requirements may vary across applications [<xref ref-type="bibr" rid="ref35">35</xref>]. Previous research on annotation support often focuses on reducing annotation burden while maintaining annotation quality [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. In a recent study on automated annotation support, the authors noted that precision exceeding 0.7 was considered &#x201C;strong performance&#x201D; for their task [<xref ref-type="bibr" rid="ref38">38</xref>]. Given the exploratory nature of this study and its intended use to support annotation workflows, we selected precision &#x2265;0.70 as an acceptable, pragmatic point to balance accuracy with meaningful efficiency gains in reducing annotation burden. At this precision point, most exemplar candidates would represent true cases while allowing a manageable proportion of false positives.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>A total of 1000 exemplar candidates were retrieved using the semantic similarity search approach (<xref ref-type="fig" rid="figure1">Figure 1</xref>). These exemplar candidates were derived from 443 clinical notes of 403 patients. The number of exemplar candidates by note type is provided in <xref ref-type="table" rid="table2">Table 2</xref>. Among the 1000 exemplar candidates retrieved, most were derived from obstetric admission notes (n=295, 29.5%), followed by obstetric postpartum notes (n=202, 20.2%) and miscellaneous nursing notes (n=172, 17.2%). Human annotators independently reviewed all 1000 exemplar candidates retrieved for accuracy. The interrater reliability among the annotators was good (Cohen &#x03BA;=0.71, 95% CI 0.67&#x2010;0.75) across 1000 exemplar candidates.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow diagram summarizing the number of true exemplars or exemplar candidates in the analysis and results.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e85088_fig01.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Number of exemplar candidates by note type.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Note type</td><td align="left" valign="bottom">Exemplar candidates<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Obstetric admission note</td><td align="left" valign="top">295 (29.5)</td></tr><tr><td align="left" valign="top">Obstetric postpartum note</td><td align="left" valign="top">202 (20.2)</td></tr><tr><td align="left" valign="top">Miscellaneous nursing note</td><td align="left" valign="top">172 (17.2)</td></tr><tr><td align="left" valign="top">Anesthesia resident note</td><td align="left" valign="top">137 (13.7)</td></tr><tr><td align="left" valign="top">Obstetric triage note</td><td align="left" valign="top">106 (10.6)</td></tr><tr><td align="left" valign="top">Social work initial assessment</td><td align="left" valign="top">51 (5.1)</td></tr><tr><td align="left" valign="top">Initial nutrition assessment</td><td align="left" valign="top">37 (3.7)</td></tr><tr><td align="left" valign="top">Total exemplar candidates retrieved</td><td align="left" valign="top">1000 (100)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>The total number of exemplar candidates retrieved was used as the denominator to calculate the percentages.</p></fn></table-wrap-foot></table-wrap><p>We examined how the average precision of all language categories changed when including exemplar candidates at different cosine similarity thresholds. As shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>, average precision increased with higher cosine similarity thresholds. The number of exemplar candidates retained decreased with higher cosine similarity thresholds. Average precision was highest at 1 (95% CI 0.44&#x2010;1.00) when including exemplar candidates with cosine similarity &#x2265;0.95, at which point 3 of 3 exemplar candidates were true cases. Average precision was lowest at 0.46 (95% CI 0.43&#x2010;0.49) when including exemplar candidates with cosine similarity &#x2265;0.5, at which point 383 of 831 exemplar candidates were true cases. Acceptable average precision (&#x2265;0.70) was achieved when including exemplar candidates with a cosine similarity of at least 0.75.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Line graph illustrating the average precision at different cosine similarity thresholds with corresponding values for the total number of exemplar candidates retained at each threshold, the number of true positives, and 95% CIs. The total number of exemplar candidates retained at each threshold was used as the denominator to calculate the precision.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e85088_fig02.png"/></fig><p>A total of 93 exemplar candidates had a cosine similarity of at least 0.75. Among those, 64 (68.8%) accurately represented the true cases across stigmatizing and positive language categories (<xref ref-type="table" rid="table3">Table 3</xref>). In specific language categories, 100% (5/5) of exemplar candidates for autonomy for birth, 73.2% (41/56) for preferred language, and 71.4% (5/7) for unilateral/authoritarian decisions represented the true cases. While the proportions of matched cases were high for power/privilege, questioning patient credibility, and disapproval, the number of retrieved exemplar candidates for these categories was small at the 0.75 cosine similarity threshold (eg, 1 of 1 exemplar candidate was a true case for power/privilege and disapproval, respectively). In contrast, 28.6% (2/7) of exemplar candidates for difficult patient and 33.3% (3/9) of exemplar candidates for marginalized identities represented the true cases, with lower proportions of matched cases than other language categories at the 0.75 cosine similarity threshold.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Proportion of matched cases by language category at a cosine similarity threshold of 0.75.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Language category</td><td align="left" valign="bottom">Exemplar candidates, N<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">True positives (matched cases), n (%)</td><td align="left" valign="bottom">False positives (nonmatched cases), n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Preferred language</td><td align="left" valign="top">56</td><td align="left" valign="top">41 (73.2)</td><td align="left" valign="top">15 (26.8)</td></tr><tr><td align="left" valign="top">Autonomy for birth</td><td align="left" valign="top">5</td><td align="left" valign="top">5 (100)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Difficult patient</td><td align="left" valign="top">7</td><td align="left" valign="top">2 (28.6)</td><td align="left" valign="top">5 (71.4)</td></tr><tr><td align="left" valign="top">Unilateral/authoritarian decisions</td><td align="left" valign="top">7</td><td align="left" valign="top">5 (71.4)</td><td align="left" valign="top">2 (28.6)</td></tr><tr><td align="left" valign="top">Power/privilege</td><td align="left" valign="top">1</td><td align="left" valign="top">1 (100)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Structural/interprofessional hierarchy</td><td align="left" valign="top">4</td><td align="left" valign="top">3 (75)</td><td align="left" valign="top">1 (25)</td></tr><tr><td align="left" valign="top">Marginalized identities</td><td align="left" valign="top">9</td><td align="left" valign="top">3 (33.3)</td><td align="left" valign="top">6 (66.7)</td></tr><tr><td align="left" valign="top">Questioning patient credibility</td><td align="left" valign="top">3</td><td align="left" valign="top">3 (100)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Disapproval</td><td align="left" valign="top">1</td><td align="left" valign="top">1 (100)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">Total</td><td align="left" valign="top">93</td><td align="left" valign="top">64 (68.8)</td><td align="left" valign="top">29 (31.2)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>The total N column was used as the denominator to calculate the proportion of matched and nonmatched cases.</p></fn></table-wrap-foot></table-wrap><p>Detailed results for precision at different cosine similarity thresholds by specific language categories and corresponding CIs are reported in Tables S1 and S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Of note, several precision values were based on small numbers of exemplar candidates, particularly at higher thresholds, and are statistically unstable for these sparse categories. For categories with relatively larger numbers of retained candidates, such as preferred language, acceptable precision (&#x2265;0.70) was observed at a cosine similarity of 0.70, where 74 of 105 exemplar candidates represented true cases. False positives most commonly occurred in exemplars containing more complex social or contextual information. For example, exemplars describing objective social circumstances were misclassified as stigmatizing despite the absence of judgmental language (eg, false positive in the marginalized identities category: &#x201C;Patient may want to be discharged to home tonight. She lives nearby and will be able to visit infant&#x201D;). In other cases, exemplars containing stigmatizing language were misclassified into incorrect categories. For example, &#x201C;Patient stated she doesn&#x2019;t have any help in the country and FOB [father of baby] is in DR [Dominican Republic]. Patient needs social worker, will continue to monitor...&#x201D; was incorrectly classified into the power/privilege category.</p></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>We explored a semantic similarity search approach to identify additional exemplars from obstetric clinical notes with less manual human effort. We found that we can increase accuracy by adjusting cosine similarity boundaries when selecting exemplar candidates. The average precision increased with higher cosine similarity thresholds, achieving an acceptable average precision &#x2265;0.70 when exemplar candidates with a cosine similarity of 0.75 or higher were included. On average, 68.8% (64/93) of exemplar candidates accurately represented true cases across various language categories at the 0.75 cosine similarity threshold. These thresholds should be guided by each use case in future research and application. Lower thresholds may be useful for exploratory analyses, where broader retrieval is desired to examine patterns of expressions and language use. Lower thresholds may also be appropriate when maximum coverage is prioritized over precision, such as for rarely occurring cases, where missing potential matches would be more consequential than reviewing additional false positives. In contrast, higher thresholds may be suitable when precision should be prioritized, such as using retrieved texts for direct application in clinical settings, including quality review or clinical documentation audit to identify potential stigmatizing language. Higher thresholds may also be useful with large datasets, where even a strict precision cutoff can yield a sufficient number of exemplar candidates, whereas the same threshold applied to smaller datasets may result in too few retrieved exemplars to be useful.</p><p>It is important to note that the reported accuracy in our study reflects only positive retrieval quality, measured by precision, rather than the overall effectiveness or completeness of the retrieval approach. Given the exploratory nature of the study, we did not annotate nonretrieved texts to calculate more comprehensive evaluation metrics, including recall and <italic>F</italic><sub>1</sub>-scores. As such, our findings do not provide information regarding the extent to which the approach identified all relevant exemplars within the dataset. For example, it remains unknown how many relevant exemplars may not have been retrieved, limiting the ability to assess the completeness of retrieval. In addition, the number of exemplar candidates retained decreased substantially across language categories as cosine similarity thresholds increased (detailed category-specific results in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). As such, the current study may be underpowered to support reliable category-level conclusions. When few exemplar candidates were retained, high precision values should not be misinterpreted as strong performance, as they are statistically unstable and reflect limited sample size. At lower cosine thresholds, where more candidates were retained, precision varied across language categories. Although these should be interpreted cautiously, they may reflect different language complexity. For example, at the 0.5 cosine similarity threshold, higher precision was observed for preferred language and autonomy for birth compared to other categories. These 2 categories were more straightforward with well-defined, identifiable keywords (eg, &#x201C;desires natural birth,&#x201D; &#x201C;reports pain&#x201D;). Conversely, we observed lower precision for marginalized identities, questioning patient credibility, and disapproval categories at the 0.5 cosine similarity threshold. These categories were more nuanced and context-dependent. For example, there was a subtle undertone of disapproving the patient&#x2019;s choice of birth control in the following: &#x201C;postpartum birth control method-patient states that she prefers to use condoms and will continue to readdress.&#x201D;</p><p>Accurately identifying such nuances requires a deeper understanding and interpretation of contextual meanings and subtleties, which may be challenging for semantic similarity search. Semantic similarity search primarily relies on vector-based representations of text, including static embeddings like Word2Vec and GloVe [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref39">39</xref>] and contextual embeddings like bidirectional encoder representations from transformers (BERT) [<xref ref-type="bibr" rid="ref40">40</xref>]. Static embeddings can effectively capture general semantic relationships [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref39">39</xref>] but can be limited in detecting context-specific nuances and meanings [<xref ref-type="bibr" rid="ref41">41</xref>]. Although contextual embeddings, such as those generated by the sentence-transformer model used in the current study, can better capture contextual information, they may still be challenged by complex and subtle nuances [<xref ref-type="bibr" rid="ref42">42</xref>]. Finally, when suitable training data are available, sentence-transformer models could be further improved through data- and task-specific training by fine-tuning.</p><p>While semantic similarity has been previously used for tasks such as information retrieval, its application to support annotation workflows, specifically for retrieving additional exemplars, remains relatively limited. The current study extends this line of work by applying semantic similarity search to more efficiently expand annotated datasets. This approach is particularly useful for low-prevalence concepts, where traditional active learning strategies may be less effective. In contrast to data augmentation strategies, such as synthetic data generation, it can preserve naturally occurring language while enabling targeted expansion of context-dependent concepts. Furthermore, while prior retrieval-based approaches have used similarity to identify relevant text for tasks such as information retrieval or retrieval-augmented generation, the effectiveness of these approaches is typically evaluated based on automated metrics or downstream model performance, rather than human validation [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. In contrast, we conducted human evaluation of retrieved exemplar candidates and examined precision across cosine similarity thresholds, providing insight into the utility of this approach as an annotation support strategy. Our findings highlight how cosine similarity thresholds influence precision, offering practical guidance for applying this approach in settings where annotated data are sparse.</p><p>Although our findings are based on the context of stigmatizing and positive language in obstetric clinical notes, which may limit the generalizability to other specialized datasets or domains, semantic similarity search itself is not inherently domain-specific. Because this approach operates by comparing a given sentence with other sentences to identify those most closely related in meaning, it is not dependent on or limited to specific models but can be incorporated into existing clinical NLP pipelines as a data augmentation or annotation-support step. For example, retrieved exemplar candidates may be used to efficiently expand training datasets prior to NLP model development or to support targeted human review in resource-constrained settings. Prior research has demonstrated the practical benefits of targeted text selection using active learning to support annotation workflows. For example, active learning approaches have been shown to reduce annotation time by approximately 20% to 35% for clinically explicit concepts, such as medical problems and tests, by prioritizing informative texts for manual annotation [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. More recently, active learning has also been shown to reduce other forms of annotation burden. For example, Nachtegael et al [<xref ref-type="bibr" rid="ref45">45</xref>] used active learning to select unlabeled texts for manual annotation. They found that models trained on selectively annotated datasets achieved performance comparable to models trained on fully labeled datasets while reducing the amount of labeled data required by 6% to 38% [<xref ref-type="bibr" rid="ref45">45</xref>]. Although these studies employed different active learning approaches than the current study, they indicate that strategically selecting texts for manual review can reduce annotation burden.</p></sec><sec id="s4-2"><title>Limitations</title><p>This study has some limitations. First, the effectiveness of semantic similarity search depends on the quality of the initial human-annotated dataset. If true exemplars do not accurately reflect positive cases, the extracted candidates are also likely to be inaccurate. The relatively modest initial agreement among annotators (Cohen &#x03BA;=0.4) highlights the subjective nature of identifying stigmatizing language. Consequently, some retrieved exemplars may have been classified differently among annotators, which could have influenced downstream retrieval accuracy. Additionally, if true exemplars do not represent diverse language categories, the extracted candidates may not be comprehensive. Second, while semantic similarity search reduced the labor required for annotation, it does not eliminate the need for human review. Due to the risk of false positives, a rapid human review of the retrieved candidates remains essential to validate the accuracy. Third, our evaluation was limited to precision to prioritize annotation efficiency. In our approach, the model retrieved exemplars predicted to be positive matches to the true exemplars. We did not annotate nonretrieved texts to identify false negatives, which is required to calculate recall. As such, the expanded dataset should be useful for augmenting existing annotated datasets with positive cases, rather than providing exhaustive or population-representative datasets. Fourth, although language categories and operational definitions were developed through qualitative analysis and informed by prior research, several categories remain inherently subjective and context-dependent. The operational definitions for these categories may remain somewhat ambiguous in practice, which can introduce variability in interpretation even among trained annotators. This reflects broader challenges in objectively operationalizing stigmatizing language in clinical notes, and findings should be considered given this inherent measurement limitation. While all discrepancies were resolved through consensus to establish the final annotated dataset, some degree of uncertainty may remain. As such, the retrieved exemplars may reflect alignment with the study-specific definitions and exemplars rather than universally accepted definitions. Fifth, we used a model trained on general-domain data because it is designed for similarity-based retrieval tasks [<xref ref-type="bibr" rid="ref32">32</xref>], enabling its direct and efficient application in the current study without additional modifications. In addition, stigmatizing and positive language often requires capturing broader sociolinguistic expressions and context beyond clinical language, for which training on diverse data sources may be helpful. However, clinical language often includes specialized terminology, abbreviations, and domain-specific cues, which may not be fully captured by general-domain embedding models. As such, semantic similarity estimates may be less accurate for certain clinical expressions. Domain-specific models (eg, ClinicalBERT or BioClinicalBERT) are trained on clinical text and may provide better representations of clinical terminology and context [<xref ref-type="bibr" rid="ref46">46</xref>]. However, these models are not specifically designed or explicitly optimized for sentence-level similarity search or retrieval tasks. Thus, additional methodological adaptations, such as deriving sentence embeddings and optimizing the models for similarity-based retrieval tasks, would be required to apply domain-specific models to the task explored in the current study. For example, ClinicalBERT produces token-level contextual representations rather than sentence-level embeddings that can be directly compared using cosine similarity to identify semantically similar text. Token-level representations need to be transformed into sentence-level representations using a pooling strategy (eg, mean or max pooling), and different strategies may yield varying retrieval accuracy. Consequently, a meaningful empirical comparison with domain-specific models would be challenging because differences in retrieval accuracy could reflect these adaptations rather than the underlying models themselves. Therefore, rigorous comparison of clinical embedding models was considered beyond the scope of the current exploratory study. Finally, several language categories had very small sample sizes at higher cosine similarity thresholds, resulting in high but statistically unstable precision that should not be misinterpreted as strong or reliable performance.</p></sec><sec id="s4-3"><title>Future Research</title><p>Future research may consider evaluating recall and <italic>F</italic><sub>1</sub>-scores, which can provide additional insight into semantic similarity search performance, particularly given the potential trade-off between precision and recall [<xref ref-type="bibr" rid="ref47">47</xref>]. Exploring additional techniques to refine the candidate selection process may further contribute to optimizing accuracy. For instance, combining other similarity metrics, such as pragmatic similarity, could provide a more comprehensive assessment. Pragmatic similarity considers the intent and attitude behind the text, which could help to ensure that the intended meanings between texts are aligned [<xref ref-type="bibr" rid="ref48">48</xref>]. In addition, comparing the performance of general-domain and clinical-domain models in similarity-based retrieval tasks for annotation support could provide further insight into how domain-specific models affect retrieval accuracy and annotation efficiency. Future work is also needed to further clarify, refine, and standardize the operationalization of these language categories to improve annotation consistency and reproducibility. Finally, application of this approach across diverse clinical settings and datasets will further inform generalizability.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This study applies semantic similarity search in the context of stigmatizing language to support annotation workflows. Unlike previous research using uncertainty-based querying or synthetic data augmentation, we used human-annotated true exemplars as queries to directly retrieve additional exemplars from real-world clinical notes to support targeted expansion of context-dependent concepts. Findings demonstrate the potential of semantic similarity search to efficiently extract additional exemplars and increase the volume of training data while reducing annotation burden. By optimizing the cosine similarity thresholds, we may further achieve higher precision. These findings provide a foundation for further refinement and application of this approach in other domains requiring efficient ways to extract additional exemplars to augment NLP training data.</p></sec></sec></body><back><ack><p>We thank Arielle Hazi for her assistance during the revision process, including providing additional information on previously generated data and results. During the revision process, the first author used ChatGPT (GPT-5.3; OpenAI) to assist with proofreading tasks, specifically for identifying and correcting grammatical errors, under full human supervision. All outputs were reviewed and edited by the authors as needed, who take full responsibility for the content of the publication.</p></ack><notes><sec><title>Funding</title><p>Columbia University Data Science Institute Seed Funds and the Gordon and Betty Moore Foundation grant (GBMF9048) supported this project.</p></sec><sec><title>Data Availability</title><p>Clinical data used in this study are restricted by the institutional review board and cannot be publicly shared.</p></sec></notes><fn-group><fn fn-type="con"><p>Analysis: JKS, IIH, VB, AD, SH, DS</p><p>Conceptualization: M Topaz</p><p>Data curation: IIH, KC, M Tadiello</p><p>Funding acquisition: VB, KC, M Topaz</p><p>Methodology: AD, HM, M Topaz</p><p>Supervision: VB, M Topaz</p><p>Visualization: JKS</p><p>Writing &#x2013; original draft: JKS, IIH</p><p>Writing &#x2013; review &#x0026; editing: JKS, IIH, VB, AD, HM, SH, DS, KC, M Tadiello, M Topaz</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb2">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb3">IRB</term><def><p>institutional review board</p></def></def-item><def-item><term id="abb4">NLP</term><def><p>natural language processing</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Horan</surname><given-names>MR</given-names> </name><etal/></person-group><article-title>Natural language processing with machine learning methods to analyze unstructured patient-reported outcomes derived from electronic health records: a systematic review</article-title><source>Artif Intell Med</source><year>2023</year><month>12</month><volume>146</volume><fpage>102701</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2023.102701</pub-id><pub-id pub-id-type="medline">38042599</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Locke</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bashall</surname><given-names>A</given-names> </name><name name-style="western"><surname>Al-Adely</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moore</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kitchen</surname><given-names>GB</given-names> </name></person-group><article-title>Natural language processing in medicine: a review</article-title><source>Trends Anaesth Crit Care</source><year>2021</year><month>06</month><volume>38</volume><fpage>4</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1016/j.tacc.2021.02.007</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Franklin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name></person-group><article-title>Clinical text annotation - what factors are associated with the cost of time?</article-title><source>AMIA Annu Symp Proc</source><year>2018</year><volume>2018</volume><fpage>1552</fpage><lpage>1560</lpage><pub-id pub-id-type="medline">30815201</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcia</surname><given-names>EA</given-names> </name></person-group><article-title>Learning from imbalanced data</article-title><source>IEEE Trans Knowl Data Eng</source><year>2009</year><volume>21</volume><issue>9</issue><fpage>1263</fpage><lpage>1284</lpage><pub-id pub-id-type="doi">10.1109/TKDE.2008.239</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Krishnakumar</surname><given-names>A</given-names> </name></person-group><article-title>Active learning literature survey</article-title><year>2007</year><access-date>2026-08-26</access-date><publisher-name>University of California</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/228971426">https://www.researchgate.net/publication/228971426</ext-link></comment></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Settles</surname><given-names>B</given-names> </name></person-group><article-title>Active learning literature survey</article-title><year>2009</year><access-date>2026-08-20</access-date><publisher-name>University of Wisconsin&#x2013;Madison</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://minds.wisc.edu/server/api/core/bitstreams/8a78cf83-0702-4157-aeff-3cacb62f9ad5/content">https://minds.wisc.edu/server/api/core/bitstreams/8a78cf83-0702-4157-aeff-3cacb62f9ad5/content</ext-link></comment></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Mani</surname><given-names>S</given-names> </name></person-group><article-title>Active learning for unbalanced data in the challenge with multiple models and biasing</article-title><access-date>2026-08-20</access-date><conf-name>Active Learning and Experimental Design Workshop in Conjunction with AISTATS 2010</conf-name><conf-date>May 26, 2010</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v16/chen11a/chen11a.pdf">https://proceedings.mlr.press/v16/chen11a/chen11a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>K</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names></name><name name-style="western"><surname>Ng</surname><given-names>V</given-names></name><name name-style="western"><surname>Wan</surname><given-names>X</given-names></name></person-group><article-title>EDA: easy data augmentation techniques for boosting performance on text classification tasks</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>6382</fpage><lpage>6388</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1670</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bouamor</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pino</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name></person-group><article-title>Synthetic data generation with large language models for text classification: potential and limitations</article-title><source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>10443</fpage><lpage>10461</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.647</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>SY</given-names> </name><name name-style="western"><surname>Gangal</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Zong</surname><given-names>C</given-names></name><name name-style="western"><surname>Xia</surname><given-names>F</given-names></name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Navigli</surname><given-names>R</given-names></name></person-group><article-title>A survey of data augmentation approaches for NLP</article-title><source>Findings of the Association for Computational Linguistics</source><year>2021</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>968</fpage><lpage>988</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.findings-acl.84</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>&#x015E;ahin</surname><given-names>GG</given-names> </name></person-group><article-title>To augment or not to augment? A comparative study on text augmentation techniques for low-resource NLP</article-title><source>Computational Linguistics</source><year>2022</year><month>04</month><day>4</day><volume>48</volume><issue>1</issue><fpage>5</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1162/coli_a_00425</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scroggins</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Barcelona</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hulchafo</surname><given-names>II</given-names> </name><etal/></person-group><article-title>Assessing the quality and performance of synthetic data augmentation to identify stigmatizing language in obstetric clinical notes</article-title><source>Nurs Outlook</source><year>2026</year><volume>74</volume><issue>3</issue><fpage>102757</fpage><pub-id pub-id-type="doi">10.1016/j.outlook.2026.102757</pub-id><pub-id pub-id-type="medline">41962499</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Mikolov</surname><given-names>T</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Corrado</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>J</given-names> </name></person-group><article-title>Efficient estimation of word representations in vector space</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 16, 2013</comment><pub-id pub-id-type="doi">10.48550/arXiv.1301.3781</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Salton</surname><given-names>G</given-names> </name></person-group><source>Automatic Text Processing: The Transformation, Analysis, and Retrieval of Information by Computer</source><year>1989</year><publisher-name>Addison-Wesley</publisher-name><pub-id pub-id-type="other">978-0-201-12227-5</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>3982</fpage><lpage>3992</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mahajan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Poddar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>JJ</given-names> </name><etal/></person-group><article-title>Identification of semantically similar sentences in clinical notes: iterative intermediate training using multi-task learning</article-title><source>JMIR Med Inform</source><year>2020</year><month>11</month><day>27</day><volume>8</volume><issue>11</issue><fpage>e22508</fpage><pub-id pub-id-type="doi">10.2196/22508</pub-id><pub-id pub-id-type="medline">33245284</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ormerod</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mart&#x00ED;nez Del Rinc&#x00F3;n</surname><given-names>J</given-names> </name><name name-style="western"><surname>Devereux</surname><given-names>B</given-names> </name></person-group><article-title>Predicting semantic similarity between clinical sentence pairs using transformer models: evaluation and representational analysis</article-title><source>JMIR Med Inform</source><year>2021</year><month>05</month><day>26</day><volume>9</volume><issue>5</issue><fpage>e23099</fpage><pub-id pub-id-type="doi">10.2196/23099</pub-id><pub-id pub-id-type="medline">34037527</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lopez</surname><given-names>I</given-names> </name><name name-style="western"><surname>Swaminathan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vedula</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Clinical entity augmented retrieval for clinical information extraction</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>19</day><volume>8</volume><issue>1</issue><fpage>45</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01377-1</pub-id><pub-id pub-id-type="medline">39828800</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arzideh</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sch&#x00E4;fer</surname><given-names>H</given-names> </name><name name-style="western"><surname>Idrissi-Yaghir</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Improving retrieval augmented generation for health care by fine-tuning clinical embedding models: development and evaluation study</article-title><source>J Med Internet Res</source><year>2026</year><month>03</month><day>25</day><volume>28</volume><fpage>e82997</fpage><pub-id pub-id-type="doi">10.2196/82997</pub-id><pub-id pub-id-type="medline">41880603</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rouhizadeh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation in biomedicine: a survey of technologies, datasets, and clinical applications</article-title><source>Research Square</source><comment>Preprint posted online on  Dec 15, 2025</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-8330917/v1</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shattell</surname><given-names>MM</given-names> </name></person-group><article-title>Stigmatizing language with unintended meanings: &#x201C;persons with mental illness&#x201D; or &#x201C;mentally ill persons&#x201D;?</article-title><source>Issues Ment Health Nurs</source><year>2009</year><month>03</month><volume>30</volume><issue>3</issue><fpage>199</fpage><pub-id pub-id-type="doi">10.1080/01612840802694668</pub-id><pub-id pub-id-type="medline">19291498</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>M</given-names> </name><name name-style="western"><surname>Oliwa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Peek</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Tung</surname><given-names>EL</given-names> </name></person-group><article-title>Negative patient descriptors: documenting racial bias in the electronic health record</article-title><source>Health Aff (Millwood)</source><year>2022</year><month>02</month><volume>41</volume><issue>2</issue><fpage>203</fpage><lpage>211</lpage><pub-id pub-id-type="doi">10.1377/hlthaff.2021.01423</pub-id><pub-id pub-id-type="medline">35044842</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benkert</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cuevas</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Dove-Meadows</surname><given-names>E</given-names> </name><name name-style="western"><surname>Knuckles</surname><given-names>D</given-names> </name></person-group><article-title>Ubiquitous yet unclear: a systematic review of medical mistrust</article-title><source>Behav Med</source><year>2019</year><volume>45</volume><issue>2</issue><fpage>86</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1080/08964289.2019.1588220</pub-id><pub-id pub-id-type="medline">31343961</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Martin</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Hamilton</surname><given-names>BE</given-names> </name><name name-style="western"><surname>Osterman</surname><given-names>MJK</given-names> </name></person-group><article-title>Births in the United States, 2023</article-title><year>2022</year><access-date>2026-08-20</access-date><publisher-name>National Center for Health Statistics</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/nchs/data/databriefs/db507.pdf">https://www.cdc.gov/nchs/data/databriefs/db507.pdf</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Himmelstein</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bates</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name></person-group><article-title>Examination of stigmatizing language in the electronic health record</article-title><source>JAMA Netw Open</source><year>2022</year><month>01</month><day>4</day><volume>5</volume><issue>1</issue><fpage>e2144967</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.44967</pub-id><pub-id pub-id-type="medline">35084481</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barcelona</surname><given-names>V</given-names> </name><name name-style="western"><surname>Scharp</surname><given-names>D</given-names> </name><name name-style="western"><surname>Idnay</surname><given-names>BR</given-names> </name><etal/></person-group><article-title>A qualitative analysis of stigmatizing language in birth admission clinical notes</article-title><source>Nurs Inq</source><year>2023</year><month>07</month><volume>30</volume><issue>3</issue><fpage>37073504</fpage><pub-id pub-id-type="doi">10.1111/nin.12557</pub-id><pub-id pub-id-type="medline">37073504</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barcelona</surname><given-names>V</given-names> </name><name name-style="western"><surname>Horton</surname><given-names>RL</given-names> </name><name name-style="western"><surname>Rivlin</surname><given-names>K</given-names> </name><etal/></person-group><article-title>The power of language in hospital care for pregnant and birthing people: a vision for change</article-title><source>Obstet Gynecol</source><year>2023</year><month>10</month><day>1</day><volume>142</volume><issue>4</issue><fpage>795</fpage><lpage>803</lpage><pub-id pub-id-type="doi">10.1097/AOG.0000000000005333</pub-id><pub-id pub-id-type="medline">37678895</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barcelona</surname><given-names>V</given-names> </name><name name-style="western"><surname>Scharp</surname><given-names>D</given-names> </name><name name-style="western"><surname>Moen</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Using natural language processing to identify stigmatizing language in labor and birth clinical notes</article-title><source>Matern Child Health J</source><year>2024</year><month>03</month><volume>28</volume><issue>3</issue><fpage>578</fpage><lpage>586</lpage><pub-id pub-id-type="doi">10.1007/s10995-023-03857-4</pub-id><pub-id pub-id-type="medline">38147277</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barcelona</surname><given-names>V</given-names> </name><name name-style="western"><surname>Scroggins</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Scharp</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Secondary qualitative analysis of stigmatizing and nonstigmatizing language used in hospital birth settings</article-title><source>J Obstet Gynecol Neonatal Nurs</source><year>2025</year><month>01</month><volume>54</volume><issue>1</issue><fpage>112</fpage><lpage>122</lpage><pub-id pub-id-type="doi">10.1016/j.jogn.2024.10.003</pub-id><pub-id pub-id-type="medline">39577837</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Saha</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chee</surname><given-names>B</given-names> </name><name name-style="western"><surname>Taylor</surname><given-names>J</given-names> </name><name name-style="western"><surname>Beach</surname><given-names>MC</given-names> </name></person-group><article-title>Physician use of stigmatizing language in patient medical records</article-title><source>JAMA Netw Open</source><year>2021</year><month>07</month><day>1</day><volume>4</volume><issue>7</issue><fpage>e2117052</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.17052</pub-id><pub-id pub-id-type="medline">34259849</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scroggins</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Hulchafo</surname><given-names>II</given-names> </name><name name-style="western"><surname>Harkins</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Identifying stigmatizing and positive/preferred language in obstetric clinical notes using natural language processing</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>02</month><day>1</day><volume>32</volume><issue>2</issue><fpage>308</fpage><lpage>317</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae290</pub-id><pub-id pub-id-type="medline">39569431</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="web"><article-title>Sentence-transformers/multi-qa-distilbert-cos-v1</article-title><source>Hugging Face</source><year>2024</year><access-date>2026-08-20</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/sentence-transformers/multi-qa-distilbert-cos-v1">https://huggingface.co/sentence-transformers/multi-qa-distilbert-cos-v1</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Excoffier</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Roehr</surname><given-names>T</given-names> </name><name name-style="western"><surname>Figueroa</surname><given-names>A</given-names> </name><name name-style="western"><surname>Papaioannou</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Bressem</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ortala</surname><given-names>M</given-names> </name></person-group><article-title>Generalist embedding models are better at short-context clinical semantic search than specialized embedding models</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 3, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.01943</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Douze</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jegou</surname><given-names>H</given-names> </name></person-group><article-title>Billion-scale similarity search with GPUs</article-title><source>IEEE Trans Big Data</source><year>2021</year><volume>7</volume><issue>3</issue><fpage>535</fpage><lpage>547</lpage><pub-id pub-id-type="doi">10.1109/TBDATA.2019.2921572</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Manning</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Raghavan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sch&#x00FC;tze</surname><given-names>H</given-names> </name></person-group><source>Introduction to Information Retrieval</source><year>2009</year><access-date>2026-08-20</access-date><publisher-name>Cambridge University Press</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://nlp.stanford.edu/IR-book/pdf/irbookprint.pdf">https://nlp.stanford.edu/IR-book/pdf/irbookprint.pdf</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lingren</surname><given-names>T</given-names> </name><name name-style="western"><surname>Deleger</surname><given-names>L</given-names> </name><name name-style="western"><surname>Molnar</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Evaluating the impact of pre-annotation on annotation speed and potential bias: natural language processing gold standard development for clinical named entity recognition in clinical trial announcements</article-title><source>J Am Med Inform Assoc</source><year>2014</year><volume>21</volume><issue>3</issue><fpage>406</fpage><lpage>413</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2013-001837</pub-id><pub-id pub-id-type="medline">24001514</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>South</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Mowery</surname><given-names>D</given-names> </name><name name-style="western"><surname>Suo</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Evaluating the effects of machine pre-annotation and an interactive annotation interface on manual de-identification of clinical text</article-title><source>J Biomed Inform</source><year>2014</year><month>08</month><volume>50</volume><fpage>162</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2014.05.002</pub-id><pub-id pub-id-type="medline">24859155</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pangakis</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wolken</surname><given-names>S</given-names> </name></person-group><article-title>Keeping humans in the loop: human-centered automated annotation with generative AI</article-title><source>ICWSM</source><year>2025</year><volume>19</volume><fpage>1471</fpage><lpage>1492</lpage><pub-id pub-id-type="doi">10.1609/icwsm.v19i1.35883</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Pennington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Socher</surname><given-names>R</given-names> </name><name name-style="western"><surname>Manning</surname><given-names>C</given-names> </name></person-group><article-title>Glove: global vectors for word representation</article-title><conf-name>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)</conf-name><conf-date>Oct 25-29, 2014</conf-date><pub-id pub-id-type="doi">10.3115/v1/D14-1162</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Devlin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Toutanova</surname><given-names>K</given-names> </name></person-group><article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 11, 2018</comment><pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Arora</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>T</given-names> </name></person-group><article-title>A simple but tough-to-beat baseline for sentence embeddings</article-title><access-date>2026-08-20</access-date><conf-name>5th International Conference on Learning Representations (ICLR 2017)</conf-name><conf-date>Apr 24-26, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=SyK00v5xx">https://openreview.net/pdf?id=SyK00v5xx</ext-link></comment></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Ethayarajh</surname><given-names>K</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>How contextual are contextualized word representations? Comparing the geometry of BERT, ELMo, and GPT-2 embeddings</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>55</fpage><lpage>65</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1006</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kholghi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sitbon</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zuccon</surname><given-names>G</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>A</given-names> </name></person-group><article-title>Active learning reduces annotation time for clinical concept extraction</article-title><source>Int J Med Inform</source><year>2017</year><month>10</month><volume>106</volume><fpage>25</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2017.08.001</pub-id><pub-id pub-id-type="medline">28870380</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Salimi</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Cost-aware active learning for named entity recognition in clinical text</article-title><source>J Am Med Inform Assoc</source><year>2019</year><month>11</month><day>1</day><volume>26</volume><issue>11</issue><fpage>1314</fpage><lpage>1322</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocz102</pub-id><pub-id pub-id-type="medline">31294792</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nachtegael</surname><given-names>C</given-names> </name><name name-style="western"><surname>De Stefani</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lenaerts</surname><given-names>T</given-names> </name></person-group><article-title>A study of deep active learning methods to reduce labelling efforts in biomedical relation extraction</article-title><source>PLoS ONE</source><year>2023</year><volume>18</volume><issue>12</issue><fpage>e0292356</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0292356</pub-id><pub-id pub-id-type="medline">38100453</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boag</surname><given-names>W</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Rumshisky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name></person-group><article-title>Publicly available clinical</article-title><source>Proceedings of the 2nd Clinical Natural Language Processing Workshop</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>72</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sokolova</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lapalme</surname><given-names>G</given-names> </name></person-group><article-title>A systematic analysis of performance measures for classification tasks</article-title><source>Inf Process Manag</source><year>2009</year><month>07</month><volume>45</volume><issue>4</issue><fpage>427</fpage><lpage>437</lpage><pub-id pub-id-type="doi">10.1016/j.ipm.2009.03.002</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Ward</surname><given-names>N</given-names> </name><name name-style="western"><surname>Marco</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Calzolari</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kan</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Hoste</surname><given-names>V</given-names> </name><name name-style="western"><surname>Lenci</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sakti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>N</given-names> </name></person-group><article-title>A collection of pragmatic-similarity judgments over spoken dialog utterances</article-title><source>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING)</source><publisher-name>ELRA and ICCL</publisher-name><fpage>154</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.63317/5iqq22vfx8n2</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary tables presenting detailed category-specific results, synthetic note snippets, and the annotation codebook.</p><media xlink:href="medinform_v14i1e85088_app1.docx" xlink:title="DOCX File, 43 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Pseudocode for semantic similarity search pipeline.</p><media xlink:href="medinform_v14i1e85088_app2.docx" xlink:title="DOCX File, 30 KB"/></supplementary-material></app-group></back></article>